Fix multi-GPU with lazy CuPy init + rendering improvements

GPU fix:
- Revert to CUDA_VISIBLE_DEVICES approach but with lazy CuPy init
- gpu.py: CuPy is no longer imported at module level; _init_gpu()
  imports it lazily on first to_gpu() call. This allows workers to
  set CUDA_VISIBLE_DEVICES before CuPy creates a CUDA context.
- gpu.py: detect GPU count via nvidia-smi (no CUDA context needed)
- pipeline.py: each worker sets CUDA_VISIBLE_DEVICES=N before CuPy
  init, so each process uses only its assigned GPU

Rendering improvements:
- Title: split into bold title (14pt) + italic description (10pt)
  instead of single 15pt bold block
- North arrow: moved inside data area (top-right corner) with
  semi-transparent white background for readability over data
- Colorbar: full height (no gap for compass rose), added
  ScalarFormatter(useOffset=False) to avoid scientific notation
- Colorbar compass rose gap removed since north arrow is now
  inside the data area
This commit is contained in:
Antoine Jacquin
2026-05-15 12:24:57 +02:00
parent b4a0e384c9
commit a3f7b44874
3 changed files with 58 additions and 48 deletions

View File

@ -7,8 +7,9 @@ operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: when multiple GPUs are available, each worker process
can be assigned a different GPU via set_active_gpu() for balanced load.
Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before
CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel
cache incompatibilities that occur with Device.use() switching.
"""
import logging
@ -18,16 +19,13 @@ from scipy import ndimage
logger = logging.getLogger("lidar")
# Detect GPU count at import time WITHOUT importing CuPy.
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
# in worker processes.
# Detect total GPU count via nvidia-smi (no CUDA context created).
# This must happen before any CUDA_VISIBLE_DEVICES manipulation.
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
# Check if GPUs are available via nvidia-smi (no CUDA context created)
try:
import subprocess
_result = subprocess.run(
@ -49,9 +47,9 @@ try:
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy-initialized GPU module references
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
# to be set before CuPy context creation in worker processes.
# Lazy CuPy initialization — imported only when first needed.
# This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates
# a CUDA context, enabling per-process GPU assignment.
_xp = np # Default: CPU
_cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
@ -61,8 +59,8 @@ _gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use.
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
before CuPy creates a CUDA context.
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
set before the CUDA context is created.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized
if _gpu_initialized:
@ -76,39 +74,36 @@ def _init_gpu():
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
except (ImportError, Exception):
except (ImportError, Exception) as e:
logger.debug(f"CuPy non disponible: {e}")
_xp = np
_cp = None
_cp_ndimage = None
def num_gpus():
"""Return the number of available CUDA GPUs."""
"""Return the total number of CUDA GPUs in the system."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process.
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES.
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
the CUDA context is created on the correct device.
MUST be called before any GPU operation (to_gpu, etc.) to ensure
CuPy creates its CUDA context on the correct device. With lazy
initialization, CuPy is imported AFTER this call, so it only
sees the assigned GPU.
Args:
gpu_id: 0-based GPU index. Clamped to valid range.
gpu_id: 0-based GPU index (referring to the system GPU numbering).
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
# This is the most reliable way in spawn processes
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
# Reset lazy init so CuPy re-detects with the new env
global _gpu_initialized, _cp, _cp_ndimage, _xp
_gpu_initialized = False
_cp = None
_cp_ndimage = None
_xp = np
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
@ -127,8 +122,17 @@ def _gpu_available():
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if HAS_GPU:
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _gpu_available():
# Get actual device name from CuPy (after init)
try:
dev = _cp.cuda.Device()
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)"
except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)