Fix multi-GPU with lazy CuPy init + rendering improvements
GPU fix: - Revert to CUDA_VISIBLE_DEVICES approach but with lazy CuPy init - gpu.py: CuPy is no longer imported at module level; _init_gpu() imports it lazily on first to_gpu() call. This allows workers to set CUDA_VISIBLE_DEVICES before CuPy creates a CUDA context. - gpu.py: detect GPU count via nvidia-smi (no CUDA context needed) - pipeline.py: each worker sets CUDA_VISIBLE_DEVICES=N before CuPy init, so each process uses only its assigned GPU Rendering improvements: - Title: split into bold title (14pt) + italic description (10pt) instead of single 15pt bold block - North arrow: moved inside data area (top-right corner) with semi-transparent white background for readability over data - Colorbar: full height (no gap for compass rose), added ScalarFormatter(useOffset=False) to avoid scientific notation - Colorbar compass rose gap removed since north arrow is now inside the data area
This commit is contained in:
@ -7,8 +7,9 @@ operations fall back to numpy/scipy on CPU.
|
||||
GPU errors (e.g. in forked subprocesses) are caught gracefully and
|
||||
cause an automatic fallback to CPU for the current operation.
|
||||
|
||||
Multi-GPU support: when multiple GPUs are available, each worker process
|
||||
can be assigned a different GPU via set_active_gpu() for balanced load.
|
||||
Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before
|
||||
CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel
|
||||
cache incompatibilities that occur with Device.use() switching.
|
||||
"""
|
||||
|
||||
import logging
|
||||
@ -18,16 +19,13 @@ from scipy import ndimage
|
||||
|
||||
logger = logging.getLogger("lidar")
|
||||
|
||||
# Detect GPU count at import time WITHOUT importing CuPy.
|
||||
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
|
||||
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
|
||||
# in worker processes.
|
||||
# Detect total GPU count via nvidia-smi (no CUDA context created).
|
||||
# This must happen before any CUDA_VISIBLE_DEVICES manipulation.
|
||||
_NUM_GPUS = 0
|
||||
HAS_GPU = False
|
||||
_gpu_name = None
|
||||
_gpu_mem_gb = 0
|
||||
|
||||
# Check if GPUs are available via nvidia-smi (no CUDA context created)
|
||||
try:
|
||||
import subprocess
|
||||
_result = subprocess.run(
|
||||
@ -49,9 +47,9 @@ try:
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||
pass
|
||||
|
||||
# Lazy-initialized GPU module references
|
||||
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
|
||||
# to be set before CuPy context creation in worker processes.
|
||||
# Lazy CuPy initialization — imported only when first needed.
|
||||
# This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates
|
||||
# a CUDA context, enabling per-process GPU assignment.
|
||||
_xp = np # Default: CPU
|
||||
_cp = None # cupy module (or None)
|
||||
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
|
||||
@ -61,8 +59,8 @@ _gpu_initialized = False
|
||||
def _init_gpu():
|
||||
"""Lazily initialize CuPy on first GPU use.
|
||||
|
||||
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
|
||||
before CuPy creates a CUDA context.
|
||||
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
|
||||
set before the CUDA context is created.
|
||||
"""
|
||||
global _xp, _cp, _cp_ndimage, _gpu_initialized
|
||||
if _gpu_initialized:
|
||||
@ -76,39 +74,36 @@ def _init_gpu():
|
||||
_xp = _real_cupy
|
||||
_cp = _real_cupy
|
||||
_cp_ndimage = _real_cupy_ndimage
|
||||
except (ImportError, Exception):
|
||||
except (ImportError, Exception) as e:
|
||||
logger.debug(f"CuPy non disponible: {e}")
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
|
||||
|
||||
def num_gpus():
|
||||
"""Return the number of available CUDA GPUs."""
|
||||
"""Return the total number of CUDA GPUs in the system."""
|
||||
return _NUM_GPUS
|
||||
|
||||
|
||||
def set_active_gpu(gpu_id):
|
||||
"""Set the active GPU for the current process.
|
||||
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES.
|
||||
|
||||
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
|
||||
the CUDA context is created on the correct device.
|
||||
MUST be called before any GPU operation (to_gpu, etc.) to ensure
|
||||
CuPy creates its CUDA context on the correct device. With lazy
|
||||
initialization, CuPy is imported AFTER this call, so it only
|
||||
sees the assigned GPU.
|
||||
|
||||
Args:
|
||||
gpu_id: 0-based GPU index. Clamped to valid range.
|
||||
gpu_id: 0-based GPU index (referring to the system GPU numbering).
|
||||
"""
|
||||
if not HAS_GPU or _NUM_GPUS <= 1:
|
||||
return # Nothing to do for single GPU or no GPU
|
||||
|
||||
gpu_id = gpu_id % _NUM_GPUS
|
||||
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
|
||||
# This is the most reliable way in spawn processes
|
||||
|
||||
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
|
||||
# Reset lazy init so CuPy re-detects with the new env
|
||||
global _gpu_initialized, _cp, _cp_ndimage, _xp
|
||||
_gpu_initialized = False
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
_xp = np
|
||||
|
||||
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
|
||||
|
||||
@ -127,8 +122,17 @@ def _gpu_available():
|
||||
|
||||
def log_gpu_status():
|
||||
"""Log GPU detection result. Called after logging is configured."""
|
||||
if HAS_GPU:
|
||||
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||
if _gpu_available():
|
||||
# Get actual device name from CuPy (after init)
|
||||
try:
|
||||
dev = _cp.cuda.Device()
|
||||
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
|
||||
if isinstance(name, bytes):
|
||||
name = name.decode()
|
||||
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3)
|
||||
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)"
|
||||
except Exception:
|
||||
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||
if _NUM_GPUS > 1:
|
||||
gpu_info += f" × {_NUM_GPUS}"
|
||||
logger.info(gpu_info)
|
||||
|
||||
Reference in New Issue
Block a user