"""GPU acceleration helpers for LiDAR pipeline. Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU is detected, array operations are accelerated on the GPU. Otherwise, all operations fall back to numpy/scipy on CPU. GPU errors (e.g. in forked subprocesses) are caught gracefully and cause an automatic fallback to CPU for the current operation. Multi-GPU support: when multiple GPUs are available, each worker process can be assigned a different GPU via set_active_gpu() for balanced load. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # Detect GPU count at import time WITHOUT importing CuPy. # We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs, # so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation # in worker processes. _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 # Check if GPUs are available via nvidia-smi (no CUDA context created) try: import subprocess _result = subprocess.run( ['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5 ) if _result.returncode == 0: _lines = _result.stdout.strip().split('\n') _NUM_GPUS = len(_lines) # Parse first GPU info for logging _parts = _lines[0].split(',') if len(_parts) >= 3: _gpu_name = _parts[1].strip() try: _gpu_mem_gb = int(float(_parts[2].strip())) // 1024 except (ValueError, IndexError): pass HAS_GPU = True except (FileNotFoundError, subprocess.TimeoutExpired, Exception): pass # Lazy-initialized GPU module references # CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES # to be set before CuPy context creation in worker processes. _xp = np # Default: CPU _cp = None # cupy module (or None) _cp_ndimage = None # cupyx.scipy.ndimage (or None) _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use. This allows CUDA_VISIBLE_DEVICES to take effect in worker processes before CuPy creates a CUDA context. """ global _xp, _cp, _cp_ndimage, _gpu_initialized if _gpu_initialized: return _gpu_initialized = True try: import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage # Verify GPU is actually accessible _real_cupy.cuda.runtime.getDevice() _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage except (ImportError, Exception): _xp = np _cp = None _cp_ndimage = None def num_gpus(): """Return the number of available CUDA GPUs.""" return _NUM_GPUS def set_active_gpu(gpu_id): """Set the active GPU for the current process. Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure the CUDA context is created on the correct device. Args: gpu_id: 0-based GPU index. Clamped to valid range. """ if not HAS_GPU or _NUM_GPUS <= 1: return # Nothing to do for single GPU or no GPU gpu_id = gpu_id % _NUM_GPUS # Set CUDA_VISIBLE_DEVICES before CuPy context is created # This is the most reliable way in spawn processes os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id) # Reset lazy init so CuPy re-detects with the new env global _gpu_initialized, _cp, _cp_ndimage, _xp _gpu_initialized = False _cp = None _cp_ndimage = None _xp = np logger.info(f" GPU {gpu_id} sélectionnée pour ce worker") def _gpu_available(): """Check if GPU is usable right now (may fail in forked subprocesses).""" if not HAS_GPU: return False try: _init_gpu() _cp.cuda.runtime.getDevice() return True except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if HAS_GPU: gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" if _NUM_GPUS > 1: gpu_info += f" × {_NUM_GPUS}" logger.info(gpu_info) else: logger.info("Pas de GPU — mode CPU uniquement") def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy. Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU is unavailable (e.g. in forked subprocess). """ if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception: pass # Fall back to CPU return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp.asnumpy(arr) except Exception: pass # Already on CPU or GPU error return arr def xp_gaussian_filter(arr, sigma): """Gaussian filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception: arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): """Uniform filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception: arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): """Minimum filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): """Maximum filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass