"""GPU acceleration helpers for LiDAR pipeline. Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU is detected, array operations are accelerated on the GPU. Otherwise, all operations fall back to numpy/scipy on CPU. GPU errors (e.g. in forked subprocesses) are caught gracefully and cause an automatic fallback to CPU for the current operation. Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel cache incompatibilities that occur with Device.use() switching. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # Detect total GPU count via nvidia-smi (no CUDA context created). # This must happen before any CUDA_VISIBLE_DEVICES manipulation. _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 try: import subprocess _result = subprocess.run( ['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5 ) if _result.returncode == 0: _lines = _result.stdout.strip().split('\n') _NUM_GPUS = len(_lines) # Parse first GPU info for logging _parts = _lines[0].split(',') if len(_parts) >= 3: _gpu_name = _parts[1].strip() try: _gpu_mem_gb = int(float(_parts[2].strip())) // 1024 except (ValueError, IndexError): pass HAS_GPU = True except (FileNotFoundError, subprocess.TimeoutExpired, Exception): pass # Lazy CuPy initialization — imported only when first needed. # This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates # a CUDA context, enabling per-process GPU assignment. _xp = np # Default: CPU _cp = None # cupy module (or None) _cp_ndimage = None # cupyx.scipy.ndimage (or None) _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use. Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be set before the CUDA context is created. """ global _xp, _cp, _cp_ndimage, _gpu_initialized if _gpu_initialized: return _gpu_initialized = True try: import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage # Verify GPU is actually accessible _real_cupy.cuda.runtime.getDevice() _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage except (ImportError, Exception) as e: logger.debug(f"CuPy non disponible: {e}") _xp = np _cp = None _cp_ndimage = None def num_gpus(): """Return the total number of CUDA GPUs in the system.""" return _NUM_GPUS def set_active_gpu(gpu_id): """Set the active GPU for the current process via CUDA_VISIBLE_DEVICES. MUST be called before any GPU operation (to_gpu, etc.) to ensure CuPy creates its CUDA context on the correct device. With lazy initialization, CuPy is imported AFTER this call, so it only sees the assigned GPU. Args: gpu_id: 0-based GPU index (referring to the system GPU numbering). """ if not HAS_GPU or _NUM_GPUS <= 1: return # Nothing to do for single GPU or no GPU gpu_id = gpu_id % _NUM_GPUS # Set CUDA_VISIBLE_DEVICES before CuPy context creation os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id) logger.info(f" GPU {gpu_id} sélectionnée pour ce worker") def _gpu_available(): """Check if GPU is usable right now (may fail in forked subprocesses).""" if not HAS_GPU: return False try: _init_gpu() _cp.cuda.runtime.getDevice() return True except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): # Get actual device name from CuPy (after init) try: dev = _cp.cuda.Device() name = _cp.cuda.runtime.getDeviceProperties(0)['name'] if isinstance(name, bytes): name = name.decode() mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" if _NUM_GPUS > 1: gpu_info += f" × {_NUM_GPUS}" logger.info(gpu_info) else: logger.info("Pas de GPU — mode CPU uniquement") def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy. Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU is unavailable (e.g. in forked subprocess). """ if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception: pass # Fall back to CPU return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp.asnumpy(arr) except Exception: pass # Already on CPU or GPU error return arr def xp_gaussian_filter(arr, sigma): """Gaussian filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception: arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): """Uniform filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception: arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): """Minimum filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): """Maximum filter — uses GPU if array is on GPU, CPU otherwise.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass