"""GPU acceleration helpers for LiDAR pipeline. Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back to CPU if no GPU is available or usable. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # --------------------------------------------------------------------------- # GPU auto-detection via nvidia-smi (no CUDA context created) # --------------------------------------------------------------------------- _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 _best_gpu_id: int | None = None def _pick_gpu() -> int | None: """Pick the best GPU from the system. Preference order: 1. RTX 50xx (Blackwell, compute >= 12.0) 2. RTX 40xx (Ada Lovelace, compute >= 8.9) 3. Any NVIDIA GPU with highest compute capability """ try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode != 0: return None global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id gpus = [] for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) < 4: continue idx = int(parts[0]) name = parts[1] cap_str = parts[2] mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) # Score: higher compute capability first, then more VRAM score = major * 1000 + minor * 100 + mem_mi gpus.append((idx, name, cap_str, mem_mi, score)) if not gpus: return None _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) best = gpus[0] _best_gpu_id = best[0] _gpu_name = best[1] _gpu_mem_gb = best[3] // 1024 HAS_GPU = True return _best_gpu_id except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return None _best_gpu_id = _pick_gpu() # --------------------------------------------------------------------------- # Lazy CuPy initialization # --------------------------------------------------------------------------- _xp = np _cp = None _cp_ndimage = None _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use.""" global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU if _gpu_initialized: return _gpu_initialized = True if not HAS_GPU or _best_gpu_id is None: _xp = np _cp = None _cp_ndimage = None HAS_GPU = False return try: # Restrict to the selected GPU before CuPy imports os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage # Warm-up: verify kernel execution works on this GPU _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) _result = _real_cupy.sum(_test * _test) _ = _result.get() del _test, _result _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage props = _real_cupy.cuda.runtime.getDeviceProperties(0) total_mem = props['totalGlobalMem'] # Memory pool: up to 90% of VRAM (single GPU shared by workers) pool_size = int(total_mem * 0.9) _real_cupy.cuda.set_memory_pool(0, pool_size) except (ImportError, Exception) as e: logger.warning(f"GPU non disponible — mode CPU: {e}") _xp = np _cp = None _cp_ndimage = None HAS_GPU = False # --------------------------------------------------------------------------- # Public API # --------------------------------------------------------------------------- def num_gpus(): """Return 1 if GPU is active, 0 otherwise.""" return 1 if HAS_GPU else 0 def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """No-op — GPU is auto-selected at import time.""" pass def set_active_gpu(gpu_id): """No-op — GPU is auto-selected at import time.""" pass def _gpu_available(): """Check if GPU is usable right now.""" if not HAS_GPU: return False try: _init_gpu() return _cp is not None except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: name = _cp.cuda.runtime.getDeviceProperties(0)['name'] if isinstance(name, bytes): name = name.decode() mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) else: logger.info("Pas de GPU — mode CPU uniquement") # --------------------------------------------------------------------------- # Array transfer # --------------------------------------------------------------------------- def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy.""" if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception: pass return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp.asnumpy(arr) except Exception: pass return arr # --------------------------------------------------------------------------- # Filters — GPU if array is on GPU, CPU otherwise # --------------------------------------------------------------------------- def xp_gaussian_filter(arr, sigma): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception: arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception: arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) # --------------------------------------------------------------------------- # Misc # --------------------------------------------------------------------------- def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass def disable_gpu(): """Disable GPU acceleration for the rest of this process.""" global HAS_GPU, _xp, _cp, _cp_ndimage if not HAS_GPU: return logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") HAS_GPU = False _xp = np _cp = None _cp_ndimage = None def is_gpu_active(): """Check if GPU acceleration is currently active.""" return HAS_GPU def safe_gpu_call(func, *args, **kwargs): """Call a function with GPU arrays, retrying on CPU if GPU fails.""" try: return func(*args, **kwargs) except Exception as e: err_msg = str(e) if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg): logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...") disable_gpu() return func(*args, **kwargs) raise