"""GPU acceleration helpers for LiDAR pipeline. Auto-selects the best NVIDIA GPU that works with CuPy. Tries each card in order of compute capability; falls back to CPU if none work. All workers share the selected GPU. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # --------------------------------------------------------------------------- # GPU auto-detection via nvidia-smi (no CUDA context created) # --------------------------------------------------------------------------- _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 _best_gpu_id: int | None = None _gpu_reason = None def _pick_gpu() -> int | None: """Pick the best GPU that CuPy can actually use. Tries each card in order of compute capability (highest first). Returns the index of the first GPU whose compute capability is known to work with the installed CuPy version. """ try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode != 0: return None global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason gpus = [] for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) < 4: continue idx = int(parts[0]) name = parts[1] cap_str = parts[2] mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) score = major * 1000 + minor * 100 + mem_mi gpus.append((idx, name, cap_str, mem_mi, score, major)) if not gpus: return None _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) # CuPy 13.4 + CUDA 11.8 JIT supports up to sm_89 (RTX 40xx). # sm_120 (RTX 50xx) is not supported yet — skip it. for gpu in gpus: idx, name, cap_str, mem_mi, score, major = gpu if major <= 8: # sm_89 and below — works with CuPy 13.4 + JIT _best_gpu_id = idx _gpu_name = name _gpu_mem_gb = mem_mi // 1024 HAS_GPU = True return _best_gpu_id # All GPUs have unsupported compute capability _gpu_reason = "aucun GPU avec compute capability compatible CuPy 13.4" return None except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return None _best_gpu_id = _pick_gpu() # --------------------------------------------------------------------------- # Lazy CuPy initialization # --------------------------------------------------------------------------- _xp = np _cp = None _cp_ndimage = None _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use.""" global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU if _gpu_initialized: return _gpu_initialized = True if not HAS_GPU or _best_gpu_id is None: _xp = np _cp = None _cp_ndimage = None HAS_GPU = False if _gpu_reason: logger.info(f"Pas de GPU — {_gpu_reason}") return try: os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage # Warm-up: verify kernel execution works _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) _result = _real_cupy.sum(_test * _test) _ = _result.get() del _test, _result _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage # Memory pool management removed — CuPy 13.x uses its own allocator. # OOM protection handled at the application level (gpu_cleanup() calls). except (ImportError, Exception) as e: logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}") _xp = np _cp = None _cp_ndimage = None HAS_GPU = False # --------------------------------------------------------------------------- # Public API # --------------------------------------------------------------------------- def num_gpus(): """Return 1 if GPU is active, 0 otherwise.""" return 1 if HAS_GPU else 0 def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """No-op — GPU is auto-selected at import time.""" pass def set_active_gpu(gpu_id): """No-op — GPU is auto-selected at import time.""" pass def _gpu_available(): """Check if GPU is usable right now.""" if not HAS_GPU: return False try: _init_gpu() return _cp is not None except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: name = _cp.cuda.runtime.getDeviceProperties(0)['name'] if isinstance(name, bytes): name = name.decode() mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) # Warn about unsupported GPUs that exist but are not used try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode == 0: for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) >= 3: idx = int(parts[0]) if idx != _best_gpu_id: cap = parts[2] logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — non utilisé (incompatible CuPy)") except Exception: pass else: logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})") # --------------------------------------------------------------------------- # Array transfer # --------------------------------------------------------------------------- def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy.""" if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception: pass return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp.asnumpy(arr) except Exception: pass return arr # --------------------------------------------------------------------------- # Filters — GPU if array is on GPU, CPU otherwise # --------------------------------------------------------------------------- def xp_gaussian_filter(arr, sigma): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception: arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception: arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) # --------------------------------------------------------------------------- # Misc # --------------------------------------------------------------------------- def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass def disable_gpu(): """Disable GPU acceleration for the rest of this process.""" global HAS_GPU, _xp, _cp, _cp_ndimage if not HAS_GPU: return logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") HAS_GPU = False _xp = np _cp = None _cp_ndimage = None def is_gpu_active(): """Check if GPU acceleration is currently active.""" return HAS_GPU def safe_gpu_call(func, *args, **kwargs): """Call a function with GPU arrays, retrying on CPU if GPU fails.""" try: return func(*args, **kwargs) except Exception as e: err_msg = str(e) if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg): logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...") disable_gpu() return func(*args, **kwargs) raise