"""GPU acceleration helpers for LiDAR pipeline. Auto-selects the best NVIDIA GPU (highest compute capability first). Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation works correctly for any architecture (sm_89, sm_120, etc.). All workers share the selected GPU. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # --------------------------------------------------------------------------- # GPU auto-detection via nvidia-smi (no CUDA context created) # --------------------------------------------------------------------------- _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 _best_gpu_id: int | None = None _gpu_reason = None def _pick_gpu() -> list: """List all GPUs from the system, sorted by compute capability (highest first).""" try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode != 0: return [] global _NUM_GPUS gpus = [] for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) < 4: continue idx = int(parts[0]) name = parts[1] cap_str = parts[2] mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) score = major * 1000 + minor * 100 + mem_mi gpus.append((idx, name, cap_str, mem_mi, score, major)) _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) return gpus except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return [] _candidate_gpus: list = [] try: _candidate_gpus = _pick_gpu() or [] except Exception: pass # --------------------------------------------------------------------------- # Lazy CuPy initialization — tries each GPU until one works # --------------------------------------------------------------------------- _xp = np _cp = None _cp_ndimage = None _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use. Uses a subprocess to test each GPU (highest compute capability first). The subprocess sets CUDA_VISIBLE_DEVICES before importing CuPy and runs a warm-up kernel. First GPU that passes wins. This is necessary because CUDA_VISIBLE_DEVICES must be set in the process environment BEFORE CuPy imports, not via os.environ in Python. """ global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb if _gpu_initialized: return _gpu_initialized = True if not _candidate_gpus: _xp = np _cp = None _cp_ndimage = None HAS_GPU = False return # Test each GPU in a subprocess import subprocess _working_gpu = None for idx, name, cap_str, mem_mi, score, major in _candidate_gpus: result = subprocess.run( ['python3', '-c', 'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); print(cupy.sum(a).get())'], capture_output=True, text=True, timeout=120, env={**os.environ, 'CUDA_VISIBLE_DEVICES': str(idx)}, ) if result.returncode == 0 and '3.0' in result.stdout: _working_gpu = (idx, name, mem_mi) break logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) non compatible: " f"{result.stderr.strip().splitlines()[-1] if result.stderr else 'inconnue'}") if _working_gpu is None: logger.info("Pas de GPU utilisable — mode CPU uniquement") _xp = np _cp = None _cp_ndimage = None HAS_GPU = False return idx, name, mem_mi = _working_gpu os.environ['CUDA_VISIBLE_DEVICES'] = str(idx) import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage _best_gpu_id = idx _gpu_name = name _gpu_mem_gb = mem_mi // 1024 HAS_GPU = True _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage # --------------------------------------------------------------------------- # Public API # --------------------------------------------------------------------------- def num_gpus(): """Return 1 if GPU is active, 0 otherwise.""" return 1 if HAS_GPU else 0 def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """No-op — GPU is auto-selected at import time.""" pass def set_active_gpu(gpu_id): """No-op — GPU is auto-selected at import time.""" pass def _gpu_available(): """Check if GPU is usable right now.""" if not HAS_GPU: return False try: _init_gpu() return _cp is not None except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: with _cp.cuda.Device(_best_gpu_id): name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name'] if isinstance(name, bytes): name = name.decode() mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) # List other GPUs for info try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode == 0: for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) >= 3: idx = int(parts[0]) if idx != _best_gpu_id: cap = parts[2] logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — disponible") except Exception: pass else: logger.info(f"Pas de GPU utilisable — mode CPU uniquement") # --------------------------------------------------------------------------- # Array transfer # --------------------------------------------------------------------------- def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy.""" if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception: pass return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU.""" if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp.asnumpy(arr) except Exception: pass return arr # --------------------------------------------------------------------------- # Filters — GPU if array is on GPU, CPU otherwise # --------------------------------------------------------------------------- def xp_gaussian_filter(arr, sigma): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception: arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception: arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception: arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) # --------------------------------------------------------------------------- # Misc # --------------------------------------------------------------------------- def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass def disable_gpu(): """Disable GPU acceleration for the rest of this process.""" global HAS_GPU, _xp, _cp, _cp_ndimage if not HAS_GPU: return logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") HAS_GPU = False _xp = np _cp = None _cp_ndimage = None def is_gpu_active(): """Check if GPU acceleration is currently active.""" return HAS_GPU def safe_gpu_call(func, *args, **kwargs): """Call a function with GPU arrays, retrying on CPU if GPU fails.""" try: return func(*args, **kwargs) except Exception as e: err_msg = str(e) if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg): logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...") disable_gpu() return func(*args, **kwargs) raise