"""GPU acceleration helpers for LiDAR pipeline. Auto-selects the best NVIDIA GPU (highest compute capability first). Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation works correctly for any architecture (sm_89, sm_120, etc.). All workers share the selected GPU. """ import logging import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") # --------------------------------------------------------------------------- # GPU auto-detection via nvidia-smi (no CUDA context created) # --------------------------------------------------------------------------- _NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 _best_gpu_id: int | None = None _gpu_reason = None # GPU restriction from -g flag (host-level indices) _restricted_gpu_ids: list[int] | None = None # Vrai si CUDA_VISIBLE_DEVICES a été écrit par _init_gpu lui-même (choix du # meilleur GPU). Cette valeur NE DOIT PAS être traitée comme une restriction # externe : sinon available_gpu_ids() ne retourne plus que le GPU choisi et # tous les workers reçoivent le même gpu_id (GPU 1 jamais utilisé). _env_set_by_init = False # Discovered GPU candidates (populated by _pick_gpu) _gpu_candidates: list = [] def _pick_gpu() -> list: """List all GPUs from the system, sorted by compute capability (highest first).""" try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=15, ) if result.returncode != 0: return [] global _NUM_GPUS gpus = [] for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) < 4: continue idx = int(parts[0]) name = parts[1] cap_str = parts[2] mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) score = major * 1000 + minor * 100 + mem_mi gpus.append((idx, name, cap_str, mem_mi, score, major)) _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) return gpus except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return [] try: _gpu_candidates = _pick_gpu() or [] except Exception: pass # --------------------------------------------------------------------------- # Lazy CuPy initialization — tries each GPU until one works # --------------------------------------------------------------------------- _cp = None _cp_ndimage = None _cupy_ndarray = None # type ref qui survit à disable_gpu() pour que to_cpu() récupère les tableaux orphelins _gpu_initialized = False def _filter_candidates(gpus: list) -> list: """Filter GPU candidates by CUDA_VISIBLE_DEVICES and _restricted_gpu_ids. nvidia-smi lists ALL GPUs even when CUDA_VISIBLE_DEVICES is set (driver 580.x behavior), so we must filter manually. La variable écrite par _init_gpu (choix auto du meilleur GPU) est ignorée : seules les restrictions externes (run.sh -g, compose) comptent. """ # Filter by CUDA_VISIBLE_DEVICES if set externally cuda_visible = os.environ.get('CUDA_VISIBLE_DEVICES') if cuda_visible is not None and not _env_set_by_init: try: visible = {int(i.strip()) for i in cuda_visible.split(',')} gpus = [g for g in gpus if g[0] in visible] except ValueError: pass # Filter by programmatic restriction (-g 0, -g 0,2) if _restricted_gpu_ids is not None: allowed = set(_restricted_gpu_ids) gpus = [g for g in gpus if g[0] in allowed] return gpus def _runtime_candidates(): """Détection GPU de repli via le runtime CuPy (sans nvidia-smi). nvidia-smi peut dépasser son timeout quand le système est chargé : _pick_gpu() retourne alors [] et les workers passent à tort en CPU. Les indices CuPy sont renumérotés selon CUDA_VISIBLE_DEVICES — on les remappe en indices hôtes pour rester compatible avec _filter_candidates. """ try: import cupy as _cp_runtime n = _cp_runtime.cuda.runtime.getDeviceCount() cuda_visible = os.environ.get('CUDA_VISIBLE_DEVICES') try: host_ids = [int(v.strip()) for v in cuda_visible.split(',')] except (ValueError, AttributeError): host_ids = list(range(n)) gpus = [] for i in range(min(n, len(host_ids))): props = _cp_runtime.cuda.runtime.getDeviceProperties(i) name = props.get('name', b'?') if isinstance(name, bytes): name = name.decode() major = int(props.get('major', 0)) minor = int(props.get('minor', 0)) mem_mi = int(props.get('totalGlobalMem', 0)) // (1024 * 1024) cap = f"{major}.{minor}" score = major * 1000 + minor * 100 + mem_mi gpus.append((host_ids[i], name, cap, mem_mi, score, major)) gpus.sort(key=lambda g: g[4], reverse=True) return gpus except Exception: return [] def _init_gpu(): """Lazily initialize CuPy on first GPU use. 1. Filters candidates by CUDA_VISIBLE_DEVICES and _restricted_gpu_ids 2. If CUDA_VISIBLE_DEVICES is already set (e.g. run.sh -g 0): import CuPy directly (no subprocess test needed) 3. Otherwise: test each GPU in a subprocess, pick the first that works """ global _cp, _cp_ndimage, _cupy_ndarray, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb if _gpu_initialized: return _gpu_initialized = True candidates = _filter_candidates(_gpu_candidates) if not candidates: # Repli runtime : nvidia-smi a échoué (timeout système chargé) candidates = _filter_candidates(_runtime_candidates()) if not candidates: logger.info("Pas de GPU utilisable — mode CPU uniquement") _cp = None _cp_ndimage = None HAS_GPU = False return cuda_visible = os.environ.get('CUDA_VISIBLE_DEVICES') if cuda_visible is not None: # CUDA_VISIBLE_DEVICES already set (e.g. by run.sh -g 0). # Pick the best visible GPU and import CuPy directly. idx, name, cap_str, mem_mi, score, major = candidates[0] try: import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage # Warm-up kernel to verify GPU works x = _real_cupy.array([1.0, 2.0], dtype=_real_cupy.float32) s = _real_cupy.sum(x).get() if s != 3.0: raise RuntimeError("GPU warm-up failed") _best_gpu_id = idx _gpu_name = name _gpu_mem_gb = mem_mi // 1024 HAS_GPU = True _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage _cupy_ndarray = _real_cupy.ndarray return except Exception as e: logger.warning(f"GPU indisponible (CUDA_VISIBLE_DEVICES={cuda_visible}): {e}") _cp = None _cp_ndimage = None HAS_GPU = False return # No CUDA_VISIBLE_DEVICES set — test each GPU in subprocess import subprocess _working_gpu = None for idx, name, cap_str, mem_mi, score, major in candidates: result = subprocess.run( ['python3', '-c', 'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); ' 'dev=cupy.cuda.runtime.getDevice(); ' 'print(f"OK:{dev}:{cupy.sum(a).get()}")'], capture_output=True, text=True, timeout=120, env={**os.environ, 'CUDA_VISIBLE_DEVICES': str(idx)}, ) stdout = result.stdout.strip() if result.returncode == 0 and stdout.startswith('OK:') and '3.0' in stdout: # Verify the device actually used is device 0 (the GPU we targeted) parts = stdout.split(':') if len(parts) >= 2 and parts[1] == '0': _working_gpu = (idx, name, mem_mi) break else: logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) faux positif CPU fallback") logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) non compatible: " f"{result.stderr.strip().splitlines()[-1] if result.stderr else 'inconnue'}") if _working_gpu is None: logger.info("Pas de GPU utilisable — mode CPU uniquement") _cp = None _cp_ndimage = None HAS_GPU = False return idx, name, mem_mi = _working_gpu global _env_set_by_init _env_set_by_init = True os.environ['CUDA_VISIBLE_DEVICES'] = str(idx) import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage _best_gpu_id = idx _gpu_name = name _gpu_mem_gb = mem_mi // 1024 HAS_GPU = True _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage _cupy_ndarray = _real_cupy.ndarray # --------------------------------------------------------------------------- # Public API # --------------------------------------------------------------------------- def num_gpus(): """Return the number of available GPUs (after restrict_gpus filtering).""" return len(_filter_candidates(_gpu_candidates)) def available_gpu_ids(): """Return list of host-level GPU indices available for processing. Respects any prior restrict_gpus() call. """ return [c[0] for c in _filter_candidates(_gpu_candidates)] def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """Restrict GPU selection to specific host-level indices. Stores the restriction to be applied during _init_gpu(). If set_env_var is True, also sets CUDA_VISIBLE_DEVICES immediately so child processes inherit the restriction. """ global _restricted_gpu_ids _restricted_gpu_ids = gpu_ids if set_env_var and gpu_ids: os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(i) for i in gpu_ids) def set_active_gpu(gpu_id): """Restrict to a single GPU by host-level index. Appelé par les workers spawnés (_process_file_standalone) : CuPy n'y est pas encore initialisé (le worker n'exécute pas log_gpu_status), il faut donc déclencher _init_gpu() ici, sinon le worker retombe silencieusement en CPU. On réduit aussi CUDA_VISIBLE_DEVICES à ce seul GPU avant l'init pour que le device 0 du worker soit le bon : sans cela, tous les workers avec plusieurs GPU visibles partagent le premier d'entre eux. """ global _restricted_gpu_ids _restricted_gpu_ids = [gpu_id] os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id) _init_gpu() def _gpu_available(): """Check if GPU is usable right now.""" try: _init_gpu() return HAS_GPU and _cp is not None except Exception: return False def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: with _cp.cuda.Device(0): props = _cp.cuda.runtime.getDeviceProperties(0) name = props['name'] if isinstance(name, bytes): name = name.decode() mem_gb = props['totalGlobalMem'] // (1024**3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) # List other GPUs for info try: import subprocess result = subprocess.run( ['nvidia-smi', '--query-gpu=index,name,compute_cap', '--format=csv,noheader,nounits'], capture_output=True, text=True, timeout=5, ) if result.returncode == 0: for line in result.stdout.strip().split('\n'): parts = [p.strip() for p in line.split(',')] if len(parts) >= 3: idx = int(parts[0]) if idx != _best_gpu_id: cap = parts[2] logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — disponible") except Exception: pass else: logger.info(f"Pas de GPU utilisable — mode CPU uniquement") # --------------------------------------------------------------------------- # Array transfer # --------------------------------------------------------------------------- def to_gpu(arr): """Send array to GPU if available, otherwise return as float32 numpy.""" if _gpu_available(): try: return _cp.asarray(arr.astype(np.float32)) except Exception as e: # L'erreur d'origine (souvent OOM) ne doit pas être avalée : # sans elle, un repli CPU est indissociable d'un simple bug. logger.warning(f"Transfert GPU échoué ({e}) — repli CPU pour ce calcul") disable_gpu() return arr.astype(np.float32) def to_cpu(arr): """Bring array back to CPU (numpy). No-op if already on CPU. Fonctionne même après disable_gpu() grâce à la réf de type _cupy_ndarray. """ if _cupy_ndarray is not None and isinstance(arr, _cupy_ndarray): try: if _cp is not None: return _cp.asnumpy(arr) return arr.get() except Exception: try: return np.asarray(arr) except Exception: pass return arr # --------------------------------------------------------------------------- # Filters — GPU if array is on GPU, CPU otherwise # --------------------------------------------------------------------------- def xp_gaussian_filter(arr, sigma): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.gaussian_filter(arr, sigma) except Exception as e: logger.warning(f"Filtre gaussien GPU échoué ({e}) — repli CPU") arr = to_cpu(arr) return ndimage.gaussian_filter(arr, sigma) def xp_uniform_filter(arr, size): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.uniform_filter(arr, size) except Exception as e: logger.warning(f"Filtre uniform GPU échoué ({e}) — repli CPU") arr = to_cpu(arr) return ndimage.uniform_filter(arr, size) def xp_minimum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) except Exception as e: logger.warning(f"Filtre minimum GPU échoué ({e}) — repli CPU") arr = to_cpu(arr) return ndimage.minimum_filter(arr, footprint=footprint, size=size) def xp_maximum_filter(arr, footprint=None, size=None): if _cp is not None and isinstance(arr, _cp.ndarray): try: return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) except Exception as e: logger.warning(f"Filtre maximum GPU échoué ({e}) — repli CPU") arr = to_cpu(arr) return ndimage.maximum_filter(arr, footprint=footprint, size=size) # --------------------------------------------------------------------------- # Misc # --------------------------------------------------------------------------- def gpu_cleanup(): """Free GPU memory. Call between visualizations to prevent OOM.""" if _cp is not None: try: _cp.get_default_memory_pool().free_all_blocks() except Exception: pass def disable_gpu(): """Disable GPU acceleration for the rest of this process. Libère le memory pool GPU avant de nuller les références (anti-fuite VRAM). Garde _cupy_ndarray vivant pour que to_cpu() récupère les tableaux orphelins. """ global HAS_GPU, _cp, _cp_ndimage if not HAS_GPU: return logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") gpu_cleanup() HAS_GPU = False _cp = None _cp_ndimage = None def is_gpu_active(): """Check if GPU acceleration is currently active.""" return HAS_GPU def safe_gpu_call(func, *args, **kwargs): """Call a function with GPU arrays, retrying on CPU if GPU fails.""" try: return func(*args, **kwargs) except Exception as e: err_msg = str(e) if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg): logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...") disable_gpu() cpu_args = tuple(to_cpu(a) for a in args) cpu_kwargs = {k: to_cpu(v) for k, v in kwargs.items()} return func(*cpu_args, **cpu_kwargs) raise