From c58ca3f477d26ff82ca56238415cbdffc962f49f Mon Sep 17 00:00:00 2001 From: Antoine Jacquin Date: Thu, 23 Jul 2026 20:06:14 +0200 Subject: [PATCH] Fix multi-GPU detection, VRAM leak on disable, and orphaned GPU array handling _gpu_candidates was never populated (typo: _candidate_gpus), breaking num_gpus()/available_gpu_ids() and the multi-GPU round-robin. available_gpu_ids also indexed tuples with dict syntax (TypeError). disable_gpu() now frees the memory pool before clearing refs (was leaking VRAM). to_cpu() and safe_gpu_call() survive disable_gpu() via a persistent _cupy_ndarray type reference so orphaned GPU arrays are recovered to CPU. log_gpu_status() now uses Device(0) instead of the host index (CuPy renumbers to 0 after CUDA_VISIBLE_DEVICES). --- lidar_pipeline/gpu.py | 54 ++++++++++++++++++++++++++++++------------- 1 file changed, 38 insertions(+), 16 deletions(-) diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index f2f5ddd..f3331fd 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -28,7 +28,7 @@ _gpu_reason = None _restricted_gpu_ids: list[int] | None = None # Discovered GPU candidates (populated by _pick_gpu) -_gpu_candidates: list[dict] = [] +_gpu_candidates: list = [] def _pick_gpu() -> list: @@ -66,9 +66,8 @@ def _pick_gpu() -> list: return [] -_candidate_gpus: list = [] try: - _candidate_gpus = _pick_gpu() or [] + _gpu_candidates = _pick_gpu() or [] except Exception: pass @@ -78,6 +77,7 @@ except Exception: _xp = np _cp = None _cp_ndimage = None +_cupy_ndarray = None # type ref qui survit à disable_gpu() pour que to_cpu() récupère les tableaux orphelins _gpu_initialized = False @@ -112,12 +112,12 @@ def _init_gpu(): import CuPy directly (no subprocess test needed) 3. Otherwise: test each GPU in a subprocess, pick the first that works """ - global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb + global _xp, _cp, _cp_ndimage, _cupy_ndarray, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb if _gpu_initialized: return _gpu_initialized = True - candidates = _filter_candidates(_candidate_gpus) + candidates = _filter_candidates(_gpu_candidates) if not candidates: logger.info("Pas de GPU utilisable — mode CPU uniquement") _xp = np @@ -149,6 +149,7 @@ def _init_gpu(): _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage + _cupy_ndarray = _real_cupy.ndarray return except Exception as e: logger.warning(f"GPU indisponible (CUDA_VISIBLE_DEVICES={cuda_visible}): {e}") @@ -203,6 +204,7 @@ def _init_gpu(): _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage + _cupy_ndarray = _real_cupy.ndarray # --------------------------------------------------------------------------- @@ -211,7 +213,7 @@ def _init_gpu(): def num_gpus(): """Return the number of available GPUs (after restrict_gpus filtering).""" - return len(_gpu_candidates) + return len(_filter_candidates(_gpu_candidates)) def available_gpu_ids(): @@ -219,16 +221,20 @@ def available_gpu_ids(): Respects any prior restrict_gpus() call. """ - return [c['id'] for c in _gpu_candidates] + return [c[0] for c in _filter_candidates(_gpu_candidates)] def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """Restrict GPU selection to specific host-level indices. Stores the restriction to be applied during _init_gpu(). + If set_env_var is True, also sets CUDA_VISIBLE_DEVICES immediately so + child processes inherit the restriction. """ global _restricted_gpu_ids _restricted_gpu_ids = gpu_ids + if set_env_var and gpu_ids: + os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(i) for i in gpu_ids) def set_active_gpu(gpu_id): @@ -250,11 +256,12 @@ def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: - with _cp.cuda.Device(_best_gpu_id): - name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name'] + with _cp.cuda.Device(0): + props = _cp.cuda.runtime.getDeviceProperties(0) + name = props['name'] if isinstance(name, bytes): name = name.decode() - mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3) + mem_gb = props['totalGlobalMem'] // (1024**3) gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" @@ -297,12 +304,20 @@ def to_gpu(arr): def to_cpu(arr): - """Bring array back to CPU (numpy). No-op if already on CPU.""" - if _cp is not None and isinstance(arr, _cp.ndarray): + """Bring array back to CPU (numpy). No-op if already on CPU. + + Fonctionne même après disable_gpu() grâce à la réf de type _cupy_ndarray. + """ + if _cupy_ndarray is not None and isinstance(arr, _cupy_ndarray): try: - return _cp.asnumpy(arr) + if _cp is not None: + return _cp.asnumpy(arr) + return arr.get() except Exception: - pass + try: + return np.asarray(arr) + except Exception: + pass return arr @@ -360,11 +375,16 @@ def gpu_cleanup(): def disable_gpu(): - """Disable GPU acceleration for the rest of this process.""" + """Disable GPU acceleration for the rest of this process. + + Libère le memory pool GPU avant de nuller les références (anti-fuite VRAM). + Garde _cupy_ndarray vivant pour que to_cpu() récupère les tableaux orphelins. + """ global HAS_GPU, _xp, _cp, _cp_ndimage if not HAS_GPU: return logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") + gpu_cleanup() HAS_GPU = False _xp = np _cp = None @@ -385,5 +405,7 @@ def safe_gpu_call(func, *args, **kwargs): if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg): logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...") disable_gpu() - return func(*args, **kwargs) + cpu_args = tuple(to_cpu(a) for a in args) + cpu_kwargs = {k: to_cpu(v) for k, v in kwargs.items()} + return func(*cpu_args, **cpu_kwargs) raise