From 8490b5ddf004c9fa24983d3a64e491aa5d00d39f Mon Sep 17 00:00:00 2001 From: Antoine Jacquin Date: Sun, 31 May 2026 22:14:18 +0200 Subject: [PATCH] Use CuPy Device API instead of CUDA_VISIBLE_DEVICES for JIT compatibility on sm_120 --- lidar_pipeline/gpu.py | 81 +++++++++++++++++++++---------------------- 1 file changed, 40 insertions(+), 41 deletions(-) diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index 1b1f50f..1bd3865 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -1,8 +1,10 @@ """GPU acceleration helpers for LiDAR pipeline. -Auto-selects the best NVIDIA GPU that works with CuPy. -Tries each card in order of compute capability; falls back to CPU -if none work. All workers share the selected GPU. +Auto-selects the best NVIDIA GPU (highest compute capability first). +Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation +works correctly for any architecture (sm_89, sm_120, etc.). + +All workers share the selected GPU. """ import logging @@ -24,12 +26,7 @@ _gpu_reason = None def _pick_gpu() -> int | None: - """Pick the best GPU that CuPy can actually use. - - Tries each card in order of compute capability (highest first). - Returns the index of the first GPU whose compute capability is - known to work with the installed CuPy version. - """ + """Pick the best GPU from the system (highest compute capability first).""" try: import subprocess result = subprocess.run( @@ -53,7 +50,7 @@ def _pick_gpu() -> int | None: mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) score = major * 1000 + minor * 100 + mem_mi - gpus.append((idx, name, cap_str, mem_mi, score, major)) + gpus.append((idx, name, cap_str, mem_mi, score)) if not gpus: return None @@ -61,16 +58,13 @@ def _pick_gpu() -> int | None: _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) - # CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any - # architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable. - for gpu in gpus: - idx, name, cap_str, mem_mi, score, major = gpu - _best_gpu_id = idx - _gpu_name = name - _gpu_mem_gb = mem_mi // 1024 - HAS_GPU = True - return _best_gpu_id - return None + # Use the best GPU (highest compute capability) + best = gpus[0] + _best_gpu_id = best[0] + _gpu_name = best[1] + _gpu_mem_gb = best[3] // 1024 + HAS_GPU = True + return _best_gpu_id except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return None @@ -79,7 +73,7 @@ def _pick_gpu() -> int | None: _best_gpu_id = _pick_gpu() # --------------------------------------------------------------------------- -# Lazy CuPy initialization +# Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES # --------------------------------------------------------------------------- _xp = np _cp = None @@ -88,7 +82,13 @@ _gpu_initialized = False def _init_gpu(): - """Lazily initialize CuPy on first GPU use.""" + """Lazily initialize CuPy on first GPU use. + + Uses cupy.cuda.Device() to select the target GPU instead of + CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE) + has access to the full device topology and can compile kernels for + architectures not in the pre-built wheel (sm_89, sm_120, etc.). + """ global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU if _gpu_initialized: return @@ -99,29 +99,27 @@ def _init_gpu(): _cp = None _cp_ndimage = None HAS_GPU = False - if _gpu_reason: - logger.info(f"Pas de GPU — {_gpu_reason}") return try: - os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) - import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage - # Warm-up: verify kernel execution works - _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) - _result = _real_cupy.sum(_test * _test) - _ = _result.get() - del _test, _result + # Select the target GPU using Device API + with _real_cupy.cuda.Device(_best_gpu_id): + # Warm-up: verify kernel execution works on this GPU + _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) + _result = _real_cupy.sum(_test * _test) + _ = _result.get() + del _test, _result + + # Make this device the default for all subsequent operations + _real_cupy.cuda.Device(_best_gpu_id).use() _xp = _real_cupy _cp = _real_cupy _cp_ndimage = _real_cupy_ndimage - # Memory pool management removed — CuPy 13.x uses its own allocator. - # OOM protection handled at the application level (gpu_cleanup() calls). - except (ImportError, Exception) as e: logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}") _xp = np @@ -164,15 +162,16 @@ def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" if _gpu_available(): try: - name = _cp.cuda.runtime.getDeviceProperties(0)['name'] - if isinstance(name, bytes): - name = name.decode() - mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3) - gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" + with _cp.cuda.Device(_best_gpu_id): + name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name'] + if isinstance(name, bytes): + name = name.decode() + mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3) + gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) - + # List other GPUs for info try: import subprocess @@ -192,7 +191,7 @@ def log_gpu_status(): except Exception: pass else: - logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})") + logger.info(f"Pas de GPU utilisable — mode CPU uniquement") # ---------------------------------------------------------------------------