Use CuPy Device API instead of CUDA_VISIBLE_DEVICES for JIT compatibility on sm_120

This commit is contained in:
Antoine Jacquin
2026-05-31 22:14:18 +02:00
parent a5af50c043
commit 8490b5ddf0

View File

@ -1,8 +1,10 @@
"""GPU acceleration helpers for LiDAR pipeline. """GPU acceleration helpers for LiDAR pipeline.
Auto-selects the best NVIDIA GPU that works with CuPy. Auto-selects the best NVIDIA GPU (highest compute capability first).
Tries each card in order of compute capability; falls back to CPU Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation
if none work. All workers share the selected GPU. works correctly for any architecture (sm_89, sm_120, etc.).
All workers share the selected GPU.
""" """
import logging import logging
@ -24,12 +26,7 @@ _gpu_reason = None
def _pick_gpu() -> int | None: def _pick_gpu() -> int | None:
"""Pick the best GPU that CuPy can actually use. """Pick the best GPU from the system (highest compute capability first)."""
Tries each card in order of compute capability (highest first).
Returns the index of the first GPU whose compute capability is
known to work with the installed CuPy version.
"""
try: try:
import subprocess import subprocess
result = subprocess.run( result = subprocess.run(
@ -53,7 +50,7 @@ def _pick_gpu() -> int | None:
mem_mi = int(parts[3]) mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.')) major, minor = (int(x) for x in cap_str.split('.'))
score = major * 1000 + minor * 100 + mem_mi score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score, major)) gpus.append((idx, name, cap_str, mem_mi, score))
if not gpus: if not gpus:
return None return None
@ -61,16 +58,13 @@ def _pick_gpu() -> int | None:
_NUM_GPUS = len(gpus) _NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True) gpus.sort(key=lambda g: g[4], reverse=True)
# CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any # Use the best GPU (highest compute capability)
# architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable. best = gpus[0]
for gpu in gpus: _best_gpu_id = best[0]
idx, name, cap_str, mem_mi, score, major = gpu _gpu_name = best[1]
_best_gpu_id = idx _gpu_mem_gb = best[3] // 1024
_gpu_name = name HAS_GPU = True
_gpu_mem_gb = mem_mi // 1024 return _best_gpu_id
HAS_GPU = True
return _best_gpu_id
return None
except (FileNotFoundError, subprocess.TimeoutExpired, Exception): except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None return None
@ -79,7 +73,7 @@ def _pick_gpu() -> int | None:
_best_gpu_id = _pick_gpu() _best_gpu_id = _pick_gpu()
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Lazy CuPy initialization # Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
_xp = np _xp = np
_cp = None _cp = None
@ -88,7 +82,13 @@ _gpu_initialized = False
def _init_gpu(): def _init_gpu():
"""Lazily initialize CuPy on first GPU use.""" """Lazily initialize CuPy on first GPU use.
Uses cupy.cuda.Device() to select the target GPU instead of
CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE)
has access to the full device topology and can compile kernels for
architectures not in the pre-built wheel (sm_89, sm_120, etc.).
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
if _gpu_initialized: if _gpu_initialized:
return return
@ -99,29 +99,27 @@ def _init_gpu():
_cp = None _cp = None
_cp_ndimage = None _cp_ndimage = None
HAS_GPU = False HAS_GPU = False
if _gpu_reason:
logger.info(f"Pas de GPU — {_gpu_reason}")
return return
try: try:
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works # Select the target GPU using Device API
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) with _real_cupy.cuda.Device(_best_gpu_id):
_result = _real_cupy.sum(_test * _test) # Warm-up: verify kernel execution works on this GPU
_ = _result.get() _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
del _test, _result _result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
# Make this device the default for all subsequent operations
_real_cupy.cuda.Device(_best_gpu_id).use()
_xp = _real_cupy _xp = _real_cupy
_cp = _real_cupy _cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage _cp_ndimage = _real_cupy_ndimage
# Memory pool management removed — CuPy 13.x uses its own allocator.
# OOM protection handled at the application level (gpu_cleanup() calls).
except (ImportError, Exception) as e: except (ImportError, Exception) as e:
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}") logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
_xp = np _xp = np
@ -164,15 +162,16 @@ def log_gpu_status():
"""Log GPU detection result. Called after logging is configured.""" """Log GPU detection result. Called after logging is configured."""
if _gpu_available(): if _gpu_available():
try: try:
name = _cp.cuda.runtime.getDeviceProperties(0)['name'] with _cp.cuda.Device(_best_gpu_id):
if isinstance(name, bytes): name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name']
name = name.decode() if isinstance(name, bytes):
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3) name = name.decode()
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}" mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
except Exception: except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
logger.info(gpu_info) logger.info(gpu_info)
# List other GPUs for info # List other GPUs for info
try: try:
import subprocess import subprocess
@ -192,7 +191,7 @@ def log_gpu_status():
except Exception: except Exception:
pass pass
else: else:
logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})") logger.info(f"Pas de GPU utilisable — mode CPU uniquement")
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------