Use CuPy Device API instead of CUDA_VISIBLE_DEVICES for JIT compatibility on sm_120

This commit is contained in:
Antoine Jacquin
2026-05-31 22:14:18 +02:00
parent a5af50c043
commit 8490b5ddf0

View File

@ -1,8 +1,10 @@
"""GPU acceleration helpers for LiDAR pipeline.
Auto-selects the best NVIDIA GPU that works with CuPy.
Tries each card in order of compute capability; falls back to CPU
if none work. All workers share the selected GPU.
Auto-selects the best NVIDIA GPU (highest compute capability first).
Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation
works correctly for any architecture (sm_89, sm_120, etc.).
All workers share the selected GPU.
"""
import logging
@ -24,12 +26,7 @@ _gpu_reason = None
def _pick_gpu() -> int | None:
"""Pick the best GPU that CuPy can actually use.
Tries each card in order of compute capability (highest first).
Returns the index of the first GPU whose compute capability is
known to work with the installed CuPy version.
"""
"""Pick the best GPU from the system (highest compute capability first)."""
try:
import subprocess
result = subprocess.run(
@ -53,7 +50,7 @@ def _pick_gpu() -> int | None:
mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.'))
score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score, major))
gpus.append((idx, name, cap_str, mem_mi, score))
if not gpus:
return None
@ -61,16 +58,13 @@ def _pick_gpu() -> int | None:
_NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True)
# CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any
# architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable.
for gpu in gpus:
idx, name, cap_str, mem_mi, score, major = gpu
_best_gpu_id = idx
_gpu_name = name
_gpu_mem_gb = mem_mi // 1024
HAS_GPU = True
return _best_gpu_id
return None
# Use the best GPU (highest compute capability)
best = gpus[0]
_best_gpu_id = best[0]
_gpu_name = best[1]
_gpu_mem_gb = best[3] // 1024
HAS_GPU = True
return _best_gpu_id
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None
@ -79,7 +73,7 @@ def _pick_gpu() -> int | None:
_best_gpu_id = _pick_gpu()
# ---------------------------------------------------------------------------
# Lazy CuPy initialization
# Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES
# ---------------------------------------------------------------------------
_xp = np
_cp = None
@ -88,7 +82,13 @@ _gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use."""
"""Lazily initialize CuPy on first GPU use.
Uses cupy.cuda.Device() to select the target GPU instead of
CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE)
has access to the full device topology and can compile kernels for
architectures not in the pre-built wheel (sm_89, sm_120, etc.).
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
if _gpu_initialized:
return
@ -99,29 +99,27 @@ def _init_gpu():
_cp = None
_cp_ndimage = None
HAS_GPU = False
if _gpu_reason:
logger.info(f"Pas de GPU — {_gpu_reason}")
return
try:
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
# Select the target GPU using Device API
with _real_cupy.cuda.Device(_best_gpu_id):
# Warm-up: verify kernel execution works on this GPU
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
# Make this device the default for all subsequent operations
_real_cupy.cuda.Device(_best_gpu_id).use()
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
# Memory pool management removed — CuPy 13.x uses its own allocator.
# OOM protection handled at the application level (gpu_cleanup() calls).
except (ImportError, Exception) as e:
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
_xp = np
@ -164,11 +162,12 @@ def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
try:
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
with _cp.cuda.Device(_best_gpu_id):
name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
logger.info(gpu_info)
@ -192,7 +191,7 @@ def log_gpu_status():
except Exception:
pass
else:
logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})")
logger.info(f"Pas de GPU utilisable — mode CPU uniquement")
# ---------------------------------------------------------------------------