Use CuPy Device API instead of CUDA_VISIBLE_DEVICES for JIT compatibility on sm_120
This commit is contained in:
@ -1,8 +1,10 @@
|
|||||||
"""GPU acceleration helpers for LiDAR pipeline.
|
"""GPU acceleration helpers for LiDAR pipeline.
|
||||||
|
|
||||||
Auto-selects the best NVIDIA GPU that works with CuPy.
|
Auto-selects the best NVIDIA GPU (highest compute capability first).
|
||||||
Tries each card in order of compute capability; falls back to CPU
|
Uses CuPy Device API (not CUDA_VISIBLE_DEVICES) so JIT compilation
|
||||||
if none work. All workers share the selected GPU.
|
works correctly for any architecture (sm_89, sm_120, etc.).
|
||||||
|
|
||||||
|
All workers share the selected GPU.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
@ -24,12 +26,7 @@ _gpu_reason = None
|
|||||||
|
|
||||||
|
|
||||||
def _pick_gpu() -> int | None:
|
def _pick_gpu() -> int | None:
|
||||||
"""Pick the best GPU that CuPy can actually use.
|
"""Pick the best GPU from the system (highest compute capability first)."""
|
||||||
|
|
||||||
Tries each card in order of compute capability (highest first).
|
|
||||||
Returns the index of the first GPU whose compute capability is
|
|
||||||
known to work with the installed CuPy version.
|
|
||||||
"""
|
|
||||||
try:
|
try:
|
||||||
import subprocess
|
import subprocess
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
@ -53,7 +50,7 @@ def _pick_gpu() -> int | None:
|
|||||||
mem_mi = int(parts[3])
|
mem_mi = int(parts[3])
|
||||||
major, minor = (int(x) for x in cap_str.split('.'))
|
major, minor = (int(x) for x in cap_str.split('.'))
|
||||||
score = major * 1000 + minor * 100 + mem_mi
|
score = major * 1000 + minor * 100 + mem_mi
|
||||||
gpus.append((idx, name, cap_str, mem_mi, score, major))
|
gpus.append((idx, name, cap_str, mem_mi, score))
|
||||||
|
|
||||||
if not gpus:
|
if not gpus:
|
||||||
return None
|
return None
|
||||||
@ -61,16 +58,13 @@ def _pick_gpu() -> int | None:
|
|||||||
_NUM_GPUS = len(gpus)
|
_NUM_GPUS = len(gpus)
|
||||||
gpus.sort(key=lambda g: g[4], reverse=True)
|
gpus.sort(key=lambda g: g[4], reverse=True)
|
||||||
|
|
||||||
# CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any
|
# Use the best GPU (highest compute capability)
|
||||||
# architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable.
|
best = gpus[0]
|
||||||
for gpu in gpus:
|
_best_gpu_id = best[0]
|
||||||
idx, name, cap_str, mem_mi, score, major = gpu
|
_gpu_name = best[1]
|
||||||
_best_gpu_id = idx
|
_gpu_mem_gb = best[3] // 1024
|
||||||
_gpu_name = name
|
HAS_GPU = True
|
||||||
_gpu_mem_gb = mem_mi // 1024
|
return _best_gpu_id
|
||||||
HAS_GPU = True
|
|
||||||
return _best_gpu_id
|
|
||||||
return None
|
|
||||||
|
|
||||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||||
return None
|
return None
|
||||||
@ -79,7 +73,7 @@ def _pick_gpu() -> int | None:
|
|||||||
_best_gpu_id = _pick_gpu()
|
_best_gpu_id = _pick_gpu()
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Lazy CuPy initialization
|
# Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
@ -88,7 +82,13 @@ _gpu_initialized = False
|
|||||||
|
|
||||||
|
|
||||||
def _init_gpu():
|
def _init_gpu():
|
||||||
"""Lazily initialize CuPy on first GPU use."""
|
"""Lazily initialize CuPy on first GPU use.
|
||||||
|
|
||||||
|
Uses cupy.cuda.Device() to select the target GPU instead of
|
||||||
|
CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE)
|
||||||
|
has access to the full device topology and can compile kernels for
|
||||||
|
architectures not in the pre-built wheel (sm_89, sm_120, etc.).
|
||||||
|
"""
|
||||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
||||||
if _gpu_initialized:
|
if _gpu_initialized:
|
||||||
return
|
return
|
||||||
@ -99,29 +99,27 @@ def _init_gpu():
|
|||||||
_cp = None
|
_cp = None
|
||||||
_cp_ndimage = None
|
_cp_ndimage = None
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
if _gpu_reason:
|
|
||||||
logger.info(f"Pas de GPU — {_gpu_reason}")
|
|
||||||
return
|
return
|
||||||
|
|
||||||
try:
|
try:
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
|
|
||||||
|
|
||||||
import cupy as _real_cupy
|
import cupy as _real_cupy
|
||||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||||
|
|
||||||
# Warm-up: verify kernel execution works
|
# Select the target GPU using Device API
|
||||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
with _real_cupy.cuda.Device(_best_gpu_id):
|
||||||
_result = _real_cupy.sum(_test * _test)
|
# Warm-up: verify kernel execution works on this GPU
|
||||||
_ = _result.get()
|
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||||
del _test, _result
|
_result = _real_cupy.sum(_test * _test)
|
||||||
|
_ = _result.get()
|
||||||
|
del _test, _result
|
||||||
|
|
||||||
|
# Make this device the default for all subsequent operations
|
||||||
|
_real_cupy.cuda.Device(_best_gpu_id).use()
|
||||||
|
|
||||||
_xp = _real_cupy
|
_xp = _real_cupy
|
||||||
_cp = _real_cupy
|
_cp = _real_cupy
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
|
||||||
# Memory pool management removed — CuPy 13.x uses its own allocator.
|
|
||||||
# OOM protection handled at the application level (gpu_cleanup() calls).
|
|
||||||
|
|
||||||
except (ImportError, Exception) as e:
|
except (ImportError, Exception) as e:
|
||||||
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
|
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
|
||||||
_xp = np
|
_xp = np
|
||||||
@ -164,11 +162,12 @@ def log_gpu_status():
|
|||||||
"""Log GPU detection result. Called after logging is configured."""
|
"""Log GPU detection result. Called after logging is configured."""
|
||||||
if _gpu_available():
|
if _gpu_available():
|
||||||
try:
|
try:
|
||||||
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
|
with _cp.cuda.Device(_best_gpu_id):
|
||||||
if isinstance(name, bytes):
|
name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name']
|
||||||
name = name.decode()
|
if isinstance(name, bytes):
|
||||||
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
|
name = name.decode()
|
||||||
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3)
|
||||||
|
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
||||||
except Exception:
|
except Exception:
|
||||||
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||||
logger.info(gpu_info)
|
logger.info(gpu_info)
|
||||||
@ -192,7 +191,7 @@ def log_gpu_status():
|
|||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
else:
|
else:
|
||||||
logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})")
|
logger.info(f"Pas de GPU utilisable — mode CPU uniquement")
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user