Auto-detect best GPU with sm_120 skip: prefer RTX 5060 Ti, fall back to 4060 Ti when nvcc/CuPy does not support sm_120 yet

This commit is contained in:
Antoine Jacquin
2026-05-31 23:14:23 +02:00
parent 8490b5ddf0
commit 67d024a64e
2 changed files with 55 additions and 18 deletions

View File

@ -50,7 +50,7 @@ def _pick_gpu() -> int | None:
mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.'))
score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score))
gpus.append((idx, name, cap_str, mem_mi, score, major))
if not gpus:
return None
@ -58,13 +58,21 @@ def _pick_gpu() -> int | None:
_NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True)
# Use the best GPU (highest compute capability)
best = gpus[0]
_best_gpu_id = best[0]
_gpu_name = best[1]
_gpu_mem_gb = best[3] // 1024
HAS_GPU = True
return _best_gpu_id
# Try GPUs in order of capability.
# CuPy 13.4 + CUDA 11.8 JIT works for sm_89 (RTX 40xx).
# sm_120 (RTX 50xx) is NOT supported by any nvcc yet (CUDA ≤ 12.9).
for gpu in gpus:
idx, name, cap_str, mem_mi, score, major = gpu
if major >= 12:
logger.debug(f" GPU {idx}: {name} (sm_{cap_str}) — non supporté par nvcc/CuPy")
continue
_best_gpu_id = idx
_gpu_name = name
_gpu_mem_gb = mem_mi // 1024
HAS_GPU = True
return _best_gpu_id
_gpu_reason = "aucun GPU compatible (sm_120+ non supporté par nvcc/CuPy)"
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None
@ -102,19 +110,19 @@ def _init_gpu():
return
try:
# MUST set CUDA_VISIBLE_DEVICES before importing CuPy.
# If we don't, CuPy creates its CUDA context on device 0 (sm_89)
# and pre-compiled kernels won't work on device 1 (sm_120).
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Select the target GPU using Device API
with _real_cupy.cuda.Device(_best_gpu_id):
# Warm-up: verify kernel execution works on this GPU
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
# Make this device the default for all subsequent operations
_real_cupy.cuda.Device(_best_gpu_id).use()
# Warm-up: verify kernel execution works on this GPU
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
_xp = _real_cupy
_cp = _real_cupy