Auto-detect best GPU with sm_120 skip: prefer RTX 5060 Ti, fall back to 4060 Ti when nvcc/CuPy does not support sm_120 yet
This commit is contained in:
@ -50,7 +50,7 @@ def _pick_gpu() -> int | None:
|
||||
mem_mi = int(parts[3])
|
||||
major, minor = (int(x) for x in cap_str.split('.'))
|
||||
score = major * 1000 + minor * 100 + mem_mi
|
||||
gpus.append((idx, name, cap_str, mem_mi, score))
|
||||
gpus.append((idx, name, cap_str, mem_mi, score, major))
|
||||
|
||||
if not gpus:
|
||||
return None
|
||||
@ -58,13 +58,21 @@ def _pick_gpu() -> int | None:
|
||||
_NUM_GPUS = len(gpus)
|
||||
gpus.sort(key=lambda g: g[4], reverse=True)
|
||||
|
||||
# Use the best GPU (highest compute capability)
|
||||
best = gpus[0]
|
||||
_best_gpu_id = best[0]
|
||||
_gpu_name = best[1]
|
||||
_gpu_mem_gb = best[3] // 1024
|
||||
HAS_GPU = True
|
||||
return _best_gpu_id
|
||||
# Try GPUs in order of capability.
|
||||
# CuPy 13.4 + CUDA 11.8 JIT works for sm_89 (RTX 40xx).
|
||||
# sm_120 (RTX 50xx) is NOT supported by any nvcc yet (CUDA ≤ 12.9).
|
||||
for gpu in gpus:
|
||||
idx, name, cap_str, mem_mi, score, major = gpu
|
||||
if major >= 12:
|
||||
logger.debug(f" GPU {idx}: {name} (sm_{cap_str}) — non supporté par nvcc/CuPy")
|
||||
continue
|
||||
_best_gpu_id = idx
|
||||
_gpu_name = name
|
||||
_gpu_mem_gb = mem_mi // 1024
|
||||
HAS_GPU = True
|
||||
return _best_gpu_id
|
||||
|
||||
_gpu_reason = "aucun GPU compatible (sm_120+ non supporté par nvcc/CuPy)"
|
||||
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||
return None
|
||||
@ -102,19 +110,19 @@ def _init_gpu():
|
||||
return
|
||||
|
||||
try:
|
||||
# MUST set CUDA_VISIBLE_DEVICES before importing CuPy.
|
||||
# If we don't, CuPy creates its CUDA context on device 0 (sm_89)
|
||||
# and pre-compiled kernels won't work on device 1 (sm_120).
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
|
||||
|
||||
import cupy as _real_cupy
|
||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||
|
||||
# Select the target GPU using Device API
|
||||
with _real_cupy.cuda.Device(_best_gpu_id):
|
||||
# Warm-up: verify kernel execution works on this GPU
|
||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||
_result = _real_cupy.sum(_test * _test)
|
||||
_ = _result.get()
|
||||
del _test, _result
|
||||
|
||||
# Make this device the default for all subsequent operations
|
||||
_real_cupy.cuda.Device(_best_gpu_id).use()
|
||||
# Warm-up: verify kernel execution works on this GPU
|
||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||
_result = _real_cupy.sum(_test * _test)
|
||||
_ = _result.get()
|
||||
del _test, _result
|
||||
|
||||
_xp = _real_cupy
|
||||
_cp = _real_cupy
|
||||
|
||||
Reference in New Issue
Block a user