Fix GPU auto-detection for RTX 5060 Ti (sm_120) — CUDA_VISIBLE_DEVICES via run.sh -e + warm-up in Python
This commit is contained in:
@ -25,8 +25,8 @@ _best_gpu_id: int | None = None
|
|||||||
_gpu_reason = None
|
_gpu_reason = None
|
||||||
|
|
||||||
|
|
||||||
def _pick_gpu() -> int | None:
|
def _pick_gpu() -> list:
|
||||||
"""Pick the best GPU from the system (highest compute capability first)."""
|
"""List all GPUs from the system, sorted by compute capability (highest first)."""
|
||||||
try:
|
try:
|
||||||
import subprocess
|
import subprocess
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
@ -35,9 +35,9 @@ def _pick_gpu() -> int | None:
|
|||||||
capture_output=True, text=True, timeout=5,
|
capture_output=True, text=True, timeout=5,
|
||||||
)
|
)
|
||||||
if result.returncode != 0:
|
if result.returncode != 0:
|
||||||
return None
|
return []
|
||||||
|
|
||||||
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason
|
global _NUM_GPUS
|
||||||
|
|
||||||
gpus = []
|
gpus = []
|
||||||
for line in result.stdout.strip().split('\n'):
|
for line in result.stdout.strip().split('\n'):
|
||||||
@ -52,36 +52,22 @@ def _pick_gpu() -> int | None:
|
|||||||
score = major * 1000 + minor * 100 + mem_mi
|
score = major * 1000 + minor * 100 + mem_mi
|
||||||
gpus.append((idx, name, cap_str, mem_mi, score, major))
|
gpus.append((idx, name, cap_str, mem_mi, score, major))
|
||||||
|
|
||||||
if not gpus:
|
|
||||||
return None
|
|
||||||
|
|
||||||
_NUM_GPUS = len(gpus)
|
_NUM_GPUS = len(gpus)
|
||||||
gpus.sort(key=lambda g: g[4], reverse=True)
|
gpus.sort(key=lambda g: g[4], reverse=True)
|
||||||
|
return gpus
|
||||||
# Try GPUs in order of capability.
|
|
||||||
# CuPy 13.4 + CUDA 11.8 JIT works for sm_89 (RTX 40xx).
|
|
||||||
# sm_120 (RTX 50xx) is NOT supported by any nvcc yet (CUDA ≤ 12.9).
|
|
||||||
for gpu in gpus:
|
|
||||||
idx, name, cap_str, mem_mi, score, major = gpu
|
|
||||||
if major >= 12:
|
|
||||||
logger.debug(f" GPU {idx}: {name} (sm_{cap_str}) — non supporté par nvcc/CuPy")
|
|
||||||
continue
|
|
||||||
_best_gpu_id = idx
|
|
||||||
_gpu_name = name
|
|
||||||
_gpu_mem_gb = mem_mi // 1024
|
|
||||||
HAS_GPU = True
|
|
||||||
return _best_gpu_id
|
|
||||||
|
|
||||||
_gpu_reason = "aucun GPU compatible (sm_120+ non supporté par nvcc/CuPy)"
|
|
||||||
|
|
||||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||||
return None
|
return []
|
||||||
|
|
||||||
|
|
||||||
_best_gpu_id = _pick_gpu()
|
_candidate_gpus: list = []
|
||||||
|
try:
|
||||||
|
_candidate_gpus = _pick_gpu() or []
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES
|
# Lazy CuPy initialization — tries each GPU until one works
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
@ -92,48 +78,62 @@ _gpu_initialized = False
|
|||||||
def _init_gpu():
|
def _init_gpu():
|
||||||
"""Lazily initialize CuPy on first GPU use.
|
"""Lazily initialize CuPy on first GPU use.
|
||||||
|
|
||||||
Uses cupy.cuda.Device() to select the target GPU instead of
|
Uses a subprocess to test each GPU (highest compute capability first).
|
||||||
CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE)
|
The subprocess sets CUDA_VISIBLE_DEVICES before importing CuPy and
|
||||||
has access to the full device topology and can compile kernels for
|
runs a warm-up kernel. First GPU that passes wins.
|
||||||
architectures not in the pre-built wheel (sm_89, sm_120, etc.).
|
|
||||||
|
This is necessary because CUDA_VISIBLE_DEVICES must be set in the
|
||||||
|
process environment BEFORE CuPy imports, not via os.environ in Python.
|
||||||
"""
|
"""
|
||||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb
|
||||||
if _gpu_initialized:
|
if _gpu_initialized:
|
||||||
return
|
return
|
||||||
_gpu_initialized = True
|
_gpu_initialized = True
|
||||||
|
|
||||||
if not HAS_GPU or _best_gpu_id is None:
|
if not _candidate_gpus:
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
_cp_ndimage = None
|
_cp_ndimage = None
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
return
|
return
|
||||||
|
|
||||||
try:
|
# Test each GPU in a subprocess
|
||||||
# MUST set CUDA_VISIBLE_DEVICES before importing CuPy.
|
import subprocess
|
||||||
# If we don't, CuPy creates its CUDA context on device 0 (sm_89)
|
_working_gpu = None
|
||||||
# and pre-compiled kernels won't work on device 1 (sm_120).
|
for idx, name, cap_str, mem_mi, score, major in _candidate_gpus:
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
|
result = subprocess.run(
|
||||||
|
['python3', '-c',
|
||||||
|
'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); print(cupy.sum(a).get())'],
|
||||||
|
capture_output=True, text=True, timeout=120,
|
||||||
|
env={**os.environ, 'CUDA_VISIBLE_DEVICES': str(idx)},
|
||||||
|
)
|
||||||
|
if result.returncode == 0 and '3.0' in result.stdout:
|
||||||
|
_working_gpu = (idx, name, mem_mi)
|
||||||
|
break
|
||||||
|
logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) non compatible: "
|
||||||
|
f"{result.stderr.strip().splitlines()[-1] if result.stderr else 'inconnue'}")
|
||||||
|
|
||||||
import cupy as _real_cupy
|
if _working_gpu is None:
|
||||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
logger.info("Pas de GPU utilisable — mode CPU uniquement")
|
||||||
|
|
||||||
# Warm-up: verify kernel execution works on this GPU
|
|
||||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
|
||||||
_result = _real_cupy.sum(_test * _test)
|
|
||||||
_ = _result.get()
|
|
||||||
del _test, _result
|
|
||||||
|
|
||||||
_xp = _real_cupy
|
|
||||||
_cp = _real_cupy
|
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
|
||||||
|
|
||||||
except (ImportError, Exception) as e:
|
|
||||||
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
|
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
_cp_ndimage = None
|
_cp_ndimage = None
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
|
return
|
||||||
|
|
||||||
|
idx, name, mem_mi = _working_gpu
|
||||||
|
os.environ['CUDA_VISIBLE_DEVICES'] = str(idx)
|
||||||
|
|
||||||
|
import cupy as _real_cupy
|
||||||
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||||
|
|
||||||
|
_best_gpu_id = idx
|
||||||
|
_gpu_name = name
|
||||||
|
_gpu_mem_gb = mem_mi // 1024
|
||||||
|
HAS_GPU = True
|
||||||
|
_xp = _real_cupy
|
||||||
|
_cp = _real_cupy
|
||||||
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
14
run.sh
14
run.sh
@ -247,7 +247,19 @@ if [ -n "$FILE_ARGS" ]; then
|
|||||||
CMD_ARGS="$CMD_ARGS --file $FILE_ARGS"
|
CMD_ARGS="$CMD_ARGS --file $FILE_ARGS"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
docker run --rm --init $GPU_FLAG \
|
# Build CUDA_VISIBLE_DEVICES env var from GPU_ARG
|
||||||
|
CUDA_ENV_FLAG=""
|
||||||
|
if [ -n "$GPU_ARG" ]; then
|
||||||
|
# Convert GPU_ARG (e.g. "0,2" or "1" or "all") to CUDA_VISIBLE_DEVICES
|
||||||
|
if [ "$GPU_ARG" = "all" ]; then
|
||||||
|
CUDA_VISIBLE="0,1,2,3"
|
||||||
|
else
|
||||||
|
CUDA_VISIBLE="$GPU_ARG"
|
||||||
|
fi
|
||||||
|
CUDA_ENV_FLAG="-e CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
docker run --rm --init $GPU_FLAG $CUDA_ENV_FLAG \
|
||||||
--user 1000:1000 \
|
--user 1000:1000 \
|
||||||
-v "${INPUT_DIR}:/data/input:ro" \
|
-v "${INPUT_DIR}:/data/input:ro" \
|
||||||
-v "${OUTPUT_DIR}:/data/output" \
|
-v "${OUTPUT_DIR}:/data/output" \
|
||||||
|
|||||||
Reference in New Issue
Block a user