diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index 22bf65a..a7e97d3 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -25,8 +25,8 @@ _best_gpu_id: int | None = None _gpu_reason = None -def _pick_gpu() -> int | None: - """Pick the best GPU from the system (highest compute capability first).""" +def _pick_gpu() -> list: + """List all GPUs from the system, sorted by compute capability (highest first).""" try: import subprocess result = subprocess.run( @@ -35,9 +35,9 @@ def _pick_gpu() -> int | None: capture_output=True, text=True, timeout=5, ) if result.returncode != 0: - return None + return [] - global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason + global _NUM_GPUS gpus = [] for line in result.stdout.strip().split('\n'): @@ -52,36 +52,22 @@ def _pick_gpu() -> int | None: score = major * 1000 + minor * 100 + mem_mi gpus.append((idx, name, cap_str, mem_mi, score, major)) - if not gpus: - return None - _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) - - # Try GPUs in order of capability. - # CuPy 13.4 + CUDA 11.8 JIT works for sm_89 (RTX 40xx). - # sm_120 (RTX 50xx) is NOT supported by any nvcc yet (CUDA ≤ 12.9). - for gpu in gpus: - idx, name, cap_str, mem_mi, score, major = gpu - if major >= 12: - logger.debug(f" GPU {idx}: {name} (sm_{cap_str}) — non supporté par nvcc/CuPy") - continue - _best_gpu_id = idx - _gpu_name = name - _gpu_mem_gb = mem_mi // 1024 - HAS_GPU = True - return _best_gpu_id - - _gpu_reason = "aucun GPU compatible (sm_120+ non supporté par nvcc/CuPy)" + return gpus except (FileNotFoundError, subprocess.TimeoutExpired, Exception): - return None + return [] -_best_gpu_id = _pick_gpu() +_candidate_gpus: list = [] +try: + _candidate_gpus = _pick_gpu() or [] +except Exception: + pass # --------------------------------------------------------------------------- -# Lazy CuPy initialization — uses Device API, not CUDA_VISIBLE_DEVICES +# Lazy CuPy initialization — tries each GPU until one works # --------------------------------------------------------------------------- _xp = np _cp = None @@ -92,48 +78,62 @@ _gpu_initialized = False def _init_gpu(): """Lazily initialize CuPy on first GPU use. - Uses cupy.cuda.Device() to select the target GPU instead of - CUDA_VISIBLE_DEVICES, so JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE) - has access to the full device topology and can compile kernels for - architectures not in the pre-built wheel (sm_89, sm_120, etc.). + Uses a subprocess to test each GPU (highest compute capability first). + The subprocess sets CUDA_VISIBLE_DEVICES before importing CuPy and + runs a warm-up kernel. First GPU that passes wins. + + This is necessary because CUDA_VISIBLE_DEVICES must be set in the + process environment BEFORE CuPy imports, not via os.environ in Python. """ - global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU + global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb if _gpu_initialized: return _gpu_initialized = True - if not HAS_GPU or _best_gpu_id is None: + if not _candidate_gpus: _xp = np _cp = None _cp_ndimage = None HAS_GPU = False return - try: - # MUST set CUDA_VISIBLE_DEVICES before importing CuPy. - # If we don't, CuPy creates its CUDA context on device 0 (sm_89) - # and pre-compiled kernels won't work on device 1 (sm_120). - os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) + # Test each GPU in a subprocess + import subprocess + _working_gpu = None + for idx, name, cap_str, mem_mi, score, major in _candidate_gpus: + result = subprocess.run( + ['python3', '-c', + 'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); print(cupy.sum(a).get())'], + capture_output=True, text=True, timeout=120, + env={**os.environ, 'CUDA_VISIBLE_DEVICES': str(idx)}, + ) + if result.returncode == 0 and '3.0' in result.stdout: + _working_gpu = (idx, name, mem_mi) + break + logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) non compatible: " + f"{result.stderr.strip().splitlines()[-1] if result.stderr else 'inconnue'}") - import cupy as _real_cupy - import cupyx.scipy.ndimage as _real_cupy_ndimage - - # Warm-up: verify kernel execution works on this GPU - _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) - _result = _real_cupy.sum(_test * _test) - _ = _result.get() - del _test, _result - - _xp = _real_cupy - _cp = _real_cupy - _cp_ndimage = _real_cupy_ndimage - - except (ImportError, Exception) as e: - logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}") + if _working_gpu is None: + logger.info("Pas de GPU utilisable — mode CPU uniquement") _xp = np _cp = None _cp_ndimage = None HAS_GPU = False + return + + idx, name, mem_mi = _working_gpu + os.environ['CUDA_VISIBLE_DEVICES'] = str(idx) + + import cupy as _real_cupy + import cupyx.scipy.ndimage as _real_cupy_ndimage + + _best_gpu_id = idx + _gpu_name = name + _gpu_mem_gb = mem_mi // 1024 + HAS_GPU = True + _xp = _real_cupy + _cp = _real_cupy + _cp_ndimage = _real_cupy_ndimage # --------------------------------------------------------------------------- diff --git a/run.sh b/run.sh index 08ab1cc..7836595 100755 --- a/run.sh +++ b/run.sh @@ -247,7 +247,19 @@ if [ -n "$FILE_ARGS" ]; then CMD_ARGS="$CMD_ARGS --file $FILE_ARGS" fi -docker run --rm --init $GPU_FLAG \ +# Build CUDA_VISIBLE_DEVICES env var from GPU_ARG +CUDA_ENV_FLAG="" +if [ -n "$GPU_ARG" ]; then + # Convert GPU_ARG (e.g. "0,2" or "1" or "all") to CUDA_VISIBLE_DEVICES + if [ "$GPU_ARG" = "all" ]; then + CUDA_VISIBLE="0,1,2,3" + else + CUDA_VISIBLE="$GPU_ARG" + fi + CUDA_ENV_FLAG="-e CUDA_VISIBLE_DEVICES=${CUDA_VISIBLE}" +fi + +docker run --rm --init $GPU_FLAG $CUDA_ENV_FLAG \ --user 1000:1000 \ -v "${INPUT_DIR}:/data/input:ro" \ -v "${OUTPUT_DIR}:/data/output" \