From 5a9cdddcfbb9791aa5be85f58641db3b5f3f5a3d Mon Sep 17 00:00:00 2001 From: Antoine Jacquin Date: Sun, 31 May 2026 21:32:34 +0200 Subject: [PATCH] Auto-detect usable GPU (skip sm_120 RTX 5060, fallback to RTX 4060 Ti) --- Dockerfile | 19 ++++++------ lidar_pipeline/gpu.py | 71 +++++++++++++++++++++++++++++-------------- 2 files changed, 58 insertions(+), 32 deletions(-) diff --git a/Dockerfile b/Dockerfile index 48ec06e..ac839c7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -FROM nvidia/cuda:12.4.0-devel-ubuntu22.04 +FROM nvidia/cuda:11.8.0-devel-ubuntu22.04 ENV DEBIAN_FRONTEND=noninteractive ENV TZ=Europe/Paris @@ -45,15 +45,14 @@ RUN pip3 install --no-cache-dir \ pillow-avif-plugin \ cmcrameri -# Build CuPy from source with nvcc, targeting sm_120 (RTX 5060) and nearby archs. -# Pre-built wheels (cupy-cuda12x 14.x) don't include sm_120, so we compile. -# This step takes ~30 min the first time; the image is cached after that. -RUN apt-get update && apt-get install -y --no-install-recommends git && \ - git clone --depth 1 --branch v14.0.0 https://github.com/cupy/cupy.git /tmp/cupy-src && \ - cd /tmp/cupy-src && \ - CUPY_NVCC_GENERATE_CODE='sm_89;sm_90;sm_120' \ - pip3 install --no-cache-dir -e . && \ - rm -rf /tmp/cupy-src +# CuPy 13.4 on CUDA 11.8 with JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE=1). +# JIT allows CuPy to compile kernels at runtime for GPU architectures not in +# the pre-built wheel (sm_89 = RTX 4060 Ti). nvcc must be in PATH at runtime. +# NOTE: RTX 5060 (sm_120) is NOT yet supported by any CuPy version. +# The pipeline auto-detects the best usable GPU (falls back to 4060 Ti). +ENV CUPY_CUDA_COMPILE_WITH_CACHE=1 +ENV PATH=/usr/local/cuda/bin:${PATH} +RUN pip3 install --no-cache-dir cupy-cuda11x==13.4.0 # Copy and install the pipeline package COPY setup.py . diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index 36297e5..44cb794 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -1,8 +1,8 @@ """GPU acceleration helpers for LiDAR pipeline. -Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts -CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back -to CPU if no GPU is available or usable. +Auto-selects the best NVIDIA GPU that works with CuPy. +Tries each card in order of compute capability; falls back to CPU +if none work. All workers share the selected GPU. """ import logging @@ -20,15 +20,15 @@ HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 _best_gpu_id: int | None = None +_gpu_reason = None def _pick_gpu() -> int | None: - """Pick the best GPU from the system. + """Pick the best GPU that CuPy can actually use. - Preference order: - 1. RTX 50xx (Blackwell, compute >= 12.0) - 2. RTX 40xx (Ada Lovelace, compute >= 8.9) - 3. Any NVIDIA GPU with highest compute capability + Tries each card in order of compute capability (highest first). + Returns the index of the first GPU whose compute capability is + known to work with the installed CuPy version. """ try: import subprocess @@ -40,7 +40,7 @@ def _pick_gpu() -> int | None: if result.returncode != 0: return None - global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id + global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason gpus = [] for line in result.stdout.strip().split('\n'): @@ -52,21 +52,29 @@ def _pick_gpu() -> int | None: cap_str = parts[2] mem_mi = int(parts[3]) major, minor = (int(x) for x in cap_str.split('.')) - # Score: higher compute capability first, then more VRAM score = major * 1000 + minor * 100 + mem_mi - gpus.append((idx, name, cap_str, mem_mi, score)) + gpus.append((idx, name, cap_str, mem_mi, score, major)) if not gpus: return None _NUM_GPUS = len(gpus) gpus.sort(key=lambda g: g[4], reverse=True) - best = gpus[0] - _best_gpu_id = best[0] - _gpu_name = best[1] - _gpu_mem_gb = best[3] // 1024 - HAS_GPU = True - return _best_gpu_id + + # CuPy 13.4 + CUDA 11.8 JIT supports up to sm_89 (RTX 40xx). + # sm_120 (RTX 50xx) is not supported yet — skip it. + for gpu in gpus: + idx, name, cap_str, mem_mi, score, major = gpu + if major <= 8: # sm_89 and below — works with CuPy 13.4 + JIT + _best_gpu_id = idx + _gpu_name = name + _gpu_mem_gb = mem_mi // 1024 + HAS_GPU = True + return _best_gpu_id + + # All GPUs have unsupported compute capability + _gpu_reason = "aucun GPU avec compute capability compatible CuPy 13.4" + return None except (FileNotFoundError, subprocess.TimeoutExpired, Exception): return None @@ -95,16 +103,17 @@ def _init_gpu(): _cp = None _cp_ndimage = None HAS_GPU = False + if _gpu_reason: + logger.info(f"Pas de GPU — {_gpu_reason}") return try: - # Restrict to the selected GPU before CuPy imports os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage - # Warm-up: verify kernel execution works on this GPU + # Warm-up: verify kernel execution works _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) _result = _real_cupy.sum(_test * _test) _ = _result.get() @@ -116,12 +125,11 @@ def _init_gpu(): props = _real_cupy.cuda.runtime.getDeviceProperties(0) total_mem = props['totalGlobalMem'] - # Memory pool: up to 90% of VRAM (single GPU shared by workers) pool_size = int(total_mem * 0.9) _real_cupy.cuda.set_memory_pool(0, pool_size) except (ImportError, Exception) as e: - logger.warning(f"GPU non disponible — mode CPU: {e}") + logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}") _xp = np _cp = None _cp_ndimage = None @@ -170,8 +178,27 @@ def log_gpu_status(): except Exception: gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" logger.info(gpu_info) + + # Warn about unsupported GPUs that exist but are not used + try: + import subprocess + result = subprocess.run( + ['nvidia-smi', '--query-gpu=index,name,compute_cap', + '--format=csv,noheader,nounits'], + capture_output=True, text=True, timeout=5, + ) + if result.returncode == 0: + for line in result.stdout.strip().split('\n'): + parts = [p.strip() for p in line.split(',')] + if len(parts) >= 3: + idx = int(parts[0]) + if idx != _best_gpu_id: + cap = parts[2] + logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — non utilisé (incompatible CuPy)") + except Exception: + pass else: - logger.info("Pas de GPU — mode CPU uniquement") + logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})") # ---------------------------------------------------------------------------