Auto-detect usable GPU (skip sm_120 RTX 5060, fallback to RTX 4060 Ti)

This commit is contained in:
Antoine Jacquin
2026-05-31 21:32:34 +02:00
parent 9119d63bc3
commit 5a9cdddcfb
2 changed files with 58 additions and 32 deletions

View File

@ -1,4 +1,4 @@
FROM nvidia/cuda:12.4.0-devel-ubuntu22.04 FROM nvidia/cuda:11.8.0-devel-ubuntu22.04
ENV DEBIAN_FRONTEND=noninteractive ENV DEBIAN_FRONTEND=noninteractive
ENV TZ=Europe/Paris ENV TZ=Europe/Paris
@ -45,15 +45,14 @@ RUN pip3 install --no-cache-dir \
pillow-avif-plugin \ pillow-avif-plugin \
cmcrameri cmcrameri
# Build CuPy from source with nvcc, targeting sm_120 (RTX 5060) and nearby archs. # CuPy 13.4 on CUDA 11.8 with JIT compilation (CUPY_CUDA_COMPILE_WITH_CACHE=1).
# Pre-built wheels (cupy-cuda12x 14.x) don't include sm_120, so we compile. # JIT allows CuPy to compile kernels at runtime for GPU architectures not in
# This step takes ~30 min the first time; the image is cached after that. # the pre-built wheel (sm_89 = RTX 4060 Ti). nvcc must be in PATH at runtime.
RUN apt-get update && apt-get install -y --no-install-recommends git && \ # NOTE: RTX 5060 (sm_120) is NOT yet supported by any CuPy version.
git clone --depth 1 --branch v14.0.0 https://github.com/cupy/cupy.git /tmp/cupy-src && \ # The pipeline auto-detects the best usable GPU (falls back to 4060 Ti).
cd /tmp/cupy-src && \ ENV CUPY_CUDA_COMPILE_WITH_CACHE=1
CUPY_NVCC_GENERATE_CODE='sm_89;sm_90;sm_120' \ ENV PATH=/usr/local/cuda/bin:${PATH}
pip3 install --no-cache-dir -e . && \ RUN pip3 install --no-cache-dir cupy-cuda11x==13.4.0
rm -rf /tmp/cupy-src
# Copy and install the pipeline package # Copy and install the pipeline package
COPY setup.py . COPY setup.py .

View File

@ -1,8 +1,8 @@
"""GPU acceleration helpers for LiDAR pipeline. """GPU acceleration helpers for LiDAR pipeline.
Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts Auto-selects the best NVIDIA GPU that works with CuPy.
CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back Tries each card in order of compute capability; falls back to CPU
to CPU if no GPU is available or usable. if none work. All workers share the selected GPU.
""" """
import logging import logging
@ -20,15 +20,15 @@ HAS_GPU = False
_gpu_name = None _gpu_name = None
_gpu_mem_gb = 0 _gpu_mem_gb = 0
_best_gpu_id: int | None = None _best_gpu_id: int | None = None
_gpu_reason = None
def _pick_gpu() -> int | None: def _pick_gpu() -> int | None:
"""Pick the best GPU from the system. """Pick the best GPU that CuPy can actually use.
Preference order: Tries each card in order of compute capability (highest first).
1. RTX 50xx (Blackwell, compute >= 12.0) Returns the index of the first GPU whose compute capability is
2. RTX 40xx (Ada Lovelace, compute >= 8.9) known to work with the installed CuPy version.
3. Any NVIDIA GPU with highest compute capability
""" """
try: try:
import subprocess import subprocess
@ -40,7 +40,7 @@ def _pick_gpu() -> int | None:
if result.returncode != 0: if result.returncode != 0:
return None return None
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason
gpus = [] gpus = []
for line in result.stdout.strip().split('\n'): for line in result.stdout.strip().split('\n'):
@ -52,22 +52,30 @@ def _pick_gpu() -> int | None:
cap_str = parts[2] cap_str = parts[2]
mem_mi = int(parts[3]) mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.')) major, minor = (int(x) for x in cap_str.split('.'))
# Score: higher compute capability first, then more VRAM
score = major * 1000 + minor * 100 + mem_mi score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score)) gpus.append((idx, name, cap_str, mem_mi, score, major))
if not gpus: if not gpus:
return None return None
_NUM_GPUS = len(gpus) _NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True) gpus.sort(key=lambda g: g[4], reverse=True)
best = gpus[0]
_best_gpu_id = best[0] # CuPy 13.4 + CUDA 11.8 JIT supports up to sm_89 (RTX 40xx).
_gpu_name = best[1] # sm_120 (RTX 50xx) is not supported yet — skip it.
_gpu_mem_gb = best[3] // 1024 for gpu in gpus:
idx, name, cap_str, mem_mi, score, major = gpu
if major <= 8: # sm_89 and below — works with CuPy 13.4 + JIT
_best_gpu_id = idx
_gpu_name = name
_gpu_mem_gb = mem_mi // 1024
HAS_GPU = True HAS_GPU = True
return _best_gpu_id return _best_gpu_id
# All GPUs have unsupported compute capability
_gpu_reason = "aucun GPU avec compute capability compatible CuPy 13.4"
return None
except (FileNotFoundError, subprocess.TimeoutExpired, Exception): except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None return None
@ -95,16 +103,17 @@ def _init_gpu():
_cp = None _cp = None
_cp_ndimage = None _cp_ndimage = None
HAS_GPU = False HAS_GPU = False
if _gpu_reason:
logger.info(f"Pas de GPU — {_gpu_reason}")
return return
try: try:
# Restrict to the selected GPU before CuPy imports
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id) os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works on this GPU # Warm-up: verify kernel execution works
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test) _result = _real_cupy.sum(_test * _test)
_ = _result.get() _ = _result.get()
@ -116,12 +125,11 @@ def _init_gpu():
props = _real_cupy.cuda.runtime.getDeviceProperties(0) props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem'] total_mem = props['totalGlobalMem']
# Memory pool: up to 90% of VRAM (single GPU shared by workers)
pool_size = int(total_mem * 0.9) pool_size = int(total_mem * 0.9)
_real_cupy.cuda.set_memory_pool(0, pool_size) _real_cupy.cuda.set_memory_pool(0, pool_size)
except (ImportError, Exception) as e: except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}") logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
_xp = np _xp = np
_cp = None _cp = None
_cp_ndimage = None _cp_ndimage = None
@ -170,8 +178,27 @@ def log_gpu_status():
except Exception: except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
logger.info(gpu_info) logger.info(gpu_info)
# Warn about unsupported GPUs that exist but are not used
try:
import subprocess
result = subprocess.run(
['nvidia-smi', '--query-gpu=index,name,compute_cap',
'--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5,
)
if result.returncode == 0:
for line in result.stdout.strip().split('\n'):
parts = [p.strip() for p in line.split(',')]
if len(parts) >= 3:
idx = int(parts[0])
if idx != _best_gpu_id:
cap = parts[2]
logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — non utilisé (incompatible CuPy)")
except Exception:
pass
else: else:
logger.info("Pas de GPU — mode CPU uniquement") logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})")
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------