Auto-detect best GPU (RTX 5060 preferred) + build CuPy from source for sm_120

This commit is contained in:
Antoine Jacquin
2026-05-31 20:55:06 +02:00
parent d2f382c94d
commit 9119d63bc3
2 changed files with 135 additions and 230 deletions

View File

@ -1,4 +1,4 @@
FROM nvidia/cuda:11.8.0-devel-ubuntu22.04 FROM nvidia/cuda:12.4.0-devel-ubuntu22.04
ENV DEBIAN_FRONTEND=noninteractive ENV DEBIAN_FRONTEND=noninteractive
ENV TZ=Europe/Paris ENV TZ=Europe/Paris
@ -45,15 +45,15 @@ RUN pip3 install --no-cache-dir \
pillow-avif-plugin \ pillow-avif-plugin \
cmcrameri cmcrameri
# Install CuPy for GPU acceleration (optional - will fallback to numpy if not available) # Build CuPy from source with nvcc, targeting sm_120 (RTX 5060) and nearby archs.
# We use CuPy 13.4 (CUDA 11.x wheel) because CuPy 14.x dropped JIT compilation # Pre-built wheels (cupy-cuda12x 14.x) don't include sm_120, so we compile.
# support. With CUPY_CUDA_COMPILE_WITH_CACHE=1, CuPy 13.4 compiles kernels at # This step takes ~30 min the first time; the image is cached after that.
# runtime for GPU architectures not in the pre-built wheel (e.g. sm_89 / RTX 4060 Ti). RUN apt-get update && apt-get install -y --no-install-recommends git && \
# The devel image includes nvcc, required for JIT compilation. git clone --depth 1 --branch v14.0.0 https://github.com/cupy/cupy.git /tmp/cupy-src && \
# nvcc stays in PATH for the 'lidar' user. cd /tmp/cupy-src && \
ENV CUPY_CUDA_COMPILE_WITH_CACHE=1 CUPY_NVCC_GENERATE_CODE='sm_89;sm_90;sm_120' \
ENV PATH=/usr/local/cuda/bin:${PATH} pip3 install --no-cache-dir -e . && \
RUN pip3 install --no-cache-dir cupy-cuda11x==13.4.0 || echo "CuPy not available - GPU acceleration disabled" rm -rf /tmp/cupy-src
# Copy and install the pipeline package # Copy and install the pipeline package
COPY setup.py . COPY setup.py .

View File

@ -1,15 +1,8 @@
"""GPU acceleration helpers for LiDAR pipeline. """GPU acceleration helpers for LiDAR pipeline.
Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts
is detected, array operations are accelerated on the GPU. Otherwise, all CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back
operations fall back to numpy/scipy on CPU. to CPU if no GPU is available or usable.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before
CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel
cache incompatibilities that occur with Device.use() switching.
""" """
import logging import logging
@ -19,125 +12,113 @@ from scipy import ndimage
logger = logging.getLogger("lidar") logger = logging.getLogger("lidar")
# Detect total GPU count via nvidia-smi (no CUDA context created). # ---------------------------------------------------------------------------
# This must happen before any CUDA_VISIBLE_DEVICES manipulation. # GPU auto-detection via nvidia-smi (no CUDA context created)
# ---------------------------------------------------------------------------
_NUM_GPUS = 0 _NUM_GPUS = 0
HAS_GPU = False HAS_GPU = False
_gpu_name = None _gpu_name = None
_gpu_mem_gb = 0 _gpu_mem_gb = 0
# System-level GPU IDs that are currently visible (after restrict_gpus). _best_gpu_id: int | None = None
# Populated by restrict_gpus() or auto-detected at import time.
_available_gpu_ids: list[int] = []
try:
import subprocess def _pick_gpu() -> int | None:
_result = subprocess.run( """Pick the best GPU from the system.
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5 Preference order:
) 1. RTX 50xx (Blackwell, compute >= 12.0)
if _result.returncode == 0: 2. RTX 40xx (Ada Lovelace, compute >= 8.9)
_lines = _result.stdout.strip().split('\n') 3. Any NVIDIA GPU with highest compute capability
_NUM_GPUS = len(_lines) """
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try: try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024 import subprocess
except (ValueError, IndexError): result = subprocess.run(
pass ['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total',
HAS_GPU = True '--format=csv,noheader,nounits'],
# All GPUs are visible by default capture_output=True, text=True, timeout=5,
_available_gpu_ids = list(range(_NUM_GPUS)) )
except (FileNotFoundError, subprocess.TimeoutExpired, Exception): if result.returncode != 0:
pass return None
# Lazy CuPy initialization — imported only when first needed. global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id
# This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates
# a CUDA context, enabling per-process GPU assignment. gpus = []
_xp = np # Default: CPU for line in result.stdout.strip().split('\n'):
_cp = None # cupy module (or None) parts = [p.strip() for p in line.split(',')]
_cp_ndimage = None # cupyx.scipy.ndimage (or None) if len(parts) < 4:
continue
idx = int(parts[0])
name = parts[1]
cap_str = parts[2]
mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.'))
# Score: higher compute capability first, then more VRAM
score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score))
if not gpus:
return None
_NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True)
best = gpus[0]
_best_gpu_id = best[0]
_gpu_name = best[1]
_gpu_mem_gb = best[3] // 1024
HAS_GPU = True
return _best_gpu_id
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None
_best_gpu_id = _pick_gpu()
# ---------------------------------------------------------------------------
# Lazy CuPy initialization
# ---------------------------------------------------------------------------
_xp = np
_cp = None
_cp_ndimage = None
_gpu_initialized = False _gpu_initialized = False
def _init_gpu(): def _init_gpu():
"""Lazily initialize CuPy on first GPU use. """Lazily initialize CuPy on first GPU use."""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
set before the CUDA context is created.
With CUPY_CUDA_COMPILE_WITH_CACHE=1, JIT-compiled kernels are cached
on disk and shared across processes. We warm up on ALL visible GPUs
so workers that later restrict to a single GPU find pre-compiled kernels.
Per-GPU warm-up errors are caught individually: if GPU 1 fails to
compile a kernel (e.g. CUDA_ERROR_NO_BINARY_FOR_GPU on sm_89), we
record it and continue warming up GPU 0. Workers assigned to a
failed GPU fall back to CPU automatically.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _gpu_failed_ids
if _gpu_initialized: if _gpu_initialized:
return return
_gpu_initialized = True _gpu_initialized = True
_gpu_failed_ids = set()
try: if not HAS_GPU or _best_gpu_id is None:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
n_devs = _real_cupy.cuda.runtime.getDeviceCount()
all_failed = False
# Warm up each GPU independently — one failure doesn't kill the others.
for dev_id in range(n_devs):
try:
with _real_cupy.cuda.Device(dev_id):
props = _real_cupy.cuda.runtime.getDeviceProperties(dev_id)
name = props['name'].decode() if isinstance(props['name'], bytes) else props['name']
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
logger.info(f" GPU {dev_id}: {name} — kernels pré-compilés")
except Exception as dev_err:
_gpu_failed_ids.add(dev_id)
logger.warning(
f" GPU {dev_id}: échec warm-up ({dev_err.__class__.__name__}: "
f"{dev_err}) — workers sur ce GPU passeront en CPU"
)
# If ALL visible GPUs failed, disable GPU entirely.
# This is critical for worker subprocesses that restricted
# CUDA_VISIBLE_DEVICES to a single GPU before importing CuPy.
if len(_gpu_failed_ids) >= n_devs:
all_failed = True
logger.warning("Tous les GPUs échoués au warm-up — passage en mode CPU")
if not all_failed:
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
# Limit GPU memory pool per worker to avoid OOM when multiple
# workers share one GPU. Each worker gets at most 3.5 GB (or
# 50 % of total VRAM on smaller cards).
try:
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem']
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
logger.info(
f" Pool mémoire GPU limité à "
f"{max_worker_mem // (1024**3) * 1000 // 1024} MB"
)
except Exception:
pass # pool config is best-effort
else:
_xp = np _xp = np
_cp = None _cp = None
_cp_ndimage = None _cp_ndimage = None
HAS_GPU = False HAS_GPU = False
return
try:
# Restrict to the selected GPU before CuPy imports
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works on this GPU
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem']
# Memory pool: up to 90% of VRAM (single GPU shared by workers)
pool_size = int(total_mem * 0.9)
_real_cupy.cuda.set_memory_pool(0, pool_size)
except (ImportError, Exception) as e: except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}") logger.warning(f"GPU non disponible — mode CPU: {e}")
@ -147,94 +128,32 @@ def _init_gpu():
HAS_GPU = False HAS_GPU = False
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): # ---------------------------------------------------------------------------
"""Restrict which GPUs are visible to the process. # Public API
# ---------------------------------------------------------------------------
By default (set_env_var=False), only records the GPU IDs for
num_gpus() and logging. Does NOT touch CUDA_VISIBLE_DEVICES
in the main process because CuPy 13.x JIT compilation (sm_89)
needs ALL GPUs visible during warm-up to pre-compile kernels.
Workers call this with set_env_var=True (via assign_gpu_to_worker)
after forking, when CuPy hasn't been imported yet in the child.
Args:
gpu_ids: List of system-level GPU indices to make visible.
set_env_var: If True, actually set CUDA_VISIBLE_DEVICES.
Default False (safe for main process).
"""
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
if not gpu_ids or not HAS_GPU:
return
# Validate IDs against total GPU count from nvidia-smi
total_count = _NUM_GPUS or 1
valid_ids = [gid % total_count for gid in gpu_ids]
_available_gpu_ids = valid_ids
_NUM_GPUS = len(valid_ids)
if set_env_var:
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
logger.info(f"GPU visibles: {_available_gpu_ids}")
def num_gpus(): def num_gpus():
"""Return the number of visible (available) GPUs.""" """Return 1 if GPU is active, 0 otherwise."""
return _NUM_GPUS return 1 if HAS_GPU else 0
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
"""No-op — GPU is auto-selected at import time."""
pass
def set_active_gpu(gpu_id): def set_active_gpu(gpu_id):
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES. """No-op — GPU is auto-selected at import time."""
pass
gpu_id is an index into the currently visible GPU list
(_available_gpu_ids), not a system-level ID.
MUST be called before any GPU operation (to_gpu, etc.) to ensure
CuPy creates its CUDA context on the correct device. With lazy
initialization, CuPy is imported AFTER this call, so it only
sees the assigned GPU.
If the assigned GPU failed warm-up (e.g. NO_BINARY_FOR_GPU), this
function disables GPU for this worker so all operations fall back
to CPU without crashing.
Args:
gpu_id: 0-based index into the visible GPU list.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Map visible-GPU index back to the real system GPU ID
system_gpu_id = _available_gpu_ids[gpu_id]
# If this GPU failed warm-up, disable GPU for this worker
if system_gpu_id in _gpu_failed_ids:
logger.warning(
f" GPU {system_gpu_id} échouée au warm-up — "
f"worker passe en mode CPU"
)
disable_gpu()
return
# Set CUDA_VISIBLE_DEVICES to isolate this worker to one GPU.
# This MUST happen before CuPy is imported (lazy init).
# The JIT kernels were already pre-compiled by the main process
# on all GPUs, so the worker finds them in the cache.
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
def _gpu_available(): def _gpu_available():
"""Check if GPU is usable right now (may fail in forked subprocesses).""" """Check if GPU is usable right now."""
if not HAS_GPU: if not HAS_GPU:
return False return False
try: try:
_init_gpu() _init_gpu()
_cp.cuda.runtime.getDevice() return _cp is not None
return True
except Exception: except Exception:
return False return False
@ -242,34 +161,30 @@ def _gpu_available():
def log_gpu_status(): def log_gpu_status():
"""Log GPU detection result. Called after logging is configured.""" """Log GPU detection result. Called after logging is configured."""
if _gpu_available(): if _gpu_available():
# Get actual device name from CuPy (after init)
try: try:
dev = _cp.cuda.Device()
name = _cp.cuda.runtime.getDeviceProperties(0)['name'] name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes): if isinstance(name, bytes):
name = name.decode() name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3) mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)" gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
except Exception: except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info) logger.info(gpu_info)
else: else:
logger.info("Pas de GPU — mode CPU uniquement") logger.info("Pas de GPU — mode CPU uniquement")
def to_gpu(arr): # ---------------------------------------------------------------------------
"""Send array to GPU if available, otherwise return as float32 numpy. # Array transfer
# ---------------------------------------------------------------------------
Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU def to_gpu(arr):
is unavailable (e.g. in forked subprocess). """Send array to GPU if available, otherwise return as float32 numpy."""
"""
if _gpu_available(): if _gpu_available():
try: try:
return _cp.asarray(arr.astype(np.float32)) return _cp.asarray(arr.astype(np.float32))
except Exception: except Exception:
pass # Fall back to CPU pass
return arr.astype(np.float32) return arr.astype(np.float32)
@ -279,12 +194,15 @@ def to_cpu(arr):
try: try:
return _cp.asnumpy(arr) return _cp.asnumpy(arr)
except Exception: except Exception:
pass # Already on CPU or GPU error pass
return arr return arr
# ---------------------------------------------------------------------------
# Filters — GPU if array is on GPU, CPU otherwise
# ---------------------------------------------------------------------------
def xp_gaussian_filter(arr, sigma): def xp_gaussian_filter(arr, sigma):
"""Gaussian filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray): if _cp is not None and isinstance(arr, _cp.ndarray):
try: try:
return _cp_ndimage.gaussian_filter(arr, sigma) return _cp_ndimage.gaussian_filter(arr, sigma)
@ -294,7 +212,6 @@ def xp_gaussian_filter(arr, sigma):
def xp_uniform_filter(arr, size): def xp_uniform_filter(arr, size):
"""Uniform filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray): if _cp is not None and isinstance(arr, _cp.ndarray):
try: try:
return _cp_ndimage.uniform_filter(arr, size) return _cp_ndimage.uniform_filter(arr, size)
@ -304,7 +221,6 @@ def xp_uniform_filter(arr, size):
def xp_minimum_filter(arr, footprint=None, size=None): def xp_minimum_filter(arr, footprint=None, size=None):
"""Minimum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray): if _cp is not None and isinstance(arr, _cp.ndarray):
try: try:
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size) return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
@ -314,7 +230,6 @@ def xp_minimum_filter(arr, footprint=None, size=None):
def xp_maximum_filter(arr, footprint=None, size=None): def xp_maximum_filter(arr, footprint=None, size=None):
"""Maximum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray): if _cp is not None and isinstance(arr, _cp.ndarray):
try: try:
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size) return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
@ -323,6 +238,10 @@ def xp_maximum_filter(arr, footprint=None, size=None):
return ndimage.maximum_filter(arr, footprint=footprint, size=size) return ndimage.maximum_filter(arr, footprint=footprint, size=size)
# ---------------------------------------------------------------------------
# Misc
# ---------------------------------------------------------------------------
def gpu_cleanup(): def gpu_cleanup():
"""Free GPU memory. Call between visualizations to prevent OOM.""" """Free GPU memory. Call between visualizations to prevent OOM."""
if _cp is not None: if _cp is not None:
@ -333,15 +252,10 @@ def gpu_cleanup():
def disable_gpu(): def disable_gpu():
"""Disable GPU acceleration for the rest of this process. """Disable GPU acceleration for the rest of this process."""
Called when a CUDA error indicates the GPU is unusable (e.g.
CUDA_ERROR_NO_BINARY_FOR_GPU). Falls back to numpy for all
subsequent operations.
"""
global HAS_GPU, _xp, _cp, _cp_ndimage global HAS_GPU, _xp, _cp, _cp_ndimage
if not HAS_GPU: if not HAS_GPU:
return # Already disabled return
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus") logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
HAS_GPU = False HAS_GPU = False
_xp = np _xp = np
@ -350,21 +264,12 @@ def disable_gpu():
def is_gpu_active(): def is_gpu_active():
"""Check if GPU acceleration is currently active. """Check if GPU acceleration is currently active."""
Unlike the HAS_GPU module-level variable (which can go stale if
imported directly), this always reflects the current runtime state.
Use this in logging tags and conditional GPU paths.
"""
return HAS_GPU return HAS_GPU
def safe_gpu_call(func, *args, **kwargs): def safe_gpu_call(func, *args, **kwargs):
"""Call a function with GPU arrays, retrying on CPU if GPU fails. """Call a function with GPU arrays, retrying on CPU if GPU fails."""
Usage:
result = safe_gpu_call(generate_svf, dem_file, basename, vis_dir, resolution, shared=shared)
"""
try: try:
return func(*args, **kwargs) return func(*args, **kwargs)
except Exception as e: except Exception as e: