Files
lidar_rendu/lidar_pipeline/gpu.py

282 lines
8.4 KiB
Python

"""GPU acceleration helpers for LiDAR pipeline.
Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts
CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back
to CPU if no GPU is available or usable.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# ---------------------------------------------------------------------------
# GPU auto-detection via nvidia-smi (no CUDA context created)
# ---------------------------------------------------------------------------
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
_best_gpu_id: int | None = None
def _pick_gpu() -> int | None:
"""Pick the best GPU from the system.
Preference order:
1. RTX 50xx (Blackwell, compute >= 12.0)
2. RTX 40xx (Ada Lovelace, compute >= 8.9)
3. Any NVIDIA GPU with highest compute capability
"""
try:
import subprocess
result = subprocess.run(
['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total',
'--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5,
)
if result.returncode != 0:
return None
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id
gpus = []
for line in result.stdout.strip().split('\n'):
parts = [p.strip() for p in line.split(',')]
if len(parts) < 4:
continue
idx = int(parts[0])
name = parts[1]
cap_str = parts[2]
mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.'))
# Score: higher compute capability first, then more VRAM
score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score))
if not gpus:
return None
_NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True)
best = gpus[0]
_best_gpu_id = best[0]
_gpu_name = best[1]
_gpu_mem_gb = best[3] // 1024
HAS_GPU = True
return _best_gpu_id
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None
_best_gpu_id = _pick_gpu()
# ---------------------------------------------------------------------------
# Lazy CuPy initialization
# ---------------------------------------------------------------------------
_xp = np
_cp = None
_cp_ndimage = None
_gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use."""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
if _gpu_initialized:
return
_gpu_initialized = True
if not HAS_GPU or _best_gpu_id is None:
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
return
try:
# Restrict to the selected GPU before CuPy imports
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works on this GPU
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem']
# Memory pool: up to 90% of VRAM (single GPU shared by workers)
pool_size = int(total_mem * 0.9)
_real_cupy.cuda.set_memory_pool(0, pool_size)
except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}")
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def num_gpus():
"""Return 1 if GPU is active, 0 otherwise."""
return 1 if HAS_GPU else 0
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
"""No-op — GPU is auto-selected at import time."""
pass
def set_active_gpu(gpu_id):
"""No-op — GPU is auto-selected at import time."""
pass
def _gpu_available():
"""Check if GPU is usable right now."""
if not HAS_GPU:
return False
try:
_init_gpu()
return _cp is not None
except Exception:
return False
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
try:
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
logger.info(gpu_info)
else:
logger.info("Pas de GPU — mode CPU uniquement")
# ---------------------------------------------------------------------------
# Array transfer
# ---------------------------------------------------------------------------
def to_gpu(arr):
"""Send array to GPU if available, otherwise return as float32 numpy."""
if _gpu_available():
try:
return _cp.asarray(arr.astype(np.float32))
except Exception:
pass
return arr.astype(np.float32)
def to_cpu(arr):
"""Bring array back to CPU (numpy). No-op if already on CPU."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp.asnumpy(arr)
except Exception:
pass
return arr
# ---------------------------------------------------------------------------
# Filters — GPU if array is on GPU, CPU otherwise
# ---------------------------------------------------------------------------
def xp_gaussian_filter(arr, sigma):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.gaussian_filter(arr, sigma)
except Exception:
arr = to_cpu(arr)
return ndimage.gaussian_filter(arr, sigma)
def xp_uniform_filter(arr, size):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.uniform_filter(arr, size)
except Exception:
arr = to_cpu(arr)
return ndimage.uniform_filter(arr, size)
def xp_minimum_filter(arr, footprint=None, size=None):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.minimum_filter(arr, footprint=footprint, size=size)
def xp_maximum_filter(arr, footprint=None, size=None):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
# ---------------------------------------------------------------------------
# Misc
# ---------------------------------------------------------------------------
def gpu_cleanup():
"""Free GPU memory. Call between visualizations to prevent OOM."""
if _cp is not None:
try:
_cp.get_default_memory_pool().free_all_blocks()
except Exception:
pass
def disable_gpu():
"""Disable GPU acceleration for the rest of this process."""
global HAS_GPU, _xp, _cp, _cp_ndimage
if not HAS_GPU:
return
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
HAS_GPU = False
_xp = np
_cp = None
_cp_ndimage = None
def is_gpu_active():
"""Check if GPU acceleration is currently active."""
return HAS_GPU
def safe_gpu_call(func, *args, **kwargs):
"""Call a function with GPU arrays, retrying on CPU if GPU fails."""
try:
return func(*args, **kwargs)
except Exception as e:
err_msg = str(e)
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
disable_gpu()
return func(*args, **kwargs)
raise