Files
lidar_rendu/lidar_pipeline/gpu.py

307 lines
9.7 KiB
Python

"""GPU acceleration helpers for LiDAR pipeline.
Auto-selects the best NVIDIA GPU that works with CuPy.
Tries each card in order of compute capability; falls back to CPU
if none work. All workers share the selected GPU.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# ---------------------------------------------------------------------------
# GPU auto-detection via nvidia-smi (no CUDA context created)
# ---------------------------------------------------------------------------
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
_best_gpu_id: int | None = None
_gpu_reason = None
def _pick_gpu() -> int | None:
"""Pick the best GPU that CuPy can actually use.
Tries each card in order of compute capability (highest first).
Returns the index of the first GPU whose compute capability is
known to work with the installed CuPy version.
"""
try:
import subprocess
result = subprocess.run(
['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total',
'--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5,
)
if result.returncode != 0:
return None
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason
gpus = []
for line in result.stdout.strip().split('\n'):
parts = [p.strip() for p in line.split(',')]
if len(parts) < 4:
continue
idx = int(parts[0])
name = parts[1]
cap_str = parts[2]
mem_mi = int(parts[3])
major, minor = (int(x) for x in cap_str.split('.'))
score = major * 1000 + minor * 100 + mem_mi
gpus.append((idx, name, cap_str, mem_mi, score, major))
if not gpus:
return None
_NUM_GPUS = len(gpus)
gpus.sort(key=lambda g: g[4], reverse=True)
# CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any
# architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable.
for gpu in gpus:
idx, name, cap_str, mem_mi, score, major = gpu
if major <= 8: # sm_89 and below — works with CuPy 13.4 + JIT
_best_gpu_id = idx
_gpu_name = name
_gpu_mem_gb = mem_mi // 1024
HAS_GPU = True
return _best_gpu_id
# All GPUs have unsupported compute capability
_gpu_reason = "aucun GPU avec compute capability compatible CuPy 13.4"
return None
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
return None
_best_gpu_id = _pick_gpu()
# ---------------------------------------------------------------------------
# Lazy CuPy initialization
# ---------------------------------------------------------------------------
_xp = np
_cp = None
_cp_ndimage = None
_gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use."""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
if _gpu_initialized:
return
_gpu_initialized = True
if not HAS_GPU or _best_gpu_id is None:
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
if _gpu_reason:
logger.info(f"Pas de GPU — {_gpu_reason}")
return
try:
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Warm-up: verify kernel execution works
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
_ = _result.get()
del _test, _result
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
# Memory pool management removed — CuPy 13.x uses its own allocator.
# OOM protection handled at the application level (gpu_cleanup() calls).
except (ImportError, Exception) as e:
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def num_gpus():
"""Return 1 if GPU is active, 0 otherwise."""
return 1 if HAS_GPU else 0
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
"""No-op — GPU is auto-selected at import time."""
pass
def set_active_gpu(gpu_id):
"""No-op — GPU is auto-selected at import time."""
pass
def _gpu_available():
"""Check if GPU is usable right now."""
if not HAS_GPU:
return False
try:
_init_gpu()
return _cp is not None
except Exception:
return False
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
try:
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
logger.info(gpu_info)
# Warn about unsupported GPUs that exist but are not used
try:
import subprocess
result = subprocess.run(
['nvidia-smi', '--query-gpu=index,name,compute_cap',
'--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5,
)
if result.returncode == 0:
for line in result.stdout.strip().split('\n'):
parts = [p.strip() for p in line.split(',')]
if len(parts) >= 3:
idx = int(parts[0])
if idx != _best_gpu_id:
cap = parts[2]
logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — non utilisé (incompatible CuPy)")
except Exception:
pass
else:
logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})")
# ---------------------------------------------------------------------------
# Array transfer
# ---------------------------------------------------------------------------
def to_gpu(arr):
"""Send array to GPU if available, otherwise return as float32 numpy."""
if _gpu_available():
try:
return _cp.asarray(arr.astype(np.float32))
except Exception:
pass
return arr.astype(np.float32)
def to_cpu(arr):
"""Bring array back to CPU (numpy). No-op if already on CPU."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp.asnumpy(arr)
except Exception:
pass
return arr
# ---------------------------------------------------------------------------
# Filters — GPU if array is on GPU, CPU otherwise
# ---------------------------------------------------------------------------
def xp_gaussian_filter(arr, sigma):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.gaussian_filter(arr, sigma)
except Exception:
arr = to_cpu(arr)
return ndimage.gaussian_filter(arr, sigma)
def xp_uniform_filter(arr, size):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.uniform_filter(arr, size)
except Exception:
arr = to_cpu(arr)
return ndimage.uniform_filter(arr, size)
def xp_minimum_filter(arr, footprint=None, size=None):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.minimum_filter(arr, footprint=footprint, size=size)
def xp_maximum_filter(arr, footprint=None, size=None):
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
# ---------------------------------------------------------------------------
# Misc
# ---------------------------------------------------------------------------
def gpu_cleanup():
"""Free GPU memory. Call between visualizations to prevent OOM."""
if _cp is not None:
try:
_cp.get_default_memory_pool().free_all_blocks()
except Exception:
pass
def disable_gpu():
"""Disable GPU acceleration for the rest of this process."""
global HAS_GPU, _xp, _cp, _cp_ndimage
if not HAS_GPU:
return
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
HAS_GPU = False
_xp = np
_cp = None
_cp_ndimage = None
def is_gpu_active():
"""Check if GPU acceleration is currently active."""
return HAS_GPU
def safe_gpu_call(func, *args, **kwargs):
"""Call a function with GPU arrays, retrying on CPU if GPU fails."""
try:
return func(*args, **kwargs)
except Exception as e:
err_msg = str(e)
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
disable_gpu()
return func(*args, **kwargs)
raise