307 lines
9.7 KiB
Python
307 lines
9.7 KiB
Python
"""GPU acceleration helpers for LiDAR pipeline.
|
|
|
|
Auto-selects the best NVIDIA GPU that works with CuPy.
|
|
Tries each card in order of compute capability; falls back to CPU
|
|
if none work. All workers share the selected GPU.
|
|
"""
|
|
|
|
import logging
|
|
import os
|
|
import numpy as np
|
|
from scipy import ndimage
|
|
|
|
logger = logging.getLogger("lidar")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# GPU auto-detection via nvidia-smi (no CUDA context created)
|
|
# ---------------------------------------------------------------------------
|
|
_NUM_GPUS = 0
|
|
HAS_GPU = False
|
|
_gpu_name = None
|
|
_gpu_mem_gb = 0
|
|
_best_gpu_id: int | None = None
|
|
_gpu_reason = None
|
|
|
|
|
|
def _pick_gpu() -> int | None:
|
|
"""Pick the best GPU that CuPy can actually use.
|
|
|
|
Tries each card in order of compute capability (highest first).
|
|
Returns the index of the first GPU whose compute capability is
|
|
known to work with the installed CuPy version.
|
|
"""
|
|
try:
|
|
import subprocess
|
|
result = subprocess.run(
|
|
['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total',
|
|
'--format=csv,noheader,nounits'],
|
|
capture_output=True, text=True, timeout=5,
|
|
)
|
|
if result.returncode != 0:
|
|
return None
|
|
|
|
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id, _gpu_reason
|
|
|
|
gpus = []
|
|
for line in result.stdout.strip().split('\n'):
|
|
parts = [p.strip() for p in line.split(',')]
|
|
if len(parts) < 4:
|
|
continue
|
|
idx = int(parts[0])
|
|
name = parts[1]
|
|
cap_str = parts[2]
|
|
mem_mi = int(parts[3])
|
|
major, minor = (int(x) for x in cap_str.split('.'))
|
|
score = major * 1000 + minor * 100 + mem_mi
|
|
gpus.append((idx, name, cap_str, mem_mi, score, major))
|
|
|
|
if not gpus:
|
|
return None
|
|
|
|
_NUM_GPUS = len(gpus)
|
|
gpus.sort(key=lambda g: g[4], reverse=True)
|
|
|
|
# CuPy 13.4 + CUDA 11.8 JIT compiles kernels at runtime for any
|
|
# architecture (sm_89, sm_120, etc.). All NVIDIA GPUs are usable.
|
|
for gpu in gpus:
|
|
idx, name, cap_str, mem_mi, score, major = gpu
|
|
if major <= 8: # sm_89 and below — works with CuPy 13.4 + JIT
|
|
_best_gpu_id = idx
|
|
_gpu_name = name
|
|
_gpu_mem_gb = mem_mi // 1024
|
|
HAS_GPU = True
|
|
return _best_gpu_id
|
|
|
|
# All GPUs have unsupported compute capability
|
|
_gpu_reason = "aucun GPU avec compute capability compatible CuPy 13.4"
|
|
return None
|
|
|
|
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
|
return None
|
|
|
|
|
|
_best_gpu_id = _pick_gpu()
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Lazy CuPy initialization
|
|
# ---------------------------------------------------------------------------
|
|
_xp = np
|
|
_cp = None
|
|
_cp_ndimage = None
|
|
_gpu_initialized = False
|
|
|
|
|
|
def _init_gpu():
|
|
"""Lazily initialize CuPy on first GPU use."""
|
|
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
|
if _gpu_initialized:
|
|
return
|
|
_gpu_initialized = True
|
|
|
|
if not HAS_GPU or _best_gpu_id is None:
|
|
_xp = np
|
|
_cp = None
|
|
_cp_ndimage = None
|
|
HAS_GPU = False
|
|
if _gpu_reason:
|
|
logger.info(f"Pas de GPU — {_gpu_reason}")
|
|
return
|
|
|
|
try:
|
|
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
|
|
|
|
import cupy as _real_cupy
|
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
|
|
|
# Warm-up: verify kernel execution works
|
|
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
|
_result = _real_cupy.sum(_test * _test)
|
|
_ = _result.get()
|
|
del _test, _result
|
|
|
|
_xp = _real_cupy
|
|
_cp = _real_cupy
|
|
_cp_ndimage = _real_cupy_ndimage
|
|
|
|
# Memory pool management removed — CuPy 13.x uses its own allocator.
|
|
# OOM protection handled at the application level (gpu_cleanup() calls).
|
|
|
|
except (ImportError, Exception) as e:
|
|
logger.warning(f"GPU {_best_gpu_id} ({_gpu_name}) non utilisable — mode CPU: {e}")
|
|
_xp = np
|
|
_cp = None
|
|
_cp_ndimage = None
|
|
HAS_GPU = False
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Public API
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def num_gpus():
|
|
"""Return 1 if GPU is active, 0 otherwise."""
|
|
return 1 if HAS_GPU else 0
|
|
|
|
|
|
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
|
"""No-op — GPU is auto-selected at import time."""
|
|
pass
|
|
|
|
|
|
def set_active_gpu(gpu_id):
|
|
"""No-op — GPU is auto-selected at import time."""
|
|
pass
|
|
|
|
|
|
def _gpu_available():
|
|
"""Check if GPU is usable right now."""
|
|
if not HAS_GPU:
|
|
return False
|
|
try:
|
|
_init_gpu()
|
|
return _cp is not None
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def log_gpu_status():
|
|
"""Log GPU detection result. Called after logging is configured."""
|
|
if _gpu_available():
|
|
try:
|
|
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
|
|
if isinstance(name, bytes):
|
|
name = name.decode()
|
|
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
|
|
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
|
except Exception:
|
|
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
|
logger.info(gpu_info)
|
|
|
|
# Warn about unsupported GPUs that exist but are not used
|
|
try:
|
|
import subprocess
|
|
result = subprocess.run(
|
|
['nvidia-smi', '--query-gpu=index,name,compute_cap',
|
|
'--format=csv,noheader,nounits'],
|
|
capture_output=True, text=True, timeout=5,
|
|
)
|
|
if result.returncode == 0:
|
|
for line in result.stdout.strip().split('\n'):
|
|
parts = [p.strip() for p in line.split(',')]
|
|
if len(parts) >= 3:
|
|
idx = int(parts[0])
|
|
if idx != _best_gpu_id:
|
|
cap = parts[2]
|
|
logger.info(f" GPU {idx}: {parts[1]} (sm_{cap}) — non utilisé (incompatible CuPy)")
|
|
except Exception:
|
|
pass
|
|
else:
|
|
logger.info(f"Pas de GPU utilisable — mode CPU uniquement ({_gpu_reason or 'aucun GPU détecté'})")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Array transfer
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def to_gpu(arr):
|
|
"""Send array to GPU if available, otherwise return as float32 numpy."""
|
|
if _gpu_available():
|
|
try:
|
|
return _cp.asarray(arr.astype(np.float32))
|
|
except Exception:
|
|
pass
|
|
return arr.astype(np.float32)
|
|
|
|
|
|
def to_cpu(arr):
|
|
"""Bring array back to CPU (numpy). No-op if already on CPU."""
|
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
try:
|
|
return _cp.asnumpy(arr)
|
|
except Exception:
|
|
pass
|
|
return arr
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Filters — GPU if array is on GPU, CPU otherwise
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def xp_gaussian_filter(arr, sigma):
|
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
try:
|
|
return _cp_ndimage.gaussian_filter(arr, sigma)
|
|
except Exception:
|
|
arr = to_cpu(arr)
|
|
return ndimage.gaussian_filter(arr, sigma)
|
|
|
|
|
|
def xp_uniform_filter(arr, size):
|
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
try:
|
|
return _cp_ndimage.uniform_filter(arr, size)
|
|
except Exception:
|
|
arr = to_cpu(arr)
|
|
return ndimage.uniform_filter(arr, size)
|
|
|
|
|
|
def xp_minimum_filter(arr, footprint=None, size=None):
|
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
try:
|
|
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
|
|
except Exception:
|
|
arr = to_cpu(arr)
|
|
return ndimage.minimum_filter(arr, footprint=footprint, size=size)
|
|
|
|
|
|
def xp_maximum_filter(arr, footprint=None, size=None):
|
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
try:
|
|
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
|
except Exception:
|
|
arr = to_cpu(arr)
|
|
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Misc
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def gpu_cleanup():
|
|
"""Free GPU memory. Call between visualizations to prevent OOM."""
|
|
if _cp is not None:
|
|
try:
|
|
_cp.get_default_memory_pool().free_all_blocks()
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def disable_gpu():
|
|
"""Disable GPU acceleration for the rest of this process."""
|
|
global HAS_GPU, _xp, _cp, _cp_ndimage
|
|
if not HAS_GPU:
|
|
return
|
|
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
|
HAS_GPU = False
|
|
_xp = np
|
|
_cp = None
|
|
_cp_ndimage = None
|
|
|
|
|
|
def is_gpu_active():
|
|
"""Check if GPU acceleration is currently active."""
|
|
return HAS_GPU
|
|
|
|
|
|
def safe_gpu_call(func, *args, **kwargs):
|
|
"""Call a function with GPU arrays, retrying on CPU if GPU fails."""
|
|
try:
|
|
return func(*args, **kwargs)
|
|
except Exception as e:
|
|
err_msg = str(e)
|
|
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
|
|
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
|
|
disable_gpu()
|
|
return func(*args, **kwargs)
|
|
raise
|