Files
lidar_rendu/lidar_pipeline/gpu.py

313 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""GPU acceleration helpers for LiDAR pipeline.
Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU
is detected, array operations are accelerated on the GPU. Otherwise, all
operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before
CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel
cache incompatibilities that occur with Device.use() switching.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# Detect total GPU count via nvidia-smi (no CUDA context created).
# This must happen before any CUDA_VISIBLE_DEVICES manipulation.
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
# System-level GPU IDs that are currently visible (after restrict_gpus).
# Populated by restrict_gpus() or auto-detected at import time.
_available_gpu_ids: list[int] = []
try:
import subprocess
_result = subprocess.run(
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5
)
if _result.returncode == 0:
_lines = _result.stdout.strip().split('\n')
_NUM_GPUS = len(_lines)
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
except (ValueError, IndexError):
pass
HAS_GPU = True
# All GPUs are visible by default
_available_gpu_ids = list(range(_NUM_GPUS))
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy CuPy initialization — imported only when first needed.
# This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates
# a CUDA context, enabling per-process GPU assignment.
_xp = np # Default: CPU
_cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
_gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use.
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
set before the CUDA context is created.
Validates that the GPU can actually execute kernels by running
a small computation. This catches CUDA_ERROR_NO_BINARY_FOR_GPU
and other compute capability mismatches before they crash visualizations.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
if _gpu_initialized:
return
_gpu_initialized = True
try:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice()
# Warm-up: run a small computation to verify kernel execution works.
# This catches CUDA_ERROR_NO_BINARY_FOR_GPU (compute capability
# mismatch) and driver errors before we commit to GPU mode.
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test)
# Force execution (CuPy is lazy — .get() ensures the kernel ran)
_ = _result.get()
del _test, _result
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
# Limit GPU memory pool per worker to avoid OOM when multiple
# workers share one GPU. Each worker gets at most 3.5 GB (or
# 50 % of total VRAM on smaller cards).
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem']
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
logger.info(f" Pool mémoire GPU limité à {max_worker_mem // (1024**3) * 1000 // 1024} MB")
except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}")
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
def restrict_gpus(gpu_ids: list[int]):
"""Restrict which GPUs are visible to the process.
Sets CUDA_VISIBLE_DEVICES so only the specified system GPU IDs
are accessible. Also updates _available_gpu_ids and _NUM_GPUS.
Must be called before any GPU operation.
Args:
gpu_ids: List of system-level GPU indices to make visible.
"""
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
if not gpu_ids or not HAS_GPU:
return
# Validate IDs against total GPU count from nvidia-smi
total_count = _NUM_GPUS or 1
valid_ids = [gid % total_count for gid in gpu_ids]
_available_gpu_ids = valid_ids
_NUM_GPUS = len(valid_ids)
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
logger.info(f"GPU visibles: {_available_gpu_ids}")
def num_gpus():
"""Return the number of visible (available) GPUs."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES.
gpu_id is an index into the currently visible GPU list
(_available_gpu_ids), not a system-level ID.
MUST be called before any GPU operation (to_gpu, etc.) to ensure
CuPy creates its CUDA context on the correct device. With lazy
initialization, CuPy is imported AFTER this call, so it only
sees the assigned GPU.
Args:
gpu_id: 0-based index into the visible GPU list.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Map visible-GPU index back to the real system GPU ID
system_gpu_id = _available_gpu_ids[gpu_id]
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
def _gpu_available():
"""Check if GPU is usable right now (may fail in forked subprocesses)."""
if not HAS_GPU:
return False
try:
_init_gpu()
_cp.cuda.runtime.getDevice()
return True
except Exception:
return False
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
# Get actual device name from CuPy (after init)
try:
dev = _cp.cuda.Device()
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
if isinstance(name, bytes):
name = name.decode()
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3)
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)"
except Exception:
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)
else:
logger.info("Pas de GPU — mode CPU uniquement")
def to_gpu(arr):
"""Send array to GPU if available, otherwise return as float32 numpy.
Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU
is unavailable (e.g. in forked subprocess).
"""
if _gpu_available():
try:
return _cp.asarray(arr.astype(np.float32))
except Exception:
pass # Fall back to CPU
return arr.astype(np.float32)
def to_cpu(arr):
"""Bring array back to CPU (numpy). No-op if already on CPU."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp.asnumpy(arr)
except Exception:
pass # Already on CPU or GPU error
return arr
def xp_gaussian_filter(arr, sigma):
"""Gaussian filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.gaussian_filter(arr, sigma)
except Exception:
arr = to_cpu(arr)
return ndimage.gaussian_filter(arr, sigma)
def xp_uniform_filter(arr, size):
"""Uniform filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.uniform_filter(arr, size)
except Exception:
arr = to_cpu(arr)
return ndimage.uniform_filter(arr, size)
def xp_minimum_filter(arr, footprint=None, size=None):
"""Minimum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.minimum_filter(arr, footprint=footprint, size=size)
def xp_maximum_filter(arr, footprint=None, size=None):
"""Maximum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
def gpu_cleanup():
"""Free GPU memory. Call between visualizations to prevent OOM."""
if _cp is not None:
try:
_cp.get_default_memory_pool().free_all_blocks()
except Exception:
pass
def disable_gpu():
"""Disable GPU acceleration for the rest of this process.
Called when a CUDA error indicates the GPU is unusable (e.g.
CUDA_ERROR_NO_BINARY_FOR_GPU). Falls back to numpy for all
subsequent operations.
"""
global HAS_GPU, _xp, _cp, _cp_ndimage
if not HAS_GPU:
return # Already disabled
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
HAS_GPU = False
_xp = np
_cp = None
_cp_ndimage = None
def is_gpu_active():
"""Check if GPU acceleration is currently active.
Unlike the HAS_GPU module-level variable (which can go stale if
imported directly), this always reflects the current runtime state.
Use this in logging tags and conditional GPU paths.
"""
return HAS_GPU
def safe_gpu_call(func, *args, **kwargs):
"""Call a function with GPU arrays, retrying on CPU if GPU fails.
Usage:
result = safe_gpu_call(generate_svf, dem_file, basename, vis_dir, resolution, shared=shared)
"""
try:
return func(*args, **kwargs)
except Exception as e:
err_msg = str(e)
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
disable_gpu()
return func(*args, **kwargs)
raise