Files
lidar_rendu/lidar_pipeline/gpu.py
Antoine Jacquin 5af53a390f Add multi-GPU support and fix scale bar / location map overlap
Multi-GPU:
- gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes
  effect before context creation in worker processes
- gpu.py: detect GPU count via nvidia-smi (no CUDA import needed)
- gpu.py: add set_active_gpu() to assign workers to specific GPUs
- pipeline.py: distribute files across GPUs (file % num_gpus) in
  parallel mode so both GPUs are used simultaneously
- pipeline.py: log GPU count when multiple GPUs detected

Layout fixes:
- rendering.py: move scale bar left of location map to avoid overlap
  (scale bar ends at fig_x=0.78, map starts at 0.82)
- rendering.py: expand location map inset to 0.16x0.13 fig coords
- rendering.py: return bounds from _download_location_map so imshow
  extent matches the actual IGN tile coverage (80km context)
- ign.py: add min_zoom parameter to download_ign_tiles, fixing the
  location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
2026-05-15 12:00:34 +02:00

209 lines
6.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""GPU acceleration helpers for LiDAR pipeline.
Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU
is detected, array operations are accelerated on the GPU. Otherwise, all
operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: when multiple GPUs are available, each worker process
can be assigned a different GPU via set_active_gpu() for balanced load.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# Detect GPU count at import time WITHOUT importing CuPy.
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
# in worker processes.
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
# Check if GPUs are available via nvidia-smi (no CUDA context created)
try:
import subprocess
_result = subprocess.run(
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5
)
if _result.returncode == 0:
_lines = _result.stdout.strip().split('\n')
_NUM_GPUS = len(_lines)
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
except (ValueError, IndexError):
pass
HAS_GPU = True
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy-initialized GPU module references
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
# to be set before CuPy context creation in worker processes.
_xp = np # Default: CPU
_cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
_gpu_initialized = False
def _init_gpu():
"""Lazily initialize CuPy on first GPU use.
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
before CuPy creates a CUDA context.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized
if _gpu_initialized:
return
_gpu_initialized = True
try:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice()
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
except (ImportError, Exception):
_xp = np
_cp = None
_cp_ndimage = None
def num_gpus():
"""Return the number of available CUDA GPUs."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process.
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
the CUDA context is created on the correct device.
Args:
gpu_id: 0-based GPU index. Clamped to valid range.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
# This is the most reliable way in spawn processes
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
# Reset lazy init so CuPy re-detects with the new env
global _gpu_initialized, _cp, _cp_ndimage, _xp
_gpu_initialized = False
_cp = None
_cp_ndimage = None
_xp = np
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
def _gpu_available():
"""Check if GPU is usable right now (may fail in forked subprocesses)."""
if not HAS_GPU:
return False
try:
_init_gpu()
_cp.cuda.runtime.getDevice()
return True
except Exception:
return False
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if HAS_GPU:
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)
else:
logger.info("Pas de GPU — mode CPU uniquement")
def to_gpu(arr):
"""Send array to GPU if available, otherwise return as float32 numpy.
Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU
is unavailable (e.g. in forked subprocess).
"""
if _gpu_available():
try:
return _cp.asarray(arr.astype(np.float32))
except Exception:
pass # Fall back to CPU
return arr.astype(np.float32)
def to_cpu(arr):
"""Bring array back to CPU (numpy). No-op if already on CPU."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp.asnumpy(arr)
except Exception:
pass # Already on CPU or GPU error
return arr
def xp_gaussian_filter(arr, sigma):
"""Gaussian filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.gaussian_filter(arr, sigma)
except Exception:
arr = to_cpu(arr)
return ndimage.gaussian_filter(arr, sigma)
def xp_uniform_filter(arr, size):
"""Uniform filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.uniform_filter(arr, size)
except Exception:
arr = to_cpu(arr)
return ndimage.uniform_filter(arr, size)
def xp_minimum_filter(arr, footprint=None, size=None):
"""Minimum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.minimum_filter(arr, footprint=footprint, size=size)
def xp_maximum_filter(arr, footprint=None, size=None):
"""Maximum filter — uses GPU if array is on GPU, CPU otherwise."""
if _cp is not None and isinstance(arr, _cp.ndarray):
try:
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
except Exception:
arr = to_cpu(arr)
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
def gpu_cleanup():
"""Free GPU memory. Call between visualizations to prevent OOM."""
if _cp is not None:
try:
_cp.get_default_memory_pool().free_all_blocks()
except Exception:
pass