Add multi-GPU support and fix scale bar / location map overlap
Multi-GPU: - gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes effect before context creation in worker processes - gpu.py: detect GPU count via nvidia-smi (no CUDA import needed) - gpu.py: add set_active_gpu() to assign workers to specific GPUs - pipeline.py: distribute files across GPUs (file % num_gpus) in parallel mode so both GPUs are used simultaneously - pipeline.py: log GPU count when multiple GPUs detected Layout fixes: - rendering.py: move scale bar left of location map to avoid overlap (scale bar ends at fig_x=0.78, map starts at 0.82) - rendering.py: expand location map inset to 0.16x0.13 fig coords - rendering.py: return bounds from _download_location_map so imshow extent matches the actual IGN tile coverage (80km context) - ign.py: add min_zoom parameter to download_ign_tiles, fixing the location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
This commit is contained in:
@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU.
|
||||
|
||||
GPU errors (e.g. in forked subprocesses) are caught gracefully and
|
||||
cause an automatic fallback to CPU for the current operation.
|
||||
|
||||
Multi-GPU support: when multiple GPUs are available, each worker process
|
||||
can be assigned a different GPU via set_active_gpu() for balanced load.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import numpy as np
|
||||
from scipy import ndimage
|
||||
|
||||
logger = logging.getLogger("lidar")
|
||||
|
||||
# GPU detection - must happen at import time
|
||||
# Detect GPU count at import time WITHOUT importing CuPy.
|
||||
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
|
||||
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
|
||||
# in worker processes.
|
||||
_NUM_GPUS = 0
|
||||
HAS_GPU = False
|
||||
_gpu_name = None
|
||||
_gpu_mem_gb = 0
|
||||
|
||||
# Check if GPUs are available via nvidia-smi (no CUDA context created)
|
||||
try:
|
||||
import subprocess
|
||||
_result = subprocess.run(
|
||||
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
|
||||
capture_output=True, text=True, timeout=5
|
||||
)
|
||||
if _result.returncode == 0:
|
||||
_lines = _result.stdout.strip().split('\n')
|
||||
_NUM_GPUS = len(_lines)
|
||||
# Parse first GPU info for logging
|
||||
_parts = _lines[0].split(',')
|
||||
if len(_parts) >= 3:
|
||||
_gpu_name = _parts[1].strip()
|
||||
try:
|
||||
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
|
||||
except (ValueError, IndexError):
|
||||
pass
|
||||
HAS_GPU = True
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||
pass
|
||||
|
||||
# Lazy-initialized GPU module references
|
||||
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
|
||||
# to be set before CuPy context creation in worker processes.
|
||||
_xp = np # Default: CPU
|
||||
_cp = None # cupy module (or None)
|
||||
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
|
||||
_gpu_initialized = False
|
||||
|
||||
try:
|
||||
import cupy as _cupy
|
||||
import cupyx.scipy.ndimage as _cupy_ndimage
|
||||
|
||||
_gpu_info = _cupy.cuda.runtime.getDeviceProperties(0)
|
||||
_gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name'])
|
||||
_gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3)
|
||||
HAS_GPU = True
|
||||
_xp = _cupy
|
||||
_cp = _cupy
|
||||
_cp_ndimage = _cupy_ndimage
|
||||
except (ImportError, Exception):
|
||||
pass
|
||||
def _init_gpu():
|
||||
"""Lazily initialize CuPy on first GPU use.
|
||||
|
||||
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
|
||||
before CuPy creates a CUDA context.
|
||||
"""
|
||||
global _xp, _cp, _cp_ndimage, _gpu_initialized
|
||||
if _gpu_initialized:
|
||||
return
|
||||
_gpu_initialized = True
|
||||
try:
|
||||
import cupy as _real_cupy
|
||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||
# Verify GPU is actually accessible
|
||||
_real_cupy.cuda.runtime.getDevice()
|
||||
_xp = _real_cupy
|
||||
_cp = _real_cupy
|
||||
_cp_ndimage = _real_cupy_ndimage
|
||||
except (ImportError, Exception):
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
|
||||
|
||||
def num_gpus():
|
||||
"""Return the number of available CUDA GPUs."""
|
||||
return _NUM_GPUS
|
||||
|
||||
|
||||
def set_active_gpu(gpu_id):
|
||||
"""Set the active GPU for the current process.
|
||||
|
||||
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
|
||||
the CUDA context is created on the correct device.
|
||||
|
||||
Args:
|
||||
gpu_id: 0-based GPU index. Clamped to valid range.
|
||||
"""
|
||||
if not HAS_GPU or _NUM_GPUS <= 1:
|
||||
return # Nothing to do for single GPU or no GPU
|
||||
|
||||
gpu_id = gpu_id % _NUM_GPUS
|
||||
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
|
||||
# This is the most reliable way in spawn processes
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
|
||||
# Reset lazy init so CuPy re-detects with the new env
|
||||
global _gpu_initialized, _cp, _cp_ndimage, _xp
|
||||
_gpu_initialized = False
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
_xp = np
|
||||
|
||||
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
|
||||
|
||||
|
||||
def _gpu_available():
|
||||
@ -42,6 +118,7 @@ def _gpu_available():
|
||||
if not HAS_GPU:
|
||||
return False
|
||||
try:
|
||||
_init_gpu()
|
||||
_cp.cuda.runtime.getDevice()
|
||||
return True
|
||||
except Exception:
|
||||
@ -50,8 +127,11 @@ def _gpu_available():
|
||||
|
||||
def log_gpu_status():
|
||||
"""Log GPU detection result. Called after logging is configured."""
|
||||
if _gpu_available():
|
||||
logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)")
|
||||
if HAS_GPU:
|
||||
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||
if _NUM_GPUS > 1:
|
||||
gpu_info += f" × {_NUM_GPUS}"
|
||||
logger.info(gpu_info)
|
||||
else:
|
||||
logger.info("Pas de GPU — mode CPU uniquement")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user