Add multi-GPU support and fix scale bar / location map overlap

Multi-GPU:
- gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes
  effect before context creation in worker processes
- gpu.py: detect GPU count via nvidia-smi (no CUDA import needed)
- gpu.py: add set_active_gpu() to assign workers to specific GPUs
- pipeline.py: distribute files across GPUs (file % num_gpus) in
  parallel mode so both GPUs are used simultaneously
- pipeline.py: log GPU count when multiple GPUs detected

Layout fixes:
- rendering.py: move scale bar left of location map to avoid overlap
  (scale bar ends at fig_x=0.78, map starts at 0.82)
- rendering.py: expand location map inset to 0.16x0.13 fig coords
- rendering.py: return bounds from _download_location_map so imshow
  extent matches the actual IGN tile coverage (80km context)
- ign.py: add min_zoom parameter to download_ign_tiles, fixing the
  location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
This commit is contained in:
Antoine Jacquin
2026-05-15 12:00:34 +02:00
parent 4848f25326
commit 5af53a390f
3 changed files with 145 additions and 31 deletions

View File

@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: when multiple GPUs are available, each worker process
can be assigned a different GPU via set_active_gpu() for balanced load.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# GPU detection - must happen at import time
# Detect GPU count at import time WITHOUT importing CuPy.
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
# in worker processes.
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
# Check if GPUs are available via nvidia-smi (no CUDA context created)
try:
import subprocess
_result = subprocess.run(
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5
)
if _result.returncode == 0:
_lines = _result.stdout.strip().split('\n')
_NUM_GPUS = len(_lines)
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
except (ValueError, IndexError):
pass
HAS_GPU = True
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy-initialized GPU module references
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
# to be set before CuPy context creation in worker processes.
_xp = np # Default: CPU
_cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
_gpu_initialized = False
try:
import cupy as _cupy
import cupyx.scipy.ndimage as _cupy_ndimage
_gpu_info = _cupy.cuda.runtime.getDeviceProperties(0)
_gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name'])
_gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3)
HAS_GPU = True
_xp = _cupy
_cp = _cupy
_cp_ndimage = _cupy_ndimage
except (ImportError, Exception):
pass
def _init_gpu():
"""Lazily initialize CuPy on first GPU use.
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
before CuPy creates a CUDA context.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized
if _gpu_initialized:
return
_gpu_initialized = True
try:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice()
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
except (ImportError, Exception):
_xp = np
_cp = None
_cp_ndimage = None
def num_gpus():
"""Return the number of available CUDA GPUs."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process.
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
the CUDA context is created on the correct device.
Args:
gpu_id: 0-based GPU index. Clamped to valid range.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
# This is the most reliable way in spawn processes
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
# Reset lazy init so CuPy re-detects with the new env
global _gpu_initialized, _cp, _cp_ndimage, _xp
_gpu_initialized = False
_cp = None
_cp_ndimage = None
_xp = np
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
def _gpu_available():
@ -42,6 +118,7 @@ def _gpu_available():
if not HAS_GPU:
return False
try:
_init_gpu()
_cp.cuda.runtime.getDevice()
return True
except Exception:
@ -50,8 +127,11 @@ def _gpu_available():
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)")
if HAS_GPU:
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)
else:
logger.info("Pas de GPU — mode CPU uniquement")