Add multi-GPU support and fix scale bar / location map overlap

Multi-GPU:
- gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes
  effect before context creation in worker processes
- gpu.py: detect GPU count via nvidia-smi (no CUDA import needed)
- gpu.py: add set_active_gpu() to assign workers to specific GPUs
- pipeline.py: distribute files across GPUs (file % num_gpus) in
  parallel mode so both GPUs are used simultaneously
- pipeline.py: log GPU count when multiple GPUs detected

Layout fixes:
- rendering.py: move scale bar left of location map to avoid overlap
  (scale bar ends at fig_x=0.78, map starts at 0.82)
- rendering.py: expand location map inset to 0.16x0.13 fig coords
- rendering.py: return bounds from _download_location_map so imshow
  extent matches the actual IGN tile coverage (80km context)
- ign.py: add min_zoom parameter to download_ign_tiles, fixing the
  location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
This commit is contained in:
Antoine Jacquin
2026-05-15 12:00:34 +02:00
parent 4848f25326
commit 5af53a390f
3 changed files with 145 additions and 31 deletions

View File

@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation. cause an automatic fallback to CPU for the current operation.
Multi-GPU support: when multiple GPUs are available, each worker process
can be assigned a different GPU via set_active_gpu() for balanced load.
""" """
import logging import logging
import os
import numpy as np import numpy as np
from scipy import ndimage from scipy import ndimage
logger = logging.getLogger("lidar") logger = logging.getLogger("lidar")
# GPU detection - must happen at import time # Detect GPU count at import time WITHOUT importing CuPy.
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
# in worker processes.
_NUM_GPUS = 0
HAS_GPU = False HAS_GPU = False
_gpu_name = None _gpu_name = None
_gpu_mem_gb = 0 _gpu_mem_gb = 0
# Check if GPUs are available via nvidia-smi (no CUDA context created)
try:
import subprocess
_result = subprocess.run(
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5
)
if _result.returncode == 0:
_lines = _result.stdout.strip().split('\n')
_NUM_GPUS = len(_lines)
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
except (ValueError, IndexError):
pass
HAS_GPU = True
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy-initialized GPU module references
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
# to be set before CuPy context creation in worker processes.
_xp = np # Default: CPU _xp = np # Default: CPU
_cp = None # cupy module (or None) _cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None) _cp_ndimage = None # cupyx.scipy.ndimage (or None)
_gpu_initialized = False
try:
import cupy as _cupy
import cupyx.scipy.ndimage as _cupy_ndimage
_gpu_info = _cupy.cuda.runtime.getDeviceProperties(0) def _init_gpu():
_gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name']) """Lazily initialize CuPy on first GPU use.
_gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3)
HAS_GPU = True This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
_xp = _cupy before CuPy creates a CUDA context.
_cp = _cupy """
_cp_ndimage = _cupy_ndimage global _xp, _cp, _cp_ndimage, _gpu_initialized
except (ImportError, Exception): if _gpu_initialized:
pass return
_gpu_initialized = True
try:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice()
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
except (ImportError, Exception):
_xp = np
_cp = None
_cp_ndimage = None
def num_gpus():
"""Return the number of available CUDA GPUs."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process.
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
the CUDA context is created on the correct device.
Args:
gpu_id: 0-based GPU index. Clamped to valid range.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
# This is the most reliable way in spawn processes
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
# Reset lazy init so CuPy re-detects with the new env
global _gpu_initialized, _cp, _cp_ndimage, _xp
_gpu_initialized = False
_cp = None
_cp_ndimage = None
_xp = np
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
def _gpu_available(): def _gpu_available():
@ -42,6 +118,7 @@ def _gpu_available():
if not HAS_GPU: if not HAS_GPU:
return False return False
try: try:
_init_gpu()
_cp.cuda.runtime.getDevice() _cp.cuda.runtime.getDevice()
return True return True
except Exception: except Exception:
@ -50,8 +127,11 @@ def _gpu_available():
def log_gpu_status(): def log_gpu_status():
"""Log GPU detection result. Called after logging is configured.""" """Log GPU detection result. Called after logging is configured."""
if _gpu_available(): if HAS_GPU:
logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)") gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)
else: else:
logger.info("Pas de GPU — mode CPU uniquement") logger.info("Pas de GPU — mode CPU uniquement")

View File

@ -63,7 +63,7 @@ from .visualizations import (
generate_roughness, generate_wavelet, generate_roughness, generate_wavelet,
generate_svf, generate_aniso_open, generate_svf, generate_aniso_open,
) )
from .gpu import gpu_cleanup from .gpu import gpu_cleanup, num_gpus
from .ign import generate_ign_overlay from .ign import generate_ign_overlay
from .rendering import tif_to_png from .rendering import tif_to_png
@ -447,14 +447,20 @@ class LidarArchaeoPipeline:
t_pipeline_start = time.time() t_pipeline_start = time.time()
if self.workers > 1 and len(files) > 1: if self.workers > 1 and len(files) > 1:
logger.info(f"Traitement parallèle avec {self.workers} workers...") n_gpus = num_gpus()
if n_gpus > 1:
logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...")
else:
logger.info(f"Traitement parallèle avec {self.workers} workers...")
logger.info(f"Fichiers: {len(files)}") logger.info(f"Fichiers: {len(files)}")
with ProcessPoolExecutor(max_workers=self.workers) as executor: with ProcessPoolExecutor(max_workers=self.workers) as executor:
# Pass resolutions as comma-separated string for multiprocessing serialization # Pass resolutions as comma-separated string for multiprocessing serialization
resolutions_str = ','.join(str(r) for r in self.resolutions) resolutions_str = ','.join(str(r) for r in self.resolutions)
n_gpus = num_gpus() or 1
future_to_file = { future_to_file = {
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format): laz_file executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file
for laz_file in files for gpu_id, laz_file in enumerate(files)
} }
done = 0 done = 0
try: try:
@ -526,11 +532,19 @@ class LidarArchaeoPipeline:
logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}") logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}")
def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif'): def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif', gpu_id=None):
"""Standalone function for multiprocessing — creates its own pipeline instance. """Standalone function for multiprocessing — creates its own pipeline instance.
Each worker gets its own temp directory to avoid file conflicts. Each worker gets its own temp directory to avoid file conflicts.
When multiple GPUs are available, each worker is assigned a GPU via
CUDA_VISIBLE_DEVICES to balance load across GPUs.
""" """
# Assign GPU FIRST — before any CuPy import happens
# This sets CUDA_VISIBLE_DEVICES and resets lazy CuPy init
if gpu_id is not None and gpu_id >= 0:
from .gpu import set_active_gpu
set_active_gpu(gpu_id)
# Configure logging in worker process (spawn doesn't inherit parent config) # Configure logging in worker process (spawn doesn't inherit parent config)
import logging import logging
import sys import sys

View File

@ -274,15 +274,16 @@ def _apply_colormap(data, tif_file):
def _download_location_map(min_x, max_x, min_y, max_y): def _download_location_map(min_x, max_x, min_y, max_y):
"""Download a wide-area IGN topographic map for location context. """Download a wide-area IGN topographic map for location context.
Downloads a zoomed-out IGN PLANIGNV2 tile covering 5-10x the processed Downloads a zoomed-out IGN PLANIGNV2 tile covering ~80km around the
zone extent, giving a wider geographic context. Results are cached to processed zone, giving wider geographic context. Results are cached to
avoid re-downloading for each visualization in the same tile. avoid re-downloading for each visualization in the same tile.
Args: Args:
min_x, max_x, min_y, max_y: DTM bounds in Lambert 93. min_x, max_x, min_y, max_y: DTM bounds in Lambert 93.
Returns: Returns:
numpy array (H, W, 3) uint8, or None on failure. Tuple (image_array, bounds_dict) where bounds_dict has keys
'min_x', 'max_x', 'min_y', 'max_y' in Lambert 93, or None on failure.
""" """
import hashlib import hashlib
@ -323,7 +324,13 @@ def _download_location_map(min_x, max_x, min_y, max_y):
) )
if result is not None: if result is not None:
_location_map_cache[cache_key] = result bounds = {
'min_x': context_min_x, 'max_x': context_max_x,
'min_y': context_min_y, 'max_y': context_max_y,
}
cached = (result, bounds)
_location_map_cache[cache_key] = cached
return cached
return result return result
except Exception as e: except Exception as e:
@ -637,6 +644,7 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
edgecolor='#cccccc', alpha=0.9)) edgecolor='#cccccc', alpha=0.9))
# Scale bar — adaptive with alternating black/white segments # Scale bar — adaptive with alternating black/white segments
# Position: left of the location map to avoid overlap
extent_m_x = max_x - min_x extent_m_x = max_x - min_x
scale_m, scale_label = _nice_scale(extent_m_x) scale_m, scale_label = _nice_scale(extent_m_x)
pixels_per_meter = 1.0 / pixel_size_x pixels_per_meter = 1.0 / pixel_size_x
@ -646,7 +654,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
bar_bottom_y = 0.55 bar_bottom_y = 0.55
bar_top_y = 0.85 bar_top_y = 0.85
bar_height = bar_top_y - bar_bottom_y bar_height = bar_top_y - bar_bottom_y
scale_start_x = 0.80 # Place scale bar so it ends before the location map (map starts at x=0.82 in fig coords)
# map_ax occupies [0.82, 0.02, 0.16, 0.13] in figure coords
# info_ax occupies [data_left, 0.015, width, 0.09]
# Scale bar end in fig coords = info_ax.left + (scale_start_x + scale_px/width) * info_ax.width
# We need: info_ax.left + (scale_start_x + scale_px/width) * info_ax.width < 0.80
scale_end_frac = scale_px / width # fraction of info_ax width
info_ax_width = data_width_frac + cbar_width + 0.02
# Calculate scale_start_x so scale bar ends at fig_x = 0.78 (leaving gap before map at 0.82)
max_scale_end_fig = 0.78
scale_end_in_info = (max_scale_end_fig - data_left) / info_ax_width
scale_start_x = max(0.05, scale_end_in_info - scale_end_frac)
for seg_i in range(n_segments): for seg_i in range(n_segments):
color = 'black' if seg_i % 2 == 0 else 'white' color = 'black' if seg_i % 2 == 0 else 'white'
@ -666,15 +684,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False) color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False)
# Location inset map — IGN topographic background with processed zone marker # Location inset map — IGN topographic background with processed zone marker
map_ax = fig.add_axes([0.84, 0.02, 0.14, 0.12]) # Positioned in lower-right corner, above the info bar
map_ax = fig.add_axes([0.82, 0.02, 0.16, 0.13])
# Try to download a wide-area IGN topo map for location context # Try to download a wide-area IGN topo map for location context
location_map = _download_location_map(min_x, max_x, min_y, max_y) location_result = _download_location_map(min_x, max_x, min_y, max_y)
if location_map is not None: if location_result is not None:
# Draw IGN topo map as background location_map, loc_bounds = location_result
# Draw IGN topo map as background with correct bounds
map_ax.imshow(location_map, aspect='equal', extent=[ map_ax.imshow(location_map, aspect='equal', extent=[
min_x - (max_x - min_x) * 2, max_x + (max_x - min_x) * 2, loc_bounds['min_x'], loc_bounds['max_x'],
min_y - (max_y - min_y) * 2, max_y + (max_y - min_y) * 2 loc_bounds['min_y'], loc_bounds['max_y']
]) ])
# Mark the processed zone with a red rectangle # Mark the processed zone with a red rectangle
rect_x1, rect_x2 = min_x, max_x rect_x1, rect_x2 = min_x, max_x