Add multi-GPU support and fix scale bar / location map overlap

Multi-GPU:
- gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes
  effect before context creation in worker processes
- gpu.py: detect GPU count via nvidia-smi (no CUDA import needed)
- gpu.py: add set_active_gpu() to assign workers to specific GPUs
- pipeline.py: distribute files across GPUs (file % num_gpus) in
  parallel mode so both GPUs are used simultaneously
- pipeline.py: log GPU count when multiple GPUs detected

Layout fixes:
- rendering.py: move scale bar left of location map to avoid overlap
  (scale bar ends at fig_x=0.78, map starts at 0.82)
- rendering.py: expand location map inset to 0.16x0.13 fig coords
- rendering.py: return bounds from _download_location_map so imshow
  extent matches the actual IGN tile coverage (80km context)
- ign.py: add min_zoom parameter to download_ign_tiles, fixing the
  location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
This commit is contained in:
Antoine Jacquin
2026-05-15 12:00:34 +02:00
parent 4848f25326
commit 5af53a390f
3 changed files with 145 additions and 31 deletions

View File

@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU.
GPU errors (e.g. in forked subprocesses) are caught gracefully and
cause an automatic fallback to CPU for the current operation.
Multi-GPU support: when multiple GPUs are available, each worker process
can be assigned a different GPU via set_active_gpu() for balanced load.
"""
import logging
import os
import numpy as np
from scipy import ndimage
logger = logging.getLogger("lidar")
# GPU detection - must happen at import time
# Detect GPU count at import time WITHOUT importing CuPy.
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
# in worker processes.
_NUM_GPUS = 0
HAS_GPU = False
_gpu_name = None
_gpu_mem_gb = 0
# Check if GPUs are available via nvidia-smi (no CUDA context created)
try:
import subprocess
_result = subprocess.run(
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
capture_output=True, text=True, timeout=5
)
if _result.returncode == 0:
_lines = _result.stdout.strip().split('\n')
_NUM_GPUS = len(_lines)
# Parse first GPU info for logging
_parts = _lines[0].split(',')
if len(_parts) >= 3:
_gpu_name = _parts[1].strip()
try:
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
except (ValueError, IndexError):
pass
HAS_GPU = True
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
pass
# Lazy-initialized GPU module references
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
# to be set before CuPy context creation in worker processes.
_xp = np # Default: CPU
_cp = None # cupy module (or None)
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
_gpu_initialized = False
try:
import cupy as _cupy
import cupyx.scipy.ndimage as _cupy_ndimage
_gpu_info = _cupy.cuda.runtime.getDeviceProperties(0)
_gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name'])
_gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3)
HAS_GPU = True
_xp = _cupy
_cp = _cupy
_cp_ndimage = _cupy_ndimage
except (ImportError, Exception):
pass
def _init_gpu():
"""Lazily initialize CuPy on first GPU use.
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
before CuPy creates a CUDA context.
"""
global _xp, _cp, _cp_ndimage, _gpu_initialized
if _gpu_initialized:
return
_gpu_initialized = True
try:
import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice()
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
except (ImportError, Exception):
_xp = np
_cp = None
_cp_ndimage = None
def num_gpus():
"""Return the number of available CUDA GPUs."""
return _NUM_GPUS
def set_active_gpu(gpu_id):
"""Set the active GPU for the current process.
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
the CUDA context is created on the correct device.
Args:
gpu_id: 0-based GPU index. Clamped to valid range.
"""
if not HAS_GPU or _NUM_GPUS <= 1:
return # Nothing to do for single GPU or no GPU
gpu_id = gpu_id % _NUM_GPUS
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
# This is the most reliable way in spawn processes
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
# Reset lazy init so CuPy re-detects with the new env
global _gpu_initialized, _cp, _cp_ndimage, _xp
_gpu_initialized = False
_cp = None
_cp_ndimage = None
_xp = np
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
def _gpu_available():
@ -42,6 +118,7 @@ def _gpu_available():
if not HAS_GPU:
return False
try:
_init_gpu()
_cp.cuda.runtime.getDevice()
return True
except Exception:
@ -50,8 +127,11 @@ def _gpu_available():
def log_gpu_status():
"""Log GPU detection result. Called after logging is configured."""
if _gpu_available():
logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)")
if HAS_GPU:
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
if _NUM_GPUS > 1:
gpu_info += f" × {_NUM_GPUS}"
logger.info(gpu_info)
else:
logger.info("Pas de GPU — mode CPU uniquement")

View File

@ -63,7 +63,7 @@ from .visualizations import (
generate_roughness, generate_wavelet,
generate_svf, generate_aniso_open,
)
from .gpu import gpu_cleanup
from .gpu import gpu_cleanup, num_gpus
from .ign import generate_ign_overlay
from .rendering import tif_to_png
@ -447,14 +447,20 @@ class LidarArchaeoPipeline:
t_pipeline_start = time.time()
if self.workers > 1 and len(files) > 1:
n_gpus = num_gpus()
if n_gpus > 1:
logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...")
else:
logger.info(f"Traitement parallèle avec {self.workers} workers...")
logger.info(f"Fichiers: {len(files)}")
with ProcessPoolExecutor(max_workers=self.workers) as executor:
# Pass resolutions as comma-separated string for multiprocessing serialization
resolutions_str = ','.join(str(r) for r in self.resolutions)
n_gpus = num_gpus() or 1
future_to_file = {
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format): laz_file
for laz_file in files
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file
for gpu_id, laz_file in enumerate(files)
}
done = 0
try:
@ -526,11 +532,19 @@ class LidarArchaeoPipeline:
logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}")
def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif'):
def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif', gpu_id=None):
"""Standalone function for multiprocessing — creates its own pipeline instance.
Each worker gets its own temp directory to avoid file conflicts.
When multiple GPUs are available, each worker is assigned a GPU via
CUDA_VISIBLE_DEVICES to balance load across GPUs.
"""
# Assign GPU FIRST — before any CuPy import happens
# This sets CUDA_VISIBLE_DEVICES and resets lazy CuPy init
if gpu_id is not None and gpu_id >= 0:
from .gpu import set_active_gpu
set_active_gpu(gpu_id)
# Configure logging in worker process (spawn doesn't inherit parent config)
import logging
import sys

View File

@ -274,15 +274,16 @@ def _apply_colormap(data, tif_file):
def _download_location_map(min_x, max_x, min_y, max_y):
"""Download a wide-area IGN topographic map for location context.
Downloads a zoomed-out IGN PLANIGNV2 tile covering 5-10x the processed
zone extent, giving a wider geographic context. Results are cached to
Downloads a zoomed-out IGN PLANIGNV2 tile covering ~80km around the
processed zone, giving wider geographic context. Results are cached to
avoid re-downloading for each visualization in the same tile.
Args:
min_x, max_x, min_y, max_y: DTM bounds in Lambert 93.
Returns:
numpy array (H, W, 3) uint8, or None on failure.
Tuple (image_array, bounds_dict) where bounds_dict has keys
'min_x', 'max_x', 'min_y', 'max_y' in Lambert 93, or None on failure.
"""
import hashlib
@ -323,7 +324,13 @@ def _download_location_map(min_x, max_x, min_y, max_y):
)
if result is not None:
_location_map_cache[cache_key] = result
bounds = {
'min_x': context_min_x, 'max_x': context_max_x,
'min_y': context_min_y, 'max_y': context_max_y,
}
cached = (result, bounds)
_location_map_cache[cache_key] = cached
return cached
return result
except Exception as e:
@ -637,6 +644,7 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
edgecolor='#cccccc', alpha=0.9))
# Scale bar — adaptive with alternating black/white segments
# Position: left of the location map to avoid overlap
extent_m_x = max_x - min_x
scale_m, scale_label = _nice_scale(extent_m_x)
pixels_per_meter = 1.0 / pixel_size_x
@ -646,7 +654,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
bar_bottom_y = 0.55
bar_top_y = 0.85
bar_height = bar_top_y - bar_bottom_y
scale_start_x = 0.80
# Place scale bar so it ends before the location map (map starts at x=0.82 in fig coords)
# map_ax occupies [0.82, 0.02, 0.16, 0.13] in figure coords
# info_ax occupies [data_left, 0.015, width, 0.09]
# Scale bar end in fig coords = info_ax.left + (scale_start_x + scale_px/width) * info_ax.width
# We need: info_ax.left + (scale_start_x + scale_px/width) * info_ax.width < 0.80
scale_end_frac = scale_px / width # fraction of info_ax width
info_ax_width = data_width_frac + cbar_width + 0.02
# Calculate scale_start_x so scale bar ends at fig_x = 0.78 (leaving gap before map at 0.82)
max_scale_end_fig = 0.78
scale_end_in_info = (max_scale_end_fig - data_left) / info_ax_width
scale_start_x = max(0.05, scale_end_in_info - scale_end_frac)
for seg_i in range(n_segments):
color = 'black' if seg_i % 2 == 0 else 'white'
@ -666,15 +684,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False)
# Location inset map — IGN topographic background with processed zone marker
map_ax = fig.add_axes([0.84, 0.02, 0.14, 0.12])
# Positioned in lower-right corner, above the info bar
map_ax = fig.add_axes([0.82, 0.02, 0.16, 0.13])
# Try to download a wide-area IGN topo map for location context
location_map = _download_location_map(min_x, max_x, min_y, max_y)
if location_map is not None:
# Draw IGN topo map as background
location_result = _download_location_map(min_x, max_x, min_y, max_y)
if location_result is not None:
location_map, loc_bounds = location_result
# Draw IGN topo map as background with correct bounds
map_ax.imshow(location_map, aspect='equal', extent=[
min_x - (max_x - min_x) * 2, max_x + (max_x - min_x) * 2,
min_y - (max_y - min_y) * 2, max_y + (max_y - min_y) * 2
loc_bounds['min_x'], loc_bounds['max_x'],
loc_bounds['min_y'], loc_bounds['max_y']
])
# Mark the processed zone with a red rectangle
rect_x1, rect_x2 = min_x, max_x