From 5af53a390fbb575674404a2d8677ff8ad8cc4eb3 Mon Sep 17 00:00:00 2001 From: Antoine Jacquin Date: Fri, 15 May 2026 12:00:34 +0200 Subject: [PATCH] Add multi-GPU support and fix scale bar / location map overlap Multi-GPU: - gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes effect before context creation in worker processes - gpu.py: detect GPU count via nvidia-smi (no CUDA import needed) - gpu.py: add set_active_gpu() to assign workers to specific GPUs - pipeline.py: distribute files across GPUs (file % num_gpus) in parallel mode so both GPUs are used simultaneously - pipeline.py: log GPU count when multiple GPUs detected Layout fixes: - rendering.py: move scale bar left of location map to avoid overlap (scale bar ends at fig_x=0.78, map starts at 0.82) - rendering.py: expand location map inset to 0.16x0.13 fig coords - rendering.py: return bounds from _download_location_map so imshow extent matches the actual IGN tile coverage (80km context) - ign.py: add min_zoom parameter to download_ign_tiles, fixing the location map that was broken (zoom 10 blocked by hardcoded min_zoom=15) --- lidar_pipeline/gpu.py | 110 +++++++++++++++++++++++++++++++----- lidar_pipeline/pipeline.py | 24 ++++++-- lidar_pipeline/rendering.py | 42 ++++++++++---- 3 files changed, 145 insertions(+), 31 deletions(-) diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index f4f357f..b1b0ec3 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU. GPU errors (e.g. in forked subprocesses) are caught gracefully and cause an automatic fallback to CPU for the current operation. + +Multi-GPU support: when multiple GPUs are available, each worker process +can be assigned a different GPU via set_active_gpu() for balanced load. """ import logging +import os import numpy as np from scipy import ndimage logger = logging.getLogger("lidar") -# GPU detection - must happen at import time +# Detect GPU count at import time WITHOUT importing CuPy. +# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs, +# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation +# in worker processes. +_NUM_GPUS = 0 HAS_GPU = False _gpu_name = None _gpu_mem_gb = 0 + +# Check if GPUs are available via nvidia-smi (no CUDA context created) +try: + import subprocess + _result = subprocess.run( + ['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'], + capture_output=True, text=True, timeout=5 + ) + if _result.returncode == 0: + _lines = _result.stdout.strip().split('\n') + _NUM_GPUS = len(_lines) + # Parse first GPU info for logging + _parts = _lines[0].split(',') + if len(_parts) >= 3: + _gpu_name = _parts[1].strip() + try: + _gpu_mem_gb = int(float(_parts[2].strip())) // 1024 + except (ValueError, IndexError): + pass + HAS_GPU = True +except (FileNotFoundError, subprocess.TimeoutExpired, Exception): + pass + +# Lazy-initialized GPU module references +# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES +# to be set before CuPy context creation in worker processes. _xp = np # Default: CPU _cp = None # cupy module (or None) _cp_ndimage = None # cupyx.scipy.ndimage (or None) +_gpu_initialized = False -try: - import cupy as _cupy - import cupyx.scipy.ndimage as _cupy_ndimage - _gpu_info = _cupy.cuda.runtime.getDeviceProperties(0) - _gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name']) - _gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3) - HAS_GPU = True - _xp = _cupy - _cp = _cupy - _cp_ndimage = _cupy_ndimage -except (ImportError, Exception): - pass +def _init_gpu(): + """Lazily initialize CuPy on first GPU use. + + This allows CUDA_VISIBLE_DEVICES to take effect in worker processes + before CuPy creates a CUDA context. + """ + global _xp, _cp, _cp_ndimage, _gpu_initialized + if _gpu_initialized: + return + _gpu_initialized = True + try: + import cupy as _real_cupy + import cupyx.scipy.ndimage as _real_cupy_ndimage + # Verify GPU is actually accessible + _real_cupy.cuda.runtime.getDevice() + _xp = _real_cupy + _cp = _real_cupy + _cp_ndimage = _real_cupy_ndimage + except (ImportError, Exception): + _xp = np + _cp = None + _cp_ndimage = None + + +def num_gpus(): + """Return the number of available CUDA GPUs.""" + return _NUM_GPUS + + +def set_active_gpu(gpu_id): + """Set the active GPU for the current process. + + Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure + the CUDA context is created on the correct device. + + Args: + gpu_id: 0-based GPU index. Clamped to valid range. + """ + if not HAS_GPU or _NUM_GPUS <= 1: + return # Nothing to do for single GPU or no GPU + + gpu_id = gpu_id % _NUM_GPUS + # Set CUDA_VISIBLE_DEVICES before CuPy context is created + # This is the most reliable way in spawn processes + os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id) + # Reset lazy init so CuPy re-detects with the new env + global _gpu_initialized, _cp, _cp_ndimage, _xp + _gpu_initialized = False + _cp = None + _cp_ndimage = None + _xp = np + + logger.info(f" GPU {gpu_id} sélectionnée pour ce worker") def _gpu_available(): @@ -42,6 +118,7 @@ def _gpu_available(): if not HAS_GPU: return False try: + _init_gpu() _cp.cuda.runtime.getDevice() return True except Exception: @@ -50,8 +127,11 @@ def _gpu_available(): def log_gpu_status(): """Log GPU detection result. Called after logging is configured.""" - if _gpu_available(): - logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)") + if HAS_GPU: + gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)" + if _NUM_GPUS > 1: + gpu_info += f" × {_NUM_GPUS}" + logger.info(gpu_info) else: logger.info("Pas de GPU — mode CPU uniquement") diff --git a/lidar_pipeline/pipeline.py b/lidar_pipeline/pipeline.py index 17814df..6e3ea38 100644 --- a/lidar_pipeline/pipeline.py +++ b/lidar_pipeline/pipeline.py @@ -63,7 +63,7 @@ from .visualizations import ( generate_roughness, generate_wavelet, generate_svf, generate_aniso_open, ) -from .gpu import gpu_cleanup +from .gpu import gpu_cleanup, num_gpus from .ign import generate_ign_overlay from .rendering import tif_to_png @@ -447,14 +447,20 @@ class LidarArchaeoPipeline: t_pipeline_start = time.time() if self.workers > 1 and len(files) > 1: - logger.info(f"Traitement parallèle avec {self.workers} workers...") + n_gpus = num_gpus() + if n_gpus > 1: + logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...") + else: + logger.info(f"Traitement parallèle avec {self.workers} workers...") logger.info(f"Fichiers: {len(files)}") + with ProcessPoolExecutor(max_workers=self.workers) as executor: # Pass resolutions as comma-separated string for multiprocessing serialization resolutions_str = ','.join(str(r) for r in self.resolutions) + n_gpus = num_gpus() or 1 future_to_file = { - executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format): laz_file - for laz_file in files + executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file + for gpu_id, laz_file in enumerate(files) } done = 0 try: @@ -526,11 +532,19 @@ class LidarArchaeoPipeline: logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}") -def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif'): +def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif', gpu_id=None): """Standalone function for multiprocessing — creates its own pipeline instance. Each worker gets its own temp directory to avoid file conflicts. + When multiple GPUs are available, each worker is assigned a GPU via + CUDA_VISIBLE_DEVICES to balance load across GPUs. """ + # Assign GPU FIRST — before any CuPy import happens + # This sets CUDA_VISIBLE_DEVICES and resets lazy CuPy init + if gpu_id is not None and gpu_id >= 0: + from .gpu import set_active_gpu + set_active_gpu(gpu_id) + # Configure logging in worker process (spawn doesn't inherit parent config) import logging import sys diff --git a/lidar_pipeline/rendering.py b/lidar_pipeline/rendering.py index 54d6bdf..3101c44 100644 --- a/lidar_pipeline/rendering.py +++ b/lidar_pipeline/rendering.py @@ -274,15 +274,16 @@ def _apply_colormap(data, tif_file): def _download_location_map(min_x, max_x, min_y, max_y): """Download a wide-area IGN topographic map for location context. - Downloads a zoomed-out IGN PLANIGNV2 tile covering 5-10x the processed - zone extent, giving a wider geographic context. Results are cached to + Downloads a zoomed-out IGN PLANIGNV2 tile covering ~80km around the + processed zone, giving wider geographic context. Results are cached to avoid re-downloading for each visualization in the same tile. Args: min_x, max_x, min_y, max_y: DTM bounds in Lambert 93. Returns: - numpy array (H, W, 3) uint8, or None on failure. + Tuple (image_array, bounds_dict) where bounds_dict has keys + 'min_x', 'max_x', 'min_y', 'max_y' in Lambert 93, or None on failure. """ import hashlib @@ -323,7 +324,13 @@ def _download_location_map(min_x, max_x, min_y, max_y): ) if result is not None: - _location_map_cache[cache_key] = result + bounds = { + 'min_x': context_min_x, 'max_x': context_max_x, + 'min_y': context_min_y, 'max_y': context_max_y, + } + cached = (result, bounds) + _location_map_cache[cache_key] = cached + return cached return result except Exception as e: @@ -637,6 +644,7 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None, edgecolor='#cccccc', alpha=0.9)) # Scale bar — adaptive with alternating black/white segments + # Position: left of the location map to avoid overlap extent_m_x = max_x - min_x scale_m, scale_label = _nice_scale(extent_m_x) pixels_per_meter = 1.0 / pixel_size_x @@ -646,7 +654,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None, bar_bottom_y = 0.55 bar_top_y = 0.85 bar_height = bar_top_y - bar_bottom_y - scale_start_x = 0.80 + # Place scale bar so it ends before the location map (map starts at x=0.82 in fig coords) + # map_ax occupies [0.82, 0.02, 0.16, 0.13] in figure coords + # info_ax occupies [data_left, 0.015, width, 0.09] + # Scale bar end in fig coords = info_ax.left + (scale_start_x + scale_px/width) * info_ax.width + # We need: info_ax.left + (scale_start_x + scale_px/width) * info_ax.width < 0.80 + scale_end_frac = scale_px / width # fraction of info_ax width + info_ax_width = data_width_frac + cbar_width + 0.02 + # Calculate scale_start_x so scale bar ends at fig_x = 0.78 (leaving gap before map at 0.82) + max_scale_end_fig = 0.78 + scale_end_in_info = (max_scale_end_fig - data_left) / info_ax_width + scale_start_x = max(0.05, scale_end_in_info - scale_end_frac) for seg_i in range(n_segments): color = 'black' if seg_i % 2 == 0 else 'white' @@ -666,15 +684,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None, color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False) # Location inset map — IGN topographic background with processed zone marker - map_ax = fig.add_axes([0.84, 0.02, 0.14, 0.12]) + # Positioned in lower-right corner, above the info bar + map_ax = fig.add_axes([0.82, 0.02, 0.16, 0.13]) # Try to download a wide-area IGN topo map for location context - location_map = _download_location_map(min_x, max_x, min_y, max_y) - if location_map is not None: - # Draw IGN topo map as background + location_result = _download_location_map(min_x, max_x, min_y, max_y) + if location_result is not None: + location_map, loc_bounds = location_result + # Draw IGN topo map as background with correct bounds map_ax.imshow(location_map, aspect='equal', extent=[ - min_x - (max_x - min_x) * 2, max_x + (max_x - min_x) * 2, - min_y - (max_y - min_y) * 2, max_y + (max_y - min_y) * 2 + loc_bounds['min_x'], loc_bounds['max_x'], + loc_bounds['min_y'], loc_bounds['max_y'] ]) # Mark the processed zone with a red rectangle rect_x1, rect_x2 = min_x, max_x