Add multi-GPU support and fix scale bar / location map overlap
Multi-GPU: - gpu.py: lazy CuPy initialization so CUDA_VISIBLE_DEVICES takes effect before context creation in worker processes - gpu.py: detect GPU count via nvidia-smi (no CUDA import needed) - gpu.py: add set_active_gpu() to assign workers to specific GPUs - pipeline.py: distribute files across GPUs (file % num_gpus) in parallel mode so both GPUs are used simultaneously - pipeline.py: log GPU count when multiple GPUs detected Layout fixes: - rendering.py: move scale bar left of location map to avoid overlap (scale bar ends at fig_x=0.78, map starts at 0.82) - rendering.py: expand location map inset to 0.16x0.13 fig coords - rendering.py: return bounds from _download_location_map so imshow extent matches the actual IGN tile coverage (80km context) - ign.py: add min_zoom parameter to download_ign_tiles, fixing the location map that was broken (zoom 10 blocked by hardcoded min_zoom=15)
This commit is contained in:
@ -6,35 +6,111 @@ operations fall back to numpy/scipy on CPU.
|
|||||||
|
|
||||||
GPU errors (e.g. in forked subprocesses) are caught gracefully and
|
GPU errors (e.g. in forked subprocesses) are caught gracefully and
|
||||||
cause an automatic fallback to CPU for the current operation.
|
cause an automatic fallback to CPU for the current operation.
|
||||||
|
|
||||||
|
Multi-GPU support: when multiple GPUs are available, each worker process
|
||||||
|
can be assigned a different GPU via set_active_gpu() for balanced load.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from scipy import ndimage
|
from scipy import ndimage
|
||||||
|
|
||||||
logger = logging.getLogger("lidar")
|
logger = logging.getLogger("lidar")
|
||||||
|
|
||||||
# GPU detection - must happen at import time
|
# Detect GPU count at import time WITHOUT importing CuPy.
|
||||||
|
# We use nvidia-smi or CUDA_VISIBLE_DEVICES to count GPUs,
|
||||||
|
# so that CUDA_VISIBLE_DEVICES can be set BEFORE CuPy context creation
|
||||||
|
# in worker processes.
|
||||||
|
_NUM_GPUS = 0
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
_gpu_name = None
|
_gpu_name = None
|
||||||
_gpu_mem_gb = 0
|
_gpu_mem_gb = 0
|
||||||
|
|
||||||
|
# Check if GPUs are available via nvidia-smi (no CUDA context created)
|
||||||
|
try:
|
||||||
|
import subprocess
|
||||||
|
_result = subprocess.run(
|
||||||
|
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
|
||||||
|
capture_output=True, text=True, timeout=5
|
||||||
|
)
|
||||||
|
if _result.returncode == 0:
|
||||||
|
_lines = _result.stdout.strip().split('\n')
|
||||||
|
_NUM_GPUS = len(_lines)
|
||||||
|
# Parse first GPU info for logging
|
||||||
|
_parts = _lines[0].split(',')
|
||||||
|
if len(_parts) >= 3:
|
||||||
|
_gpu_name = _parts[1].strip()
|
||||||
|
try:
|
||||||
|
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
|
||||||
|
except (ValueError, IndexError):
|
||||||
|
pass
|
||||||
|
HAS_GPU = True
|
||||||
|
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Lazy-initialized GPU module references
|
||||||
|
# CuPy is imported only when first needed, allowing CUDA_VISIBLE_DEVICES
|
||||||
|
# to be set before CuPy context creation in worker processes.
|
||||||
_xp = np # Default: CPU
|
_xp = np # Default: CPU
|
||||||
_cp = None # cupy module (or None)
|
_cp = None # cupy module (or None)
|
||||||
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
|
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
|
||||||
|
_gpu_initialized = False
|
||||||
|
|
||||||
try:
|
|
||||||
import cupy as _cupy
|
|
||||||
import cupyx.scipy.ndimage as _cupy_ndimage
|
|
||||||
|
|
||||||
_gpu_info = _cupy.cuda.runtime.getDeviceProperties(0)
|
def _init_gpu():
|
||||||
_gpu_name = _gpu_info['name'].decode() if isinstance(_gpu_info['name'], bytes) else str(_gpu_info['name'])
|
"""Lazily initialize CuPy on first GPU use.
|
||||||
_gpu_mem_gb = _gpu_info['totalGlobalMem'] // (1024 ** 3)
|
|
||||||
HAS_GPU = True
|
This allows CUDA_VISIBLE_DEVICES to take effect in worker processes
|
||||||
_xp = _cupy
|
before CuPy creates a CUDA context.
|
||||||
_cp = _cupy
|
"""
|
||||||
_cp_ndimage = _cupy_ndimage
|
global _xp, _cp, _cp_ndimage, _gpu_initialized
|
||||||
except (ImportError, Exception):
|
if _gpu_initialized:
|
||||||
pass
|
return
|
||||||
|
_gpu_initialized = True
|
||||||
|
try:
|
||||||
|
import cupy as _real_cupy
|
||||||
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||||
|
# Verify GPU is actually accessible
|
||||||
|
_real_cupy.cuda.runtime.getDevice()
|
||||||
|
_xp = _real_cupy
|
||||||
|
_cp = _real_cupy
|
||||||
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
except (ImportError, Exception):
|
||||||
|
_xp = np
|
||||||
|
_cp = None
|
||||||
|
_cp_ndimage = None
|
||||||
|
|
||||||
|
|
||||||
|
def num_gpus():
|
||||||
|
"""Return the number of available CUDA GPUs."""
|
||||||
|
return _NUM_GPUS
|
||||||
|
|
||||||
|
|
||||||
|
def set_active_gpu(gpu_id):
|
||||||
|
"""Set the active GPU for the current process.
|
||||||
|
|
||||||
|
Must be called BEFORE any GPU operation (to_gpu, etc.) to ensure
|
||||||
|
the CUDA context is created on the correct device.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
gpu_id: 0-based GPU index. Clamped to valid range.
|
||||||
|
"""
|
||||||
|
if not HAS_GPU or _NUM_GPUS <= 1:
|
||||||
|
return # Nothing to do for single GPU or no GPU
|
||||||
|
|
||||||
|
gpu_id = gpu_id % _NUM_GPUS
|
||||||
|
# Set CUDA_VISIBLE_DEVICES before CuPy context is created
|
||||||
|
# This is the most reliable way in spawn processes
|
||||||
|
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
|
||||||
|
# Reset lazy init so CuPy re-detects with the new env
|
||||||
|
global _gpu_initialized, _cp, _cp_ndimage, _xp
|
||||||
|
_gpu_initialized = False
|
||||||
|
_cp = None
|
||||||
|
_cp_ndimage = None
|
||||||
|
_xp = np
|
||||||
|
|
||||||
|
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
|
||||||
|
|
||||||
|
|
||||||
def _gpu_available():
|
def _gpu_available():
|
||||||
@ -42,6 +118,7 @@ def _gpu_available():
|
|||||||
if not HAS_GPU:
|
if not HAS_GPU:
|
||||||
return False
|
return False
|
||||||
try:
|
try:
|
||||||
|
_init_gpu()
|
||||||
_cp.cuda.runtime.getDevice()
|
_cp.cuda.runtime.getDevice()
|
||||||
return True
|
return True
|
||||||
except Exception:
|
except Exception:
|
||||||
@ -50,8 +127,11 @@ def _gpu_available():
|
|||||||
|
|
||||||
def log_gpu_status():
|
def log_gpu_status():
|
||||||
"""Log GPU detection result. Called after logging is configured."""
|
"""Log GPU detection result. Called after logging is configured."""
|
||||||
if _gpu_available():
|
if HAS_GPU:
|
||||||
logger.info(f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)")
|
gpu_info = f"GPU détectée: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||||
|
if _NUM_GPUS > 1:
|
||||||
|
gpu_info += f" × {_NUM_GPUS}"
|
||||||
|
logger.info(gpu_info)
|
||||||
else:
|
else:
|
||||||
logger.info("Pas de GPU — mode CPU uniquement")
|
logger.info("Pas de GPU — mode CPU uniquement")
|
||||||
|
|
||||||
|
|||||||
@ -63,7 +63,7 @@ from .visualizations import (
|
|||||||
generate_roughness, generate_wavelet,
|
generate_roughness, generate_wavelet,
|
||||||
generate_svf, generate_aniso_open,
|
generate_svf, generate_aniso_open,
|
||||||
)
|
)
|
||||||
from .gpu import gpu_cleanup
|
from .gpu import gpu_cleanup, num_gpus
|
||||||
from .ign import generate_ign_overlay
|
from .ign import generate_ign_overlay
|
||||||
from .rendering import tif_to_png
|
from .rendering import tif_to_png
|
||||||
|
|
||||||
@ -447,14 +447,20 @@ class LidarArchaeoPipeline:
|
|||||||
t_pipeline_start = time.time()
|
t_pipeline_start = time.time()
|
||||||
|
|
||||||
if self.workers > 1 and len(files) > 1:
|
if self.workers > 1 and len(files) > 1:
|
||||||
logger.info(f"Traitement parallèle avec {self.workers} workers...")
|
n_gpus = num_gpus()
|
||||||
|
if n_gpus > 1:
|
||||||
|
logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...")
|
||||||
|
else:
|
||||||
|
logger.info(f"Traitement parallèle avec {self.workers} workers...")
|
||||||
logger.info(f"Fichiers: {len(files)}")
|
logger.info(f"Fichiers: {len(files)}")
|
||||||
|
|
||||||
with ProcessPoolExecutor(max_workers=self.workers) as executor:
|
with ProcessPoolExecutor(max_workers=self.workers) as executor:
|
||||||
# Pass resolutions as comma-separated string for multiprocessing serialization
|
# Pass resolutions as comma-separated string for multiprocessing serialization
|
||||||
resolutions_str = ','.join(str(r) for r in self.resolutions)
|
resolutions_str = ','.join(str(r) for r in self.resolutions)
|
||||||
|
n_gpus = num_gpus() or 1
|
||||||
future_to_file = {
|
future_to_file = {
|
||||||
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format): laz_file
|
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file
|
||||||
for laz_file in files
|
for gpu_id, laz_file in enumerate(files)
|
||||||
}
|
}
|
||||||
done = 0
|
done = 0
|
||||||
try:
|
try:
|
||||||
@ -526,11 +532,19 @@ class LidarArchaeoPipeline:
|
|||||||
logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}")
|
logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}")
|
||||||
|
|
||||||
|
|
||||||
def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif'):
|
def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, force=False, ground_method='auto', force_classify=False, keep_tif=False, quality=98, only_viz=None, skip_viz=None, output_format='avif', gpu_id=None):
|
||||||
"""Standalone function for multiprocessing — creates its own pipeline instance.
|
"""Standalone function for multiprocessing — creates its own pipeline instance.
|
||||||
|
|
||||||
Each worker gets its own temp directory to avoid file conflicts.
|
Each worker gets its own temp directory to avoid file conflicts.
|
||||||
|
When multiple GPUs are available, each worker is assigned a GPU via
|
||||||
|
CUDA_VISIBLE_DEVICES to balance load across GPUs.
|
||||||
"""
|
"""
|
||||||
|
# Assign GPU FIRST — before any CuPy import happens
|
||||||
|
# This sets CUDA_VISIBLE_DEVICES and resets lazy CuPy init
|
||||||
|
if gpu_id is not None and gpu_id >= 0:
|
||||||
|
from .gpu import set_active_gpu
|
||||||
|
set_active_gpu(gpu_id)
|
||||||
|
|
||||||
# Configure logging in worker process (spawn doesn't inherit parent config)
|
# Configure logging in worker process (spawn doesn't inherit parent config)
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
|
|||||||
@ -274,15 +274,16 @@ def _apply_colormap(data, tif_file):
|
|||||||
def _download_location_map(min_x, max_x, min_y, max_y):
|
def _download_location_map(min_x, max_x, min_y, max_y):
|
||||||
"""Download a wide-area IGN topographic map for location context.
|
"""Download a wide-area IGN topographic map for location context.
|
||||||
|
|
||||||
Downloads a zoomed-out IGN PLANIGNV2 tile covering 5-10x the processed
|
Downloads a zoomed-out IGN PLANIGNV2 tile covering ~80km around the
|
||||||
zone extent, giving a wider geographic context. Results are cached to
|
processed zone, giving wider geographic context. Results are cached to
|
||||||
avoid re-downloading for each visualization in the same tile.
|
avoid re-downloading for each visualization in the same tile.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
min_x, max_x, min_y, max_y: DTM bounds in Lambert 93.
|
min_x, max_x, min_y, max_y: DTM bounds in Lambert 93.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
numpy array (H, W, 3) uint8, or None on failure.
|
Tuple (image_array, bounds_dict) where bounds_dict has keys
|
||||||
|
'min_x', 'max_x', 'min_y', 'max_y' in Lambert 93, or None on failure.
|
||||||
"""
|
"""
|
||||||
import hashlib
|
import hashlib
|
||||||
|
|
||||||
@ -323,7 +324,13 @@ def _download_location_map(min_x, max_x, min_y, max_y):
|
|||||||
)
|
)
|
||||||
|
|
||||||
if result is not None:
|
if result is not None:
|
||||||
_location_map_cache[cache_key] = result
|
bounds = {
|
||||||
|
'min_x': context_min_x, 'max_x': context_max_x,
|
||||||
|
'min_y': context_min_y, 'max_y': context_max_y,
|
||||||
|
}
|
||||||
|
cached = (result, bounds)
|
||||||
|
_location_map_cache[cache_key] = cached
|
||||||
|
return cached
|
||||||
|
|
||||||
return result
|
return result
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@ -637,6 +644,7 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
|
|||||||
edgecolor='#cccccc', alpha=0.9))
|
edgecolor='#cccccc', alpha=0.9))
|
||||||
|
|
||||||
# Scale bar — adaptive with alternating black/white segments
|
# Scale bar — adaptive with alternating black/white segments
|
||||||
|
# Position: left of the location map to avoid overlap
|
||||||
extent_m_x = max_x - min_x
|
extent_m_x = max_x - min_x
|
||||||
scale_m, scale_label = _nice_scale(extent_m_x)
|
scale_m, scale_label = _nice_scale(extent_m_x)
|
||||||
pixels_per_meter = 1.0 / pixel_size_x
|
pixels_per_meter = 1.0 / pixel_size_x
|
||||||
@ -646,7 +654,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
|
|||||||
bar_bottom_y = 0.55
|
bar_bottom_y = 0.55
|
||||||
bar_top_y = 0.85
|
bar_top_y = 0.85
|
||||||
bar_height = bar_top_y - bar_bottom_y
|
bar_height = bar_top_y - bar_bottom_y
|
||||||
scale_start_x = 0.80
|
# Place scale bar so it ends before the location map (map starts at x=0.82 in fig coords)
|
||||||
|
# map_ax occupies [0.82, 0.02, 0.16, 0.13] in figure coords
|
||||||
|
# info_ax occupies [data_left, 0.015, width, 0.09]
|
||||||
|
# Scale bar end in fig coords = info_ax.left + (scale_start_x + scale_px/width) * info_ax.width
|
||||||
|
# We need: info_ax.left + (scale_start_x + scale_px/width) * info_ax.width < 0.80
|
||||||
|
scale_end_frac = scale_px / width # fraction of info_ax width
|
||||||
|
info_ax_width = data_width_frac + cbar_width + 0.02
|
||||||
|
# Calculate scale_start_x so scale bar ends at fig_x = 0.78 (leaving gap before map at 0.82)
|
||||||
|
max_scale_end_fig = 0.78
|
||||||
|
scale_end_in_info = (max_scale_end_fig - data_left) / info_ax_width
|
||||||
|
scale_start_x = max(0.05, scale_end_in_info - scale_end_frac)
|
||||||
|
|
||||||
for seg_i in range(n_segments):
|
for seg_i in range(n_segments):
|
||||||
color = 'black' if seg_i % 2 == 0 else 'white'
|
color = 'black' if seg_i % 2 == 0 else 'white'
|
||||||
@ -666,15 +684,17 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
|
|||||||
color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False)
|
color='black', linewidth=1, transform=info_ax.transAxes, clip_on=False)
|
||||||
|
|
||||||
# Location inset map — IGN topographic background with processed zone marker
|
# Location inset map — IGN topographic background with processed zone marker
|
||||||
map_ax = fig.add_axes([0.84, 0.02, 0.14, 0.12])
|
# Positioned in lower-right corner, above the info bar
|
||||||
|
map_ax = fig.add_axes([0.82, 0.02, 0.16, 0.13])
|
||||||
|
|
||||||
# Try to download a wide-area IGN topo map for location context
|
# Try to download a wide-area IGN topo map for location context
|
||||||
location_map = _download_location_map(min_x, max_x, min_y, max_y)
|
location_result = _download_location_map(min_x, max_x, min_y, max_y)
|
||||||
if location_map is not None:
|
if location_result is not None:
|
||||||
# Draw IGN topo map as background
|
location_map, loc_bounds = location_result
|
||||||
|
# Draw IGN topo map as background with correct bounds
|
||||||
map_ax.imshow(location_map, aspect='equal', extent=[
|
map_ax.imshow(location_map, aspect='equal', extent=[
|
||||||
min_x - (max_x - min_x) * 2, max_x + (max_x - min_x) * 2,
|
loc_bounds['min_x'], loc_bounds['max_x'],
|
||||||
min_y - (max_y - min_y) * 2, max_y + (max_y - min_y) * 2
|
loc_bounds['min_y'], loc_bounds['max_y']
|
||||||
])
|
])
|
||||||
# Mark the processed zone with a red rectangle
|
# Mark the processed zone with a red rectangle
|
||||||
rect_x1, rect_x2 = min_x, max_x
|
rect_x1, rect_x2 = min_x, max_x
|
||||||
|
|||||||
Reference in New Issue
Block a user