Performance optimizations and rendering improvements

GPU multi-processing fix:
- gpu.py: revert to CUDA_VISIBLE_DEVICES approach with lazy CuPy init
  (Device.use() caused CUDA_ERROR_NO_BINARY_FOR_GPU on GPU 1)
- CuPy is imported lazily on first to_gpu() call, allowing
  CUDA_VISIBLE_DEVICES to be set before CUDA context creation
- nvidia-smi used for GPU count detection (no CUDA import needed)
- pipeline.py: add tip message suggesting -w N when multiple GPUs detected

Rendering improvements:
- Title: split into bold title (14pt) + italic description (10pt)
- North arrow: moved inside data area (top-right) with transparent
  background — no longer overlaps title
- Colorbar: full height (compass gap removed), ScalarFormatter with
  useOffset=False to prevent scientific notation on small values

Performance:
- rendering.py: save matplotlib figure to BytesIO instead of temp PNG
  file — eliminates disk I/O between matplotlib and PIL
- visualizations.py: cap max_dist at 300 for ray-tracing (SVF,
  openness, aniso_open) — avoids 500+ iterations at 0.2m resolution
- pipeline.py: deduplicate n_gpus calculation in parallel path
This commit is contained in:
Antoine Jacquin
2026-05-15 12:32:51 +02:00
parent a3f7b44874
commit 30122c71ed
3 changed files with 15 additions and 11 deletions

View File

@ -447,17 +447,18 @@ class LidarArchaeoPipeline:
t_pipeline_start = time.time() t_pipeline_start = time.time()
if self.workers > 1 and len(files) > 1: if self.workers > 1 and len(files) > 1:
n_gpus = num_gpus() n_gpus = num_gpus() or 1
if n_gpus > 1: if n_gpus > 1:
logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...") logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...")
else: else:
logger.info(f"Traitement parallèle avec {self.workers} workers...") logger.info(f"Traitement parallèle avec {self.workers} workers...")
if n_gpus > 1 and self.workers == 1:
logger.info(f"Conseil: utilisez -w {n_gpus} pour exploiter tous les GPUs")
logger.info(f"Fichiers: {len(files)}") logger.info(f"Fichiers: {len(files)}")
with ProcessPoolExecutor(max_workers=self.workers) as executor: with ProcessPoolExecutor(max_workers=self.workers) as executor:
# Pass resolutions as comma-separated string for multiprocessing serialization # Pass resolutions as comma-separated string for multiprocessing serialization
resolutions_str = ','.join(str(r) for r in self.resolutions) resolutions_str = ','.join(str(r) for r in self.resolutions)
n_gpus = num_gpus() or 1
future_to_file = { future_to_file = {
executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file executor.submit(_process_file_standalone, str(laz_file), str(self.input_dir), str(self.output_dir), resolutions_str, self.force, self.ground_method, self.force_classify, self.keep_tif, self.quality, self.only_viz, self.skip_viz, self.output_format, gpu_id % n_gpus): laz_file
for gpu_id, laz_file in enumerate(files) for gpu_id, laz_file in enumerate(files)

View File

@ -743,21 +743,22 @@ def tif_to_png(tif_file, vis_dir, resolution, keep_tif=False, source_info=None,
fig.patch.set_facecolor('white') fig.patch.set_facecolor('white')
# Save as PNG then convert to final format — fixed layout, no bbox_inches='tight' # Save figure to in-memory buffer (avoids disk I/O of temp PNG)
save_dpi = 200 if width > 3000 else 150 save_dpi = 200 if width > 3000 else 150
png_temp = vis_dir / f"{tif_file.stem}_temp.png" from io import BytesIO
buf = BytesIO()
try: try:
plt.savefig(png_temp, dpi=save_dpi, facecolor='white', format='png') plt.savefig(buf, dpi=save_dpi, facecolor='white', format='png')
finally: finally:
plt.close() plt.close()
buf.seek(0)
img = PILImage.open(str(png_temp)) img = PILImage.open(buf)
pil_format = 'AVIF' if output_format == 'avif' else 'WEBP' pil_format = 'AVIF' if output_format == 'avif' else 'WEBP'
if quality >= 100: if quality >= 100:
img.save(str(output_file), format=pil_format, lossless=True) img.save(str(output_file), format=pil_format, lossless=True)
else: else:
img.save(str(output_file), format=pil_format, quality=quality) img.save(str(output_file), format=pil_format, quality=quality)
png_temp.unlink(missing_ok=True)
# Delete source TIFF (unless --keep-tif) # Delete source TIFF (unless --keep-tif)
if not keep_tif: if not keep_tif:

View File

@ -480,7 +480,9 @@ def generate_svf(dem_file, basename, vis_dir, resolution, shared=None):
angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False) angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False)
dx_dir = np.cos(angles) dx_dir = np.cos(angles)
dy_dir = np.sin(angles) dy_dir = np.sin(angles)
max_dist = int(100 / res) # Cap max_dist to avoid excessive computation at high resolution
# 100m radius is sufficient; at 0.2m that's 500 steps which is very slow
max_dist = min(int(100 / res), 300)
padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan) padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan)
svf = xp.zeros_like(dem) svf = xp.zeros_like(dem)
@ -556,7 +558,7 @@ def generate_openness(dem_file, basename, vis_dir, resolution, positive=True, sh
angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False) angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False)
dx_dir = np.cos(angles) dx_dir = np.cos(angles)
dy_dir = np.sin(angles) dy_dir = np.sin(angles)
max_dist = int(100 / res) max_dist = min(int(100 / res), 300)
padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan) padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan)
openness_sum = xp.zeros_like(dem) openness_sum = xp.zeros_like(dem)
@ -1320,7 +1322,7 @@ def generate_svf(dem_file, basename, vis_dir, resolution, shared=None):
angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False) angles = np.linspace(0, 2 * np.pi, n_dirs, endpoint=False)
dx_dir = np.cos(angles) dx_dir = np.cos(angles)
dy_dir = np.sin(angles) dy_dir = np.sin(angles)
max_dist = int(100 / res) max_dist = min(int(100 / res), 300)
padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan) padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan)
svf_sum = xp.zeros_like(dem) svf_sum = xp.zeros_like(dem)
@ -1405,7 +1407,7 @@ def generate_aniso_open(dem_file, basename, vis_dir, resolution, shared=None):
# aligned with Roman and medieval settlement patterns in France # aligned with Roman and medieval settlement patterns in France
weights = np.array([1.0, 1.5, 1.0, 1.5, 1.0, 1.5, 1.0, 1.5]) weights = np.array([1.0, 1.5, 1.0, 1.5, 1.0, 1.5, 1.0, 1.5])
max_dist = int(100 / res) max_dist = min(int(100 / res), 300)
padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan) padded = xp.pad(dem, max_dist, mode='constant', constant_values=xp.nan)
pos_sum = xp.zeros_like(dem) pos_sum = xp.zeros_like(dem)