Translate the whole project to English and fix outdated comments and help

Comments, docstrings, logs, CLI help, map UI, legends, PDF sheet, scripts,
compose files and AGENTS.md are now English. Data keys stay unchanged
(relief_oriente, densite_sol, visualisations/, API JSON keys, link params).
Wrong comments and help defaults found along the way are corrected.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Antoine
2026-09-27 23:16:45 +02:00
parent cc1c22d2b8
commit fb892ea9f2
52 changed files with 4356 additions and 4323 deletions

View File

@ -1,9 +1,10 @@
"""Pipeline orchestration for LiDAR archaeological analysis.
LidarArchaeoPipeline coordinates the full processing chain:
1. Ground classification (PDAL/SMRF)
1. Ground classification (IGN pre-classification by default; SMRF/CSF via PDAL)
2. DTM generation
3. Visualization generation (17 products)
3. Visualization generation (17 available products; a default run only
produces the map layers, PANEL_VIZ in index.py)
4. Rendering (AVIF/WebP conversion)
"""
@ -28,14 +29,14 @@ logger = logging.getLogger("lidar")
def resolve_workers(value):
"""Traduit l'option -w en nombre effectif de workers.
"""Turn the -w option into an effective worker count.
Résolu au lancement de chaque run (pas au démarrage du serveur, ni par
tuile : le pool de processus vit le temps du run). « auto » = cœurs
logiques - 2 (un pour l'OS/serveur, un pour les phases I/O et
l'indexation), borné [2, 16] — chaque worker traite une tuile et peut
lancer un processus PDAL en flux, le compte reste raisonnable même sur
une grosse machine.
Resolved when each run starts (not at server startup, nor per tile: the
process pool lives for the duration of the run). "auto" = logical cores
- 2 (one for the OS/server, one for the I/O and indexing phases),
clamped to [2, 16] — each worker processes one tile and may spawn a
streaming PDAL process, so the count stays reasonable even on a large
machine.
"""
if isinstance(value, str) and value.strip().lower() == "auto":
cpus = os.cpu_count() or 4
@ -119,16 +120,16 @@ VIZ_STEPS = [
('ortho', lambda d, b, v, r: generate_ign_overlay(
d, b, v, r,
layer='ORTHOIMAGERY.ORTHOPHOTOS',
title='Photographie Aérienne IGN',
legend_label='Orthophotographie\nImage aérienne',
description='Photographie aérienne IGN (Orthophoto)',
title='IGN Aerial Photograph',
legend_label='Orthophoto\nAerial image',
description='IGN aerial photograph (orthophoto)',
out_suffix='ortho')),
('topo', lambda d, b, v, r: generate_ign_overlay(
d, b, v, r,
layer='GEOGRAPHICALGRIDSYSTEMS.PLANIGNV2',
title='Carte Topographique IGN',
legend_label='Carte IGN\nPlan topographique',
description='Carte topographique IGN (Plan IGN)',
title='IGN Topographic Map',
legend_label='IGN map\nTopographic map',
description='IGN topographic map (Plan IGN)',
out_suffix='topo')),
]
@ -159,9 +160,9 @@ class LidarArchaeoPipeline:
self.output_format = output_format
self.gpu_ids = gpu_ids
self.no_index = no_index
# Inventaire mis à jour après chaque dalle par défaut : la carte (et sa
# maintenance de pyramide) suit le rendu en cours, quel que soit le
# lanceur (carte, compose, run.sh). --no-index le désactive.
# Inventory updated after every tile by default: the map (and its
# pyramid maintenance) follows the ongoing render, whatever the
# launcher (map, compose, run.sh). --no-index disables it.
self.incremental_index = bool(incremental_index) or not no_index
self.strip_align = strip_align
self.openness_downsample = openness_downsample
@ -172,7 +173,7 @@ class LidarArchaeoPipeline:
self.temp_dir = self.output_dir / "temp"
if not self.input_dir.exists():
raise ValueError(f"Répertoire introuvable: {self.input_dir}")
raise ValueError(f"Directory not found: {self.input_dir}")
self.output_dir.mkdir(parents=True, exist_ok=True)
self.temp_dir.mkdir(exist_ok=True)
@ -188,49 +189,49 @@ class LidarArchaeoPipeline:
if only_viz:
invalid = set(only_viz) - set(all_viz_names)
if invalid:
raise ValueError(f"Visualisations inconnues: {', '.join(invalid)}. Disponibles: {', '.join(all_viz_names)}")
raise ValueError(f"Unknown visualizations: {', '.join(invalid)}. Available: {', '.join(all_viz_names)}")
self.viz_steps = [(n, f) for n, f in VIZ_STEPS if n in only_viz]
elif skip_viz:
invalid = set(skip_viz) - set(all_viz_names)
if invalid:
raise ValueError(f"Visualisations inconnues: {', '.join(invalid)}. Disponibles: {', '.join(all_viz_names)}")
raise ValueError(f"Unknown visualizations: {', '.join(invalid)}. Available: {', '.join(all_viz_names)}")
self.viz_steps = [(n, f) for n, f in VIZ_STEPS if n not in skip_viz]
else:
# Sans --only/--skip : seules les couches affichées par la carte
# (PANEL_VIZ, index.py) sont produites.
# Without --only/--skip: only the layers shown on the map
# (PANEL_VIZ, index.py) are produced.
from .index import panel_steps
default = panel_steps()
self.viz_steps = [(n, f) for n, f in VIZ_STEPS if default is None or n in default]
logger.info("Pipeline initialisé")
logger.info(f" Entrée : {self.input_dir}")
logger.info(f" Sortie : {self.output_dir}")
logger.info("Pipeline initialized")
logger.info(f" Input : {self.input_dir}")
logger.info(f" Output : {self.output_dir}")
if len(self.resolutions) > 1:
logger.info(f" Résolutions : {', '.join(f'{r}m/px' for r in self.resolutions)}")
logger.info(f" Resolutions : {', '.join(f'{r} m/px' for r in self.resolutions)}")
else:
logger.info(f" Résolution : {self.resolution}m/px")
logger.info(f" Resolution : {self.resolution} m/px")
logger.info(f" Workers : {workers}")
logger.info(f" Force : {'OUI' if self.force else 'non (skip existing)'}")
logger.info(f" Classification sol : {self.ground_method}")
logger.info(f" Force classif.: {'OUI' if self.force_classify else 'non'}")
logger.info(f" Keep TIFF : {'OUI' if self.keep_tif else 'non'}")
logger.info(f" Force : {'YES' if self.force else 'no (skip existing)'}")
logger.info(f" Ground classification: {self.ground_method}")
logger.info(f" Force classif.: {'YES' if self.force_classify else 'no'}")
logger.info(f" Keep TIFF : {'YES' if self.keep_tif else 'no'}")
if self.edge_buffer > 0:
logger.info(f" Raccord bords: {self.edge_buffer:g} m (points sol des tuiles voisines)")
logger.info(f" Qualité {self.output_format.upper()}: {self.quality if self.quality < 100 else 'lossless'}")
logger.info(f" Edge buffer : {self.edge_buffer:g} m (ground points from neighboring tiles)")
logger.info(f" {self.output_format.upper()} quality: {self.quality if self.quality < 100 else 'lossless'}")
if only_viz:
logger.info(f" Visualisations: uniquement {', '.join(only_viz)}")
logger.info(f" Visualizations: only {', '.join(only_viz)}")
elif skip_viz:
logger.info(f" Visualisations: tout sauf {', '.join(skip_viz)}")
logger.info(f" Visualisations: {len(self.viz_steps)}/{len(VIZ_STEPS)}")
logger.info(f" Visualizations: all except {', '.join(skip_viz)}")
logger.info(f" Visualizations: {len(self.viz_steps)}/{len(VIZ_STEPS)}")
def find_laz_files(self):
"""Find all LAZ/LAS files in input directory, triés du nord au sud.
"""Find all LAZ/LAS files in the input directory, sorted north to south.
Les tuiles LHD_FXX_{col}_{row} sont ordonnées par ligne décroissante
(row = nord en km) puis colonne croissante : les workers prennent les
fichiers dans l'ordre de soumission, la carte se remplit ainsi du nord
vers le sud lors des passes globales. Les fichiers hors pattern LHD
restent triés par nom, en fin de liste.
LHD_FXX_{col}_{row} tiles are ordered by decreasing row (row = northing
in km), then increasing column: workers pick files up in submission
order, so the map fills from north to south during full passes. Files
that do not match the LHD pattern are sorted by name at the end of the
list.
"""
from .index import parse_basename_coords
files = list(self.input_dir.glob("*.laz")) + list(self.input_dir.glob("*.las"))
@ -243,7 +244,7 @@ class LidarArchaeoPipeline:
return (1, 0, 0, f.name)
files.sort(key=_north_key)
logger.info(f"{len(files)} fichier(s) LiDAR trouvé(s) — triés du nord au sud")
logger.info(f"{len(files)} LiDAR file(s) found — sorted north to south")
for f in files:
logger.debug(f" {f.name}")
return files
@ -256,7 +257,7 @@ class LidarArchaeoPipeline:
version = result.stdout.strip().split('\n')[0]
logger.info(f" ✓ {name}: {version}")
except (subprocess.CalledProcessError, FileNotFoundError):
logger.error(f" ✗ {name} non disponible")
logger.error(f" ✗ {name} not available")
return False
return True
@ -265,7 +266,7 @@ class LidarArchaeoPipeline:
"""Return the expected output filename for a visualization step."""
ext = 'avif' if output_format == 'avif' else 'webp'
if name in LOSSLESS_GRAY_KEYWORDS:
ext = 'webp' # aplats de niveaux : WebP sans perte (rendering.tif_to_crop)
ext = 'webp' # flat level classes: lossless WebP (rendering.tif_to_crop)
if name == 'pos_open':
return file_vis_dir / f"{basename}_positive_openness.{ext}"
elif name == 'neg_open':
@ -278,13 +279,13 @@ class LidarArchaeoPipeline:
def generate_all_visualizations(self, dtm_file, basename, resolution=None, vis_dir=None, force_images=None):
"""Generate all archaeological visualizations for one DTM file.
Optimisation: SharedDEM is only computed if at least one visualization
needs to be generated. When all WebP outputs exist, SharedDEM is
skipped entirely (saves ~2min per file on re-runs).
Optimization: SharedDEM is only computed if at least one visualization
needs to be generated. When all output images (AVIF/WebP) exist,
SharedDEM is skipped entirely (saves time on re-runs).
"""
if resolution is None:
resolution = self.resolution
logger.info(" Génération visualisations:")
logger.info(" Generating visualizations:")
# Use provided vis_dir (for multi-resolution subdirectories) or default
file_vis_dir = vis_dir if vis_dir else (self.vis_dir / basename)
@ -298,15 +299,15 @@ class LidarArchaeoPipeline:
if force_viz:
needs_generation[name] = True
else:
expected_webp = self._expected_output_path(name, basename, file_vis_dir, self.output_format)
needs_generation[name] = not expected_webp.exists()
expected_img = self._expected_output_path(name, basename, file_vis_dir, self.output_format)
needs_generation[name] = not expected_img.exists()
to_generate = [n for n, needed in needs_generation.items() if needed]
needs_shared = any(name not in ('ortho', 'topo') for name in to_generate)
if not to_generate:
logger.info(" Toutes les visualisations déjà existantes — ignorées")
# Still need to return results dict for PDF check
logger.info(" All visualizations already exist — skipped")
# Still return the results dict (expected output paths)
vis_results = {}
for name, func in self.viz_steps:
vis_results[name] = self._expected_output_path(name, basename, file_vis_dir, self.output_format)
@ -315,16 +316,16 @@ class LidarArchaeoPipeline:
# Phase 2: compute SharedDEM only if needed
shared = None
if needs_shared:
logger.info(" Pré-calcul données partagées (gradient, LRM)...")
logger.info(" Precomputing shared data (gradient, LRM)...")
t_shared = time.time()
shared = SharedDEM(dtm_file, resolution)
logger.info(f" ✓ Données partagées prêtes ({time.time()-t_shared:.1f}s)")
logger.info(f" ✓ Shared data ready ({time.time()-t_shared:.1f}s)")
# Phase 3: generate visualizations
vis_results = {}
for idx, (name, func) in enumerate(self.viz_steps, 1):
if not needs_generation[name]:
logger.info(f" [{idx}/{total}] {name}: déjà existant, ignoré")
logger.info(f" [{idx}/{total}] {name}: already exists, skipped")
self._report(basename, "viz", "skip", name, res=resolution)
vis_results[name] = self._expected_output_path(name, basename, file_vis_dir, self.output_format)
continue
@ -371,7 +372,7 @@ class LidarArchaeoPipeline:
# Convert to output format (only newly generated TIFs, not skipped ones)
fmt_label = self.output_format.upper()
logger.info(f" Conversion images {fmt_label}:")
logger.info(f" Converting images to {fmt_label}:")
for name, tif_file in vis_results.items():
if tif_file and isinstance(tif_file, Path) and tif_file.suffix == '.tif' and tif_file.exists():
img_file = tif_to_crop(tif_file, file_vis_dir, resolution, keep_tif=self.keep_tif, quality=self.quality, output_format=self.output_format, subtiles_dir=self.output_dir)
@ -408,10 +409,11 @@ class LidarArchaeoPipeline:
return None
def _effective_ground_method(self):
"""Méthode effective pour le suivi de cache, classes IGN incluses.
"""Effective method for cache tracking, IGN classes included.
La méthode 'ign' est étiquetée avec les classes choisies (ex. 'ign_1_2')
pour qu'un changement de --ign-classes déclenche la reclassification.
The 'ign' method is labeled with the chosen classes (e.g. 'ign_1_2')
so that changing --ign-classes triggers reclassification (the default
class list, ground only, keeps the plain 'ign' label).
"""
if self.ground_method == 'ign':
from .dtm import parse_ign_classes, ign_method_label
@ -428,12 +430,13 @@ class LidarArchaeoPipeline:
return recorded is None or recorded == self._effective_ground_method()
def _strip_align_matches(self, basename, res_suffix):
"""True si le sidecar de calage des faisceaux correspond à la config.
"""True if the strip-alignment sidecar matches the current config.
Un DTM sans sidecar (antérieur au calage) est régénéré pour mesurer
et consigner ses offsets ; un sidecar de version, de seuil ou de
paramètres de gigue intra-faisceau différents aussi. Calage désactivé :
tout DTM porteur d'un sidecar (donc calé) est régénéré non calé.
A DTM without a sidecar (older than strip alignment) is regenerated to
measure and record its offsets; so is one whose sidecar has a different
version, threshold, intra-strip jitter parameters or scan-line
parameters (window, cell, model). With alignment disabled, any DTM
carrying a sidecar (hence aligned) is regenerated unaligned.
"""
sidecar = self.dtm_dir / f"{basename}_dtm{res_suffix}_stripalign.json"
if not self.strip_align:
@ -462,34 +465,34 @@ class LidarArchaeoPipeline:
pass
def _edge_buffer_matches(self, dtm_path):
"""True si le tampon de raccord du DTM correspond à la config.
"""True if the DTM edge buffer matches the current config.
Le tampon est inscrit dans le tag GeoTIFF LIDAR_EDGE_BUFFER (dtm.py).
Un DTM sans tag (antérieur au raccord) compte comme tampon 0 :
activer --edge-buffer régénère donc les DTM en cache, le désactiver
régénère les DTM raccordés.
The buffer is stored in the LIDAR_EDGE_BUFFER GeoTIFF tag (dtm.py).
A DTM without the tag (older than edge stitching) counts as buffer 0:
enabling --edge-buffer therefore regenerates cached DTMs, and disabling
it regenerates buffered ones.
"""
from .dtm import read_dtm_edge_buffer
return abs(read_dtm_edge_buffer(dtm_path) - self.edge_buffer) < 1e-6
def _gap_fill_matches(self, dtm_path):
"""True si le DTM a été comblé par la version courante (tag
LIDAR_GAP_FILL, dtm.py) ; absent = ancien comblement à distance fixe."""
"""True if the DTM gaps were filled by the current version (the
LIDAR_GAP_FILL tag, dtm.py); missing = old fixed-distance filling."""
from .dtm import read_dtm_gap_fill, GAP_FILL_VERSION
return read_dtm_gap_fill(dtm_path) == GAP_FILL_VERSION
def _fetch_edge_neighbors(self, files):
"""Télécharge les dalles LAZ voisines manquantes (raccord des bords).
"""Download missing neighboring LAZ tiles (edge stitching).
La bande de raccord lit les 8 LAZ adjacentes de chaque tuile ; celles
absentes sont téléchargées depuis le catalogue IGN avant de lancer
les workers, dans le sous-dossier input/edge_neighbors/ : elles ne
doivent PAS rejoindre input/ à plat, sinon les passes globales
(« tout input/ ») les comptent comme des tuiles à rendre et la zone
grandit d'un anneau à chaque relance. La liste est dédupliquée sur
tout le lot : la couronne d'un bloc contigu ne coûte qu'un passage.
Une dalle introuvable (zone non publiée) ou en échec laisse simplement
la bande vide — le rendu continue (best-effort).
The edge band reads the 8 LAZ tiles adjacent to each tile; missing ones
are downloaded from the IGN catalog before the workers start, into the
input/edge_neighbors/ subdirectory: they must NOT land flat in input/,
otherwise full passes ("all of input/") would count them as tiles to
render and the area would grow by one ring on every rerun. The list is
deduplicated across the whole batch: the ring around a contiguous
block costs a single pass. A tile that cannot be found (unpublished
area) or fails to download simply leaves the band empty — rendering
continues (best effort).
"""
if self.edge_buffer <= 0:
return
@ -512,30 +515,29 @@ class LidarArchaeoPipeline:
if not missing:
return
edge_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"Raccord des bords : {len(missing)} dalle(s) voisine(s) "
f"absente(s) — téléchargement dans {EDGE_NEIGHBORS_DIRNAME}/ "
f"(catalogue IGN)")
logger.info(f"Edge stitching: {len(missing)} missing neighboring "
f"tile(s) — downloading into {EDGE_NEIGHBORS_DIRNAME}/ "
f"(IGN catalog)")
t0 = time.time()
from concurrent.futures import ThreadPoolExecutor
with ThreadPoolExecutor(max_workers=4) as pool:
fetched = [path for batch in pool.map(
lambda spec: fetch_tiles(edge_dir, [spec]), missing)
for path in batch]
logger.info(f"Raccord des bords : {len(fetched)}/{len(missing)} dalle(s) "
f"voisine(s) téléchargée(s) ({time.time() - t0:.0f} s)")
logger.info(f"Edge stitching: {len(fetched)}/{len(missing)} neighboring "
f"tile(s) downloaded ({time.time() - t0:.0f} s)")
def _cleanup_edge_neighbor_duplicates(self, edge_dir):
"""Supprime de edge_neighbors/ les dalles déjà présentes dans input/.
"""Remove from edge_neighbors/ the tiles already present in input/.
Une dalle téléchargée comme voisine avant d'être requise comme tuile
principale (ou l'inverse) peut se retrouver dupliquée aux deux
emplacements (avant que fetch_tiles ne promeuve les nouvelles
voisines) ; l'exemplaire d'input/ fait foi (_neighbor_laz_files
cherche dans input/ avant edge_neighbors/), le doublon ne sert donc
à rien et peut peser plusieurs centaines de Mo. Les fichiers ".part"
(téléchargement en cours) ne sont jamais touchés, et un doublon de
taille différente est conservé (l'exemplaire d'input/ pourrait être
tronqué : on ne supprime jamais ce qui serait la seule copie saine).
A tile downloaded as a neighbor before being requested as a main tile
(or the reverse) can end up duplicated in both locations (before
fetch_tiles promotes the new neighbors); the copy in input/ is
authoritative (_neighbor_laz_files searches input/ before
edge_neighbors/), so the duplicate is useless and can weigh several
hundred MB. ".part" files (download in progress) are never touched,
and a duplicate of a different size is kept (the copy in input/ could
be truncated: never delete what might be the only sound copy).
"""
from .dtm import EDGE_NEIGHBORS_DIRNAME
if not edge_dir.is_dir():
@ -551,9 +553,9 @@ class LidarArchaeoPipeline:
try:
size = path.stat().st_size
if twin.stat().st_size != size:
logger.warning(f"Raccord des bords : {path.name} présent dans input/ "
f"et {EDGE_NEIGHBORS_DIRNAME}/ avec des tailles "
f"différentes — doublon conservé")
logger.warning(f"Edge stitching: {path.name} present in input/ "
f"and {EDGE_NEIGHBORS_DIRNAME}/ with different "
f"sizes — duplicate kept")
continue
path.unlink()
freed += size
@ -561,12 +563,12 @@ class LidarArchaeoPipeline:
except OSError:
pass
if removed:
logger.info(f"Raccord des bords : {removed} doublon(s) supprimé(s) de "
f"{EDGE_NEIGHBORS_DIRNAME}/ (déjà dans input/, "
f"{freed / 1e6:.0f} Mo libérés)")
logger.info(f"Edge stitching: {removed} duplicate(s) removed from "
f"{EDGE_NEIGHBORS_DIRNAME}/ (already in input/, "
f"{freed / 1e6:.0f} MB freed)")
def _report(self, basename, phase, state, detail=None, res=None):
"""Émet un événement de progression (file de génération, best-effort)."""
"""Emit a progress event (generation queue, best effort)."""
report_event(self.output_dir, basename, phase, state, detail=detail, res=res)
def process_file(self, laz_file):
@ -582,13 +584,13 @@ class LidarArchaeoPipeline:
t_start = time.time()
logger.info("=" * 60)
logger.info(f"FICHIER : {basename}")
logger.info(f"FILE: {basename}")
logger.info("=" * 60)
# Validate file integrity before any processing
from .dtm import validate_laz
if not validate_laz(laz_file):
self._report(basename, "tile", "fail", "fichier LAZ invalide")
self._report(basename, "tile", "fail", "invalid LAZ file")
return False
# Step 1: Ground classification (shared across all resolutions)
@ -606,17 +608,17 @@ class LidarArchaeoPipeline:
dtm_path = self.dtm_dir / f"{basename}_dtm{res_suffix}.tif"
if dtm_path.exists() and not self.force_classify:
if not self._strip_align_matches(basename, res_suffix):
logger.info(f" DTM{res_suffix} sans calage de faisceaux conforme — régénération (offsets verticaux mesurés et appliqués)")
logger.info(f" DTM{res_suffix} lacks up-to-date strip alignment — regenerating (vertical offsets measured and applied)")
dtm_path.unlink()
elif not self._gap_fill_matches(dtm_path):
logger.info(f" DTM{res_suffix} au comblement d'une version antérieure — régénération "
f"(vides comblés dans l'enveloppe des points, rayon selon la densité)")
logger.info(f" DTM{res_suffix} gap-filled by an older version — regenerating "
f"(gaps filled within the point envelope, radius based on density)")
dtm_path.unlink()
elif not self._edge_buffer_matches(dtm_path):
from .dtm import read_dtm_edge_buffer
recorded = read_dtm_edge_buffer(dtm_path)
logger.info(f" DTM{res_suffix} avec raccord de {recorded:g} m ≠ {self.edge_buffer:g} m "
f"demandé — régénération (bande de bord des tuiles voisines)")
logger.info(f" DTM{res_suffix} has a {recorded:g} m edge buffer ≠ {self.edge_buffer:g} m "
f"requested — regenerating (edge band from neighboring tiles)")
dtm_path.unlink()
elif method_matches:
import rasterio
@ -624,61 +626,62 @@ class LidarArchaeoPipeline:
with rasterio.open(dtm_path) as src:
existing_res = abs(src.transform.a)
if abs(existing_res - res) > 0.01:
logger.info(f" DTM{res_suffix} existant à {existing_res}m/px — résolution demandée {res}m/px → régénération")
logger.info(f" Existing DTM{res_suffix} at {existing_res} m/px — requested resolution {res} m/px → regenerating")
dtm_path.unlink()
else:
if i == 0:
logger.info(f"[1/5] Classification du sol — sautée (DTM existant)")
logger.info(f"[2/5] Génération DTM {res}m/px — sautée (DTM existant)")
self._report(basename, "classif", "skip", "DTM existant")
# Sidecar qualité (densité sol, dates de vol) pour
# l'encart de l'export PDF : DTM réutilisé depuis le
# cache — la classification sol a été sautée, donc
# mesure directement sur le LAZ d'entrée avec les
# classes IGN du run (ensure_quality ne recalcule
# rien si le sidecar existe déjà : coût en régime
# établi = une lecture JSON).
logger.info("[1/5] Ground classification — skipped (existing DTM)")
logger.info(f"[2/5] DTM generation {res} m/px — skipped (existing DTM)")
self._report(basename, "classif", "skip", "existing DTM")
# Quality sidecar (ground density, flight dates) for
# the PDF export inset: DTM reused from the cache —
# ground classification was skipped, so measure
# directly on the input LAZ with the run's IGN
# classes (ensure_quality recomputes nothing if the
# sidecar already exists: steady-state cost = one
# JSON read).
from .dtm import parse_ign_classes
from .quality import ensure_quality
ensure_quality(laz_file, basename, self.output_dir,
codes=tuple(parse_ign_classes(self.ign_classes)))
else:
logger.info(f" DTM {res}m/px déjà existant — ignoré")
logger.info(f" DTM {res} m/px already exists — skipped")
self._report(basename, "dtm", "skip", res=res)
continue
except Exception:
logger.warning(f"Impossible de lire le DTM existant — régénération")
logger.warning("Cannot read the existing DTM — regenerating")
dtm_path.unlink()
else:
logger.info(f" DTM{res_suffix} produit par {self._dtm_method_name(basename, primary_suffix) or '?'} ≠ {self._effective_ground_method()} → reclassification")
logger.info(f" DTM{res_suffix} produced by {self._dtm_method_name(basename, primary_suffix) or '?'} ≠ {self._effective_ground_method()} → reclassifying")
dtm_path.unlink()
# Need to classify/generate DTM for this resolution
if las_file is None:
# First time: do ground classification
logger.info("[1/5] Classification du sol...")
logger.info("[1/5] Ground classification...")
self._report(basename, "classif", "start")
t1 = time.time()
las_file = classify_ground(laz_file, self.temp_dir, method=self.ground_method, force=self.force_classify, ign_classes=self.ign_classes)
t_classif = time.time() - t1
if not las_file:
logger.error(f" ✗ Échec classification ({t_classif:.1f}s)")
logger.error(f" ✗ Classification failed ({t_classif:.1f}s)")
self._report(basename, "classif", "fail")
self._report(basename, "tile", "fail", "échec classification")
self._report(basename, "tile", "fail", "classification failed")
return False
logger.info(f" ✓ Classification terminée ({t_classif:.1f}s)")
logger.info(f" ✓ Classification done ({t_classif:.1f}s)")
self._report(basename, "classif", "ok")
# Generate DTM at this resolution
logger.info(f"{'[2/5]' if i == 0 else ' '} Génération DTM {res}m/px...")
logger.info(f"{'[2/5]' if i == 0 else ' '} DTM generation {res} m/px...")
self._report(basename, "dtm", "start", res=res)
t2 = time.time()
# Classification IGN → mode pur : DTM = rasterisation brute des
# classes choisies, sans plancher ni comblement
# IGN classification → "pure" flag. create_dtm_fast now ignores it
# (kept for compatibility): small gaps are filled and strip
# alignment applies whatever the classification method.
pure_ign = "_ground_ign" in Path(las_file).name
# Classes des voisines pour la bande de raccord : mêmes classes IGN
# que le MNT (les voisines sont lues dans leur pré-classification
# fournisseur, quelle que soit la méthode de la tuile centrale).
# Neighbor classes for the edge band: same IGN classes as the DTM
# (neighbors are read in their provider pre-classification,
# whatever the method used for the central tile).
from .dtm import parse_ign_classes
neighbor_codes = parse_ign_classes(self.ign_classes)
dtm_file = create_dtm_fast(las_file, basename, self.dtm_dir, res,
@ -691,29 +694,29 @@ class LidarArchaeoPipeline:
neighbor_classes=neighbor_codes)
t_dtm = time.time() - t2
if not dtm_file:
logger.error(f" ✗ Échec DTM {res}m/px ({t_dtm:.1f}s)")
logger.error(f" ✗ DTM {res} m/px failed ({t_dtm:.1f}s)")
self._report(basename, "dtm", "fail", res=res)
if i == 0:
self._report(basename, "tile", "fail", "échec DTM")
self._report(basename, "tile", "fail", "DTM failed")
return False # Primary resolution failure is fatal
continue # Additional resolution failure is non-fatal
logger.info(f" ✓ DTM {res}m/px terminé ({t_dtm:.1f}s)")
logger.info(f" ✓ DTM {res} m/px done ({t_dtm:.1f}s)")
self._report(basename, "dtm", "ok", res=res)
dtm_rebuilt = True
self._write_dtm_method(basename, res_suffix)
if i == 0:
# Sidecar qualité (densité sol, dates de vol) pour l'encart de
# l'export PDF : DTM fraîchement généré — mesure sur le LAS sol
# déjà filtré (le cas DTM réutilisé depuis le cache est couvert
# plus haut, avant le `continue`).
# Quality sidecar (ground density, flight dates) for the PDF
# export inset: freshly generated DTM — measured on the already
# filtered ground LAS (the cached-DTM case is handled above,
# before the `continue`).
from .quality import ensure_quality
ensure_quality(las_file, basename, self.output_dir)
# Process each resolution: visualizations + PDF
# Option de calcul (surcharge le défaut du module) appliquée ICI car les
# workers (spawn) réimportent les modules à froid : c'est le seul endroit
# qui s'exécute dans le processus qui fait le calcul.
# Process each resolution: visualizations + image conversion
# Computation option (overrides the module default) applied HERE because
# workers (spawn) re-import modules from scratch: this is the only place
# that runs inside the process doing the computation.
if self.openness_downsample is not None:
from . import visualizations as _viz_mod
_viz_mod.OPENNESS_DOWNSAMPLE = max(1, int(self.openness_downsample))
@ -723,7 +726,7 @@ class LidarArchaeoPipeline:
dtm_path = self.dtm_dir / f"{basename}_dtm{res_suffix}.tif"
if not dtm_path.exists():
logger.warning(f" DTM {res}m/px manquant — visualisations ignorées")
logger.warning(f" DTM {res} m/px missing — visualizations skipped")
continue
import rasterio
@ -731,7 +734,7 @@ class LidarArchaeoPipeline:
actual_res = abs(src.transform.a)
if len(self.resolutions) > 1:
logger.info(f" --- Résolution {res}m/px ---")
logger.info(f" --- Resolution {res} m/px ---")
# For additional resolutions, use suffixed subdirectory
if res_suffix:
@ -746,26 +749,27 @@ class LidarArchaeoPipeline:
force_images=self.force or self.force_classify or dtm_rebuilt)
t_total = time.time() - t_start
logger.info(f"✓ {basename} terminé en {t_total:.1f}s")
logger.info(f"✓ {basename} done in {t_total:.1f}s")
self._report(basename, "tile", "ok", f"{t_total:.0f}s")
_file_filter.basename = None
return True
def _rebuild_index_incremental(self):
"""Régénère la carte juste après une tuile terminée (mode incrémental).
"""Rebuild the map inventory right after a tile finishes (incremental mode).
Réécrit index_tiles.json (et rafraîchit vignettes/sous-tuiles) pour
que la carte affiche la tuile sans attendre la fin du run. Anti-rebond : 3 s minimum entre
deux passes — les tuiles terminées pendant l'intervalle sont couvertes
par la passe suivante ou par la passe finale. Les logs de build_index
sont masqués pour ne pas noyer le journal du run.
Rewrites index_tiles.json (and refreshes thumbnails/subtiles) so the
map shows the tile without waiting for the end of the run. Debounce:
at least 3 s between two passes — tiles finished during the interval
are covered by a deferred pass scheduled at the end of the interval
(or by the final pass). build_index logs are silenced so they do not
flood the run log.
"""
if self.no_index:
return
wait = 3.0 - (time.time() - self._last_index_rebuild)
if wait > 0:
# Anti-rebond : la dalle n'est pas oubliée, une passe différée la
# couvre dès la fin de l'intervalle (sans attendre la dalle suivante).
# Debounce: the tile is not forgotten, a deferred pass covers it as
# soon as the interval ends (without waiting for the next tile).
with self._index_lock:
if self._index_timer is None:
self._index_timer = threading.Timer(wait, self._rebuild_index_now)
@ -787,23 +791,23 @@ class LidarArchaeoPipeline:
finally:
logger.setLevel(saved_level)
except Exception as e:
logger.debug(f"Rebuild incrémental de l'index ignoré : {e}")
logger.debug(f"Incremental index rebuild skipped: {e}")
def process_all(self, files=None):
"""Process all LAZ files in input directory (or an explicit list)."""
files = files if files is not None else self.find_laz_files()
if not files:
logger.error("Aucun fichier LAZ/LAS trouvé !")
logger.error("No LAZ/LAS file found!")
return
logger.info("=" * 60)
logger.info("PIPELINE ARCHÉOLOGIQUE LiDAR")
logger.info("LiDAR ARCHAEOLOGICAL PIPELINE")
logger.info("=" * 60)
logger.info("Vérification des outils...")
logger.info("Checking tools...")
if not self.check_tools():
logger.error("Outils manquants — abandon")
logger.error("Missing tools — aborting")
return
results = {}
@ -813,29 +817,29 @@ class LidarArchaeoPipeline:
if self.gpu_ids is not None:
restrict_gpus(self.gpu_ids)
# Raccord des bords : pré-télécharger les voisines manquantes avant
# les workers (no-op si le raccord est désactivé).
# Edge stitching: pre-download missing neighbors before starting the
# workers (no-op when edge stitching is disabled).
self._fetch_edge_neighbors(files)
if self.workers > 1 and len(files) > 1:
n_gpus = num_gpus() or 1
if n_gpus > 1:
logger.info(f"Traitement parallèle avec {self.workers} workers sur {n_gpus} GPUs...")
logger.info(f"Parallel processing with {self.workers} workers on {n_gpus} GPUs...")
else:
logger.info(f"Traitement parallèle avec {self.workers} workers...")
logger.info(f"Fichiers: {len(files)}")
logger.info(f"Parallel processing with {self.workers} workers...")
logger.info(f"Files: {len(files)}")
# Une place fixe par processus du pool (GPU ou CPU), prise à sa
# création : au plus « VRAM libre / pic d'un worker » processus par
# GPU. L'affectation par numéro de fichier (round-robin) laissait
# 6 workers sur chaque GPU de 8 Go avec LIDAR_WORKERS=auto : OOM.
# One fixed slot per pool process (GPU or CPU), taken when the
# process starts: at most "free VRAM / peak per worker" processes
# per GPU. Assignment by file number (round-robin) used to put
# 6 workers on each 8 GB GPU with LIDAR_WORKERS=auto: OOM.
active_ids = self.gpu_ids if self.gpu_ids else available_gpu_ids()
slots = gpu_worker_slots(active_ids, self.workers)
if active_ids:
per_gpu = ", ".join(f"GPU {g} : {slots.count(g)}" for g in active_ids)
per_gpu = ", ".join(f"GPU {g}: {slots.count(g)}" for g in active_ids)
n_cpu = slots.count(-1)
logger.info(f"Répartition des workers : {per_gpu}"
+ (f", CPU : {n_cpu} (VRAM insuffisante)" if n_cpu else ""))
logger.info(f"Worker distribution: {per_gpu}"
+ (f", CPU: {n_cpu} (insufficient VRAM)" if n_cpu else ""))
slot_queue = multiprocessing.Queue()
for slot in slots:
slot_queue.put(slot)
@ -850,9 +854,9 @@ class LidarArchaeoPipeline:
done = 0
t_deadline = time.time() + 7200
try:
# timeout= : sans lui, as_completed bloque entre deux
# complétions et le délai de 2 h n'est jamais évalué si
# aucun worker ne rend la main (run figé pour toujours).
# timeout=: without it, as_completed blocks between two
# completions and the 2 h deadline is never evaluated if
# no worker returns (run stuck forever).
try:
futures_iter = as_completed(future_to_file,
timeout=max(1.0, t_deadline - time.time()))
@ -868,38 +872,38 @@ class LidarArchaeoPipeline:
self._rebuild_index_incremental()
except Exception as e:
logger.error(f" [{done}/{len(files)}] ✗ {laz_file.name}: {e}")
logger.debug(f" Traceback:", exc_info=True)
logger.debug(" Traceback:", exc_info=True)
report_event(self.output_dir, _file_basename(laz_file),
"tile", "fail", detail=str(e))
results[laz_file.name] = False
except FuturesTimeoutError:
logger.error("Délai dépassé (2h) — annulation des workers restants")
logger.error("Timeout exceeded (2 h) — cancelling remaining workers")
for f in future_to_file:
f.cancel()
except KeyboardInterrupt:
logger.info("Interruption — annulation des travaux en cours...")
logger.info("Interrupted — cancelling running jobs...")
for f in future_to_file:
f.cancel()
executor.shutdown(wait=False, cancel_futures=True)
logger.info("Travaux annulés.")
logger.info("Jobs cancelled.")
return
else:
total = len(files)
if self.workers == 1 and len(files) > 1:
n_gpus = num_gpus() or 1
if n_gpus > 1:
logger.info(f"Conseil : utilisez -w {n_gpus} pour exploiter tous les GPUs")
logger.info(f"Tip: use -w {n_gpus} to take advantage of all GPUs")
for idx, laz_file in enumerate(files, 1):
logger.info(f"--- Fichier {idx}/{total} ---")
logger.info(f"--- File {idx}/{total} ---")
try:
results[laz_file.name] = self.process_file(laz_file)
if results[laz_file.name] and self.incremental_index:
self._rebuild_index_incremental()
except KeyboardInterrupt:
logger.info("Interruption — arrêt immédiat.")
logger.info("Interrupted — stopping immediately.")
return
except Exception as e:
logger.error(f"✗ Erreur traitement {laz_file.name}: {e}")
logger.error(f"✗ Error processing {laz_file.name}: {e}")
logger.debug("Traceback:", exc_info=True)
report_event(self.output_dir, _file_basename(laz_file),
"tile", "fail", detail=str(e))
@ -911,46 +915,46 @@ class LidarArchaeoPipeline:
fail_count = sum(1 for v in results.values() if not v)
logger.info("=" * 60)
logger.info("RÉSUMÉ")
logger.info("SUMMARY")
logger.info("=" * 60)
for name, ok in results.items():
status = "✓" if ok else "✗"
logger.info(f" {status} {name}")
logger.info("-" * 60)
logger.info(f" Succès : {success_count}/{len(results)}")
logger.info(f" Succeeded: {success_count}/{len(results)}")
if fail_count:
logger.info(f" Échecs : {fail_count}/{len(results)}")
logger.info(f" Durée totale : {t_pipeline_total:.1f}s ({t_pipeline_total/60:.1f}min)")
logger.info(f" Failed: {fail_count}/{len(results)}")
logger.info(f" Total time: {t_pipeline_total:.1f}s ({t_pipeline_total/60:.1f} min)")
logger.info(f"\nRésultats dans: {self.output_dir}")
logger.info(f"\nResults in: {self.output_dir}")
logger.info(f" • DTM : {self.dtm_dir}")
logger.info(f" • Visualisations: {self.vis_dir}")
logger.info(f" • Visualizations: {self.vis_dir}")
# Génère le catalogue des tuiles traitées (vignettes + inventaire)
# Build the catalog of processed tiles (thumbnails + inventory)
if not self.no_index:
try:
from .index import build_index
index_path = build_index(self.output_dir, self.output_format)
if index_path:
logger.info(f" • Catalogue : {index_path}")
logger.info(f" • Catalog: {index_path}")
except Exception as e:
logger.warning(f"Index global non généré: {e}")
logger.warning(f"Global index not generated: {e}")
# Clean up temporary files
logger.info("Nettoyage des fichiers temporaires...")
logger.info("Cleaning up temporary files...")
try:
if self.temp_dir.exists():
shutil.rmtree(self.temp_dir)
logger.info(" ✓ Fichiers temporaires supprimés")
logger.info(" ✓ Temporary files removed")
except Exception as e:
logger.warning(f" Note: Impossible de supprimer les fichiers temporaires: {e}")
logger.warning(f" Note: could not remove temporary files: {e}")
def _init_worker_slot(slot_queue):
"""Initialiseur du pool : le processus prend sa place une fois pour toutes.
"""Pool initializer: the process takes its slot once and for all.
Indice GPU → set_active_gpu ; -1 → CPU forcé (VRAM insuffisante) ;
None ou file vide → choix laissé au worker (pas de GPU détecté).
GPU index → set_active_gpu; -1 → forced CPU (insufficient VRAM);
None or empty queue → left to the worker (no GPU detected).
"""
import queue
from . import gpu
@ -970,8 +974,9 @@ def _process_file_standalone(laz_file_str, input_dir, output_dir, resolution, fo
"""Standalone function for multiprocessing — creates its own pipeline instance.
Each worker gets its own temp directory to avoid file conflicts.
When multiple GPUs are available, each worker is assigned a GPU via
CUDA_VISIBLE_DEVICES to balance load across GPUs.
The GPU is normally assigned once per pool process by _init_worker_slot
(VRAM-bounded slots); gpu_id (process_all passes None) only forces a GPU
for a direct call.
"""
if gpu_id is not None and gpu_id >= 0:
from .gpu import set_active_gpu