Files
lidar_rendu/lidar_pipeline/cli.py
Antoine fb892ea9f2 Translate the whole project to English and fix outdated comments and help
Comments, docstrings, logs, CLI help, map UI, legends, PDF sheet, scripts,
compose files and AGENTS.md are now English. Data keys stay unchanged
(relief_oriente, densite_sol, visualisations/, API JSON keys, link params).
Wrong comments and help defaults found along the way are corrected.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-27 23:16:45 +02:00

475 lines
18 KiB
Python

"""Command-line interface for the LiDAR archaeological pipeline.
Handles argument parsing, logging configuration, and entry point.
"""
import argparse
import logging
import signal
import sys
from .pipeline import resolve_workers
from .pipeline import LidarArchaeoPipeline
from .gpu import log_gpu_status
logger = logging.getLogger("lidar")
# pkill pattern for the PDAL processes of THIS run (set in main) — see
# _kill_orphan_pdal: scoped to our output to spare other runs.
_pdal_kill_pattern = None
def setup_logging(verbose=False, debug=False):
"""Configure the 'lidar' logger.
Args:
verbose: If True, include timestamps and level names.
debug: If True, set level to DEBUG and add file:line info.
"""
# Ensure UTF-8 output for non-ASCII characters in messages (✓, ✗, →, ≠, etc.)
if hasattr(sys.stdout, 'reconfigure'):
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
if hasattr(sys.stderr, 'reconfigure'):
sys.stderr.reconfigure(encoding='utf-8', errors='replace')
if debug:
level = logging.DEBUG
fmt = "%(asctime)s.%(msecs)03d %(levelname)-5s [%(filename)s:%(lineno)d] %(message)s"
elif verbose:
level = logging.INFO
fmt = "%(asctime)s %(levelname)-5s %(message)s"
else:
level = logging.INFO
fmt = "%(message)s"
handler = logging.StreamHandler(sys.stdout)
handler.setFormatter(logging.Formatter(fmt, datefmt="%H:%M:%S"))
logger.setLevel(level)
logger.handlers.clear()
logger.addHandler(handler)
logger.propagate = False # Prevent double logging via root logger
# Also configure the root logger so worker processes log properly
root_logger = logging.getLogger()
root_logger.setLevel(level)
if not root_logger.handlers:
root_handler = logging.StreamHandler(sys.stdout)
root_handler.setFormatter(logging.Formatter(fmt, datefmt="%H:%M:%S"))
root_logger.addHandler(root_handler)
return logger
def main():
"""Entry point for the LiDAR archaeological pipeline."""
parser = argparse.ArgumentParser(
description="LiDAR pipeline for archaeological detection",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""\
Examples:
Standard processing (0.2 m/px):
python -m lidar_pipeline /data/input -o /data/output
Coarser 0.5 m/px resolution:
python -m lidar_pipeline /data/input -o /data/output -r 0.5
Select a specific GPU (index 0):
python -m lidar_pipeline /data/input -o /data/output -g 0
Select several GPUs (indices 0 and 2):
python -m lidar_pipeline /data/input -o /data/output -g 0,2
Use all available GPUs:
python -m lidar_pipeline /data/input -o /data/output -g all
Verbose mode (timestamps):
python -m lidar_pipeline /data/input -o /data/output -v
Debug mode (internal details):
python -m lidar_pipeline /data/input -o /data/output --debug
Force regeneration of all files:
python -m lidar_pipeline /data/input -o /data/output --force
Process a single file (for testing):
python -m lidar_pipeline /data/input -o /data/output --file LHD_FXX_1000_6881_PTS_LAMB93_IGN69.copc
Parallel processing (4 workers):
python -m lidar_pipeline /data/input -o /data/output -w 4
"""
)
parser.add_argument(
"input",
nargs="?",
default="/data/input",
help="Directory containing the LAZ/LAS files (default: /data/input; "
"optional with --rebuild-index)"
)
parser.add_argument(
"-o", "--output",
default="/data/output",
help="Output directory (default: /data/output)"
)
parser.add_argument(
"-r", "--resolution",
type=str,
default="0.2",
help="Resolution in m/px, or several comma-separated values (default: 0.2)"
)
parser.add_argument(
"-w", "--workers",
type=str,
default="auto",
help="Number of workers for parallel processing, or \"auto\" = "
"cores - 2 clamped to [2, 16], evaluated when the run starts "
"(default: auto)"
)
parser.add_argument(
"-g", "--gpu",
nargs="?",
const="all",
default=None,
metavar="LIST",
help="Select the GPU(s) to use: an index (e.g. -g 0), "
"a list (e.g. -g 0,2), or 'all' for all of them. "
"Without -g: all available GPUs; workers are spread across them "
"within the free VRAM, the excess runs on CPU."
)
parser.add_argument(
"-f", "--force",
action="store_true",
help="Regenerate all files even if the output images (AVIF/WebP) already exist"
)
parser.add_argument(
"--force-classification",
action="store_true",
help="Reclassify the ground even if the method is unchanged (also regenerates the "
"DTM and the images). Without this flag, changing --ground-classification is "
"enough: the recorded method is compared and a change triggers reclassification."
)
parser.add_argument(
"--openness-downsample",
type=int,
default=None,
metavar="FACTOR",
help="Downsampling factor for the openness computation (ray tracing): "
"block-decimated grid, then resampling. Default: 2 (cost divided by "
"factor³, ~13x faster measured, near-identical rendering); 1 = full resolution"
)
parser.add_argument(
"--edge-buffer",
type=float,
default=100.0,
metavar="METERS",
help="Edge stitching: extend the DTM by an N-meter band filled with the "
"ground points of the 8 neighboring LAZ tiles, so that large-kernel "
"renders (openness, SVF, LRM) are continuous from one tile to the next. "
"100 m covers all radii; final images are cropped to the exact 1 km "
"tile. Default: 100 (always applied in production); 0 = disabled. "
"Changing the value regenerates the affected DTMs."
)
parser.add_argument(
"--no-strip-align",
action="store_true",
help="Disable vertical alignment of flight strips. By default, vertical offsets "
"≥ 0.5 cm between the flight strips (PointSourceId) of a tile are measured on "
"the ground points and corrected before DTM rasterization, then each scan line "
"is adjusted jointly (offset and roll) against the other strips, with a "
"per-strip scan-angle calibration profile (GPS-time window jitter correction "
"when scan_angle is missing); corrections are recorded in "
"DTM/*_stripalign.json"
)
parser.add_argument(
"--keep-tif",
action="store_true",
help="Keep the TIFF files (DTM + visualizations) so the output images can be regenerated without recomputing"
)
parser.add_argument(
"--ground-classification",
choices=["auto", "ign", "smrf", "csf"],
default="ign",
help="Ground classification method: auto (prefers the IGN pre-classification when "
"present — fast path — otherwise SMRF/CSF detection), ign, smrf, csf. "
"With ign, the DTM is rasterized from the chosen classes (--ign-classes); "
"with smrf/csf, ground is classified by PDAL first. In every case small gaps "
"are filled within the point envelope. "
"(default: ign, enforced in production)"
)
parser.add_argument(
"--ign-classes",
default="sol",
help="LAS classes extracted for the DTM with the IGN classification (ign/auto "
"method): comma-separated list of names or codes — "
"sol(2, ground), unclassified(1), eau(9, water), virtuel(66, virtual), "
"pont(17, bridge), sursol(64, above-ground). "
"E.g. --ign-classes sol,unclassified. Changing the list reclassifies the "
"affected tiles. (default: sol)"
)
parser.add_argument(
"--quality",
type=int,
default=60,
help="Image quality (1-100, default: 60). 100 requests lossless (honored in WebP; "
"Pillow ignores it in AVIF)."
)
parser.add_argument(
"--lossless",
action="store_true",
help="Force lossless compression (same as --quality 100)"
)
parser.add_argument(
"--format",
choices=["webp", "avif"],
default="avif",
help="Output format: avif (default, better quality) or webp"
)
parser.add_argument(
"--only",
nargs="+",
type=str,
default=None,
help="Generate only these visualizations (e.g. --only hillshade svf mslrm)"
)
parser.add_argument(
"--skip",
nargs="+",
type=str,
default=None,
help="Exclude these visualizations (e.g. --skip ortho topo)"
)
parser.add_argument(
"--file",
nargs="+",
type=str,
default=None,
help="Process one or more LAZ/LAS files (full name without extension, e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69.copc)"
)
parser.add_argument(
"--fetch-tiles",
nargs="+",
default=None,
metavar="COL,ROW",
help="Download these LiDAR HD tiles from IGN before processing "
"(tiles not generated yet, e.g. --fetch-tiles 1055,6882 1056,6883)"
)
parser.add_argument(
"-v", "--verbose",
action="store_true",
help="Verbose mode: show timestamps and levels"
)
parser.add_argument(
"--debug",
action="store_true",
help="Debug mode: show internal details (file:line)"
)
parser.add_argument(
"--rebuild-index",
action="store_true",
help="Only rebuild the catalog of processed tiles (thumbnails, subtiles, inventory — no reprocessing)"
)
parser.add_argument(
"--no-index",
action="store_true",
help="Do not build the catalog (thumbnails + inventory), neither after each tile nor at the end of processing"
)
parser.add_argument(
"--incremental-index",
action="store_true",
help="Kept for compatibility: the catalog is now rebuilt after every "
"finished tile by default (unless --no-index)"
)
parser.add_argument(
"--quality-backfill",
action="store_true",
help="Only compute the quality sidecars (ground density, flight dates) "
"of the input LAZ files that lack one, then rebuild the inventory "
"(no DTM or visualization, ~5 s per tile)"
)
args = parser.parse_args()
# Configure logging before any other output
setup_logging(verbose=args.verbose, debug=args.debug)
# Add file prefix filter for parallel processing
from .pipeline import _file_filter
logger.addFilter(_file_filter)
logger.info("=" * 60)
logger.info("LiDAR Archaeological Pipeline")
logger.info("=" * 60)
# Parse --gpu into a list of GPU IDs
gpu_ids = None
gpu_arg = args.gpu
if gpu_arg is not None:
if gpu_arg == 'all':
gpu_ids = None # Don't restrict — use all GPUs
else:
try:
gpu_ids = [int(g.strip()) for g in gpu_arg.split(',')]
except ValueError:
parser.error(f"Invalid GPU: {gpu_arg!r}. Use an index, a list (0,2) or 'all'.")
if gpu_ids is not None:
from .gpu import restrict_gpus
restrict_gpus(gpu_ids)
# Kill orphan PDAL processes on interrupt or termination. The pattern is
# scoped to OUR temp directory (pdal pipeline <output>/temp/...): a global
# pkill would slaughter the PDAL processes of other runs on the machine
# (e.g. a concurrent generation job). No atexit hook: a normal exit has no
# reason to kill anything.
global _pdal_kill_pattern
_pdal_kill_pattern = f"pdal pipeline {args.output}/temp/"
signal.signal(signal.SIGINT, _kill_orphan_pdal)
signal.signal(signal.SIGTERM, _kill_orphan_pdal)
log_gpu_status()
try:
# --rebuild-index mode: only rebuild the catalog, no reprocessing
if args.rebuild_index:
from .index import build_index
index_path = build_index(args.output, args.format)
if index_path:
logger.info(f"Catalog built: {index_path}")
else:
logger.warning("No processed tile found — catalog not built")
return
if args.quality_backfill:
from .dtm import parse_ign_classes
from .index import build_index
from .quality import backfill_quality
n = backfill_quality(args.input, args.output,
codes=tuple(parse_ign_classes(args.ign_classes)))
logger.info(f"{n} quality sidecar(s) written")
build_index(args.output, args.format)
return
# New run: reset the event log (generation queue). Done here, before
# downloading — download events survive the start of processing
# (process_all no longer truncates the log).
from .progress import reset_events
reset_events(args.output)
quality = 100 if args.lossless else args.quality
# Parse --only and --skip: accept comma-separated values
only_viz = None
if args.only:
only_viz = [v.strip() for item in args.only for v in item.split(',')]
skip_viz = None
if args.skip:
skip_viz = [v.strip() for item in args.skip for v in item.split(',')]
resolutions = [float(r.strip()) for r in args.resolution.split(',') if r.strip()]
# Download missing IGN tiles before processing.
# only_viz/resolutions: a tile is skipped only if it already has all
# the requested visualizations — an incomplete tile is
# (re)downloaded to generate its missing visualizations.
if args.fetch_tiles:
from .fetch_ign import fetch_tiles, parse_tile_specs
try:
specs = parse_tile_specs(args.fetch_tiles)
except ValueError as e:
logger.error(str(e))
return
logger.info(f"Downloading {len(specs)} LiDAR HD tile(s) from IGN...")
fetched = fetch_tiles(args.input, specs, args.output,
only_viz=only_viz, resolutions=resolutions,
force=args.force or args.force_classification)
if fetched:
logger.info(f"{len(fetched)} tile(s) downloaded — processing...")
else:
logger.warning("No tile downloaded (already present or not found)")
pipeline = LidarArchaeoPipeline(
input_dir=args.input,
output_dir=args.output,
resolution=args.resolution,
workers=resolve_workers(args.workers),
force=args.force,
ground_method=args.ground_classification,
ign_classes=args.ign_classes,
force_classify=args.force_classification,
keep_tif=args.keep_tif,
strip_align=not args.no_strip_align,
openness_downsample=args.openness_downsample,
edge_buffer=args.edge_buffer,
quality=quality,
only_viz=only_viz,
skip_viz=skip_viz,
output_format=args.format,
gpu_ids=gpu_ids,
no_index=args.no_index,
incremental_index=args.incremental_index,
)
# If --file is specified, process only matching files
if args.file:
from pathlib import Path
input_dir = Path(args.input)
# Each pattern is the full filename without extension (e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69.copc)
# Also supports bare name without .copc (e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69)
selected_files = []
for pattern in args.file:
# Try exact filename first (e.g. LHD_FXX_...copc.laz)
exact_match = input_dir / pattern
if exact_match.exists() and exact_match.is_file():
matches = [exact_match]
else:
# Try with added extensions (e.g. pattern=LHD_FXX_...IGN69)
matches = (list(input_dir.glob(f"{pattern}.laz"))
+ list(input_dir.glob(f"{pattern}.las"))
+ list(input_dir.glob(f"{pattern}.copc.laz"))
+ list(input_dir.glob(f"{pattern}.copc.las")))
# Remove duplicates
matches = list(dict.fromkeys(matches))
if not matches:
logger.warning(f"No file found for: {pattern}")
continue
selected_files.extend(matches)
# Remove duplicates across patterns
seen = set()
unique_files = []
for f in selected_files:
if f not in seen:
seen.add(f)
unique_files.append(f)
if not unique_files:
logger.error("No file found for the given patterns")
sys.exit(1)
logger.info(f"Processing {len(unique_files)} selected file(s)")
for laz_file in unique_files:
logger.info(f" → {laz_file.name}")
# Reuse process_all: parallel workers, summary, index, cleanup
pipeline.process_all(files=unique_files)
else:
pipeline.process_all()
except Exception as e:
logger.error(f"Fatal error: {e}", exc_info=True)
sys.exit(1)
def _kill_orphan_pdal(signum=None, frame=None):
"""Kill orphan PDAL processes of THIS run on interrupt or termination."""
import subprocess
# Belt-and-suspenders: os.killpg below is the primary mechanism.
# This handles the edge case where a child escapes the process group.
# Pattern scoped to our output (see main) so the PDAL processes of other
# runs on the machine are not killed.
if _pdal_kill_pattern:
try:
import re
subprocess.run(["pkill", "-9", "-f", re.escape(_pdal_kill_pattern)],
capture_output=True, timeout=3)
except Exception:
pass
if signum is not None:
logger.info("Interrupted — cleaning up processes")
# Force-kill all child processes immediately
try:
import os
os.killpg(os.getpgrp(), signal.SIGKILL)
except Exception:
pass
sys.exit(130)