"""Command-line interface for the LiDAR archaeological pipeline. Handles argument parsing, logging configuration, and entry point. """ import argparse import logging import signal import sys from .pipeline import resolve_workers from .pipeline import LidarArchaeoPipeline from .gpu import log_gpu_status logger = logging.getLogger("lidar") # pkill pattern for the PDAL processes of THIS run (set in main) — see # _kill_orphan_pdal: scoped to our output to spare other runs. _pdal_kill_pattern = None def setup_logging(verbose=False, debug=False): """Configure the 'lidar' logger. Args: verbose: If True, include timestamps and level names. debug: If True, set level to DEBUG and add file:line info. """ # Ensure UTF-8 output for non-ASCII characters in messages (✓, ✗, →, ≠, etc.) if hasattr(sys.stdout, 'reconfigure'): sys.stdout.reconfigure(encoding='utf-8', errors='replace') if hasattr(sys.stderr, 'reconfigure'): sys.stderr.reconfigure(encoding='utf-8', errors='replace') if debug: level = logging.DEBUG fmt = "%(asctime)s.%(msecs)03d %(levelname)-5s [%(filename)s:%(lineno)d] %(message)s" elif verbose: level = logging.INFO fmt = "%(asctime)s %(levelname)-5s %(message)s" else: level = logging.INFO fmt = "%(message)s" handler = logging.StreamHandler(sys.stdout) handler.setFormatter(logging.Formatter(fmt, datefmt="%H:%M:%S")) logger.setLevel(level) logger.handlers.clear() logger.addHandler(handler) logger.propagate = False # Prevent double logging via root logger # Also configure the root logger so worker processes log properly root_logger = logging.getLogger() root_logger.setLevel(level) if not root_logger.handlers: root_handler = logging.StreamHandler(sys.stdout) root_handler.setFormatter(logging.Formatter(fmt, datefmt="%H:%M:%S")) root_logger.addHandler(root_handler) return logger def main(): """Entry point for the LiDAR archaeological pipeline.""" parser = argparse.ArgumentParser( description="LiDAR pipeline for archaeological detection", formatter_class=argparse.RawDescriptionHelpFormatter, epilog="""\ Examples: Standard processing (0.2 m/px): python -m lidar_pipeline /data/input -o /data/output Coarser 0.5 m/px resolution: python -m lidar_pipeline /data/input -o /data/output -r 0.5 Select a specific GPU (index 0): python -m lidar_pipeline /data/input -o /data/output -g 0 Select several GPUs (indices 0 and 2): python -m lidar_pipeline /data/input -o /data/output -g 0,2 Use all available GPUs: python -m lidar_pipeline /data/input -o /data/output -g all Verbose mode (timestamps): python -m lidar_pipeline /data/input -o /data/output -v Debug mode (internal details): python -m lidar_pipeline /data/input -o /data/output --debug Force regeneration of all files: python -m lidar_pipeline /data/input -o /data/output --force Process a single file (for testing): python -m lidar_pipeline /data/input -o /data/output --file LHD_FXX_1000_6881_PTS_LAMB93_IGN69.copc Parallel processing (4 workers): python -m lidar_pipeline /data/input -o /data/output -w 4 """ ) parser.add_argument( "input", nargs="?", default="/data/input", help="Directory containing the LAZ/LAS files (default: /data/input; " "optional with --rebuild-index)" ) parser.add_argument( "-o", "--output", default="/data/output", help="Output directory (default: /data/output)" ) parser.add_argument( "-r", "--resolution", type=str, default="0.2", help="Resolution in m/px, or several comma-separated values (default: 0.2)" ) parser.add_argument( "-w", "--workers", type=str, default="auto", help="Number of workers for parallel processing, or \"auto\" = " "cores - 2 clamped to [2, 16], evaluated when the run starts " "(default: auto)" ) parser.add_argument( "-g", "--gpu", nargs="?", const="all", default=None, metavar="LIST", help="Select the GPU(s) to use: an index (e.g. -g 0), " "a list (e.g. -g 0,2), or 'all' for all of them. " "Without -g: all available GPUs; workers are spread across them " "within the free VRAM, the excess runs on CPU." ) parser.add_argument( "-f", "--force", action="store_true", help="Regenerate all files even if the output images (AVIF/WebP) already exist" ) parser.add_argument( "--force-classification", action="store_true", help="Reclassify the ground even if the method is unchanged (also regenerates the " "DTM and the images). Without this flag, changing --ground-classification is " "enough: the recorded method is compared and a change triggers reclassification." ) parser.add_argument( "--openness-downsample", type=int, default=None, metavar="FACTOR", help="Downsampling factor for the openness computation (ray tracing): " "block-decimated grid, then resampling. Default: 2 (cost divided by " "factor³, ~13x faster measured, near-identical rendering); 1 = full resolution" ) parser.add_argument( "--edge-buffer", type=float, default=100.0, metavar="METERS", help="Edge stitching: extend the DTM by an N-meter band filled with the " "ground points of the 8 neighboring LAZ tiles, so that large-kernel " "renders (openness, SVF, LRM) are continuous from one tile to the next. " "100 m covers all radii; final images are cropped to the exact 1 km " "tile. Default: 100 (always applied in production); 0 = disabled. " "Changing the value regenerates the affected DTMs." ) parser.add_argument( "--no-strip-align", action="store_true", help="Disable vertical alignment of flight strips. By default, vertical offsets " "≥ 0.5 cm between the flight strips (PointSourceId) of a tile are measured on " "the ground points and corrected before DTM rasterization, then each scan line " "is adjusted jointly (offset and roll) against the other strips, with a " "per-strip scan-angle calibration profile (GPS-time window jitter correction " "when scan_angle is missing); corrections are recorded in " "DTM/*_stripalign.json" ) parser.add_argument( "--keep-tif", action="store_true", help="Keep the TIFF files (DTM + visualizations) so the output images can be regenerated without recomputing" ) parser.add_argument( "--ground-classification", choices=["auto", "ign", "smrf", "csf"], default="ign", help="Ground classification method: auto (prefers the IGN pre-classification when " "present — fast path — otherwise SMRF/CSF detection), ign, smrf, csf. " "With ign, the DTM is rasterized from the chosen classes (--ign-classes); " "with smrf/csf, ground is classified by PDAL first. In every case small gaps " "are filled within the point envelope. " "(default: ign, enforced in production)" ) parser.add_argument( "--ign-classes", default="sol", help="LAS classes extracted for the DTM with the IGN classification (ign/auto " "method): comma-separated list of names or codes — " "sol(2, ground), unclassified(1), eau(9, water), virtuel(66, virtual), " "pont(17, bridge), sursol(64, above-ground). " "E.g. --ign-classes sol,unclassified. Changing the list reclassifies the " "affected tiles. (default: sol)" ) parser.add_argument( "--quality", type=int, default=60, help="Image quality (1-100, default: 60). 100 requests lossless (honored in WebP; " "Pillow ignores it in AVIF)." ) parser.add_argument( "--lossless", action="store_true", help="Force lossless compression (same as --quality 100)" ) parser.add_argument( "--format", choices=["webp", "avif"], default="avif", help="Output format: avif (default, better quality) or webp" ) parser.add_argument( "--only", nargs="+", type=str, default=None, help="Generate only these visualizations (e.g. --only hillshade svf mslrm)" ) parser.add_argument( "--skip", nargs="+", type=str, default=None, help="Exclude these visualizations (e.g. --skip ortho topo)" ) parser.add_argument( "--file", nargs="+", type=str, default=None, help="Process one or more LAZ/LAS files (full name without extension, e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69.copc)" ) parser.add_argument( "--fetch-tiles", nargs="+", default=None, metavar="COL,ROW", help="Download these LiDAR HD tiles from IGN before processing " "(tiles not generated yet, e.g. --fetch-tiles 1055,6882 1056,6883)" ) parser.add_argument( "-v", "--verbose", action="store_true", help="Verbose mode: show timestamps and levels" ) parser.add_argument( "--debug", action="store_true", help="Debug mode: show internal details (file:line)" ) parser.add_argument( "--rebuild-index", action="store_true", help="Only rebuild the catalog of processed tiles (thumbnails, subtiles, inventory — no reprocessing)" ) parser.add_argument( "--no-index", action="store_true", help="Do not build the catalog (thumbnails + inventory), neither after each tile nor at the end of processing" ) parser.add_argument( "--incremental-index", action="store_true", help="Kept for compatibility: the catalog is now rebuilt after every " "finished tile by default (unless --no-index)" ) parser.add_argument( "--quality-backfill", action="store_true", help="Only compute the quality sidecars (ground density, flight dates) " "of the input LAZ files that lack one, then rebuild the inventory " "(no DTM or visualization, ~5 s per tile)" ) args = parser.parse_args() # Configure logging before any other output setup_logging(verbose=args.verbose, debug=args.debug) # Add file prefix filter for parallel processing from .pipeline import _file_filter logger.addFilter(_file_filter) logger.info("=" * 60) logger.info("LiDAR Archaeological Pipeline") logger.info("=" * 60) # Parse --gpu into a list of GPU IDs gpu_ids = None gpu_arg = args.gpu if gpu_arg is not None: if gpu_arg == 'all': gpu_ids = None # Don't restrict — use all GPUs else: try: gpu_ids = [int(g.strip()) for g in gpu_arg.split(',')] except ValueError: parser.error(f"Invalid GPU: {gpu_arg!r}. Use an index, a list (0,2) or 'all'.") if gpu_ids is not None: from .gpu import restrict_gpus restrict_gpus(gpu_ids) # Kill orphan PDAL processes on interrupt or termination. The pattern is # scoped to OUR temp directory (pdal pipeline /temp/...): a global # pkill would slaughter the PDAL processes of other runs on the machine # (e.g. a concurrent generation job). No atexit hook: a normal exit has no # reason to kill anything. global _pdal_kill_pattern _pdal_kill_pattern = f"pdal pipeline {args.output}/temp/" signal.signal(signal.SIGINT, _kill_orphan_pdal) signal.signal(signal.SIGTERM, _kill_orphan_pdal) log_gpu_status() try: # --rebuild-index mode: only rebuild the catalog, no reprocessing if args.rebuild_index: from .index import build_index index_path = build_index(args.output, args.format) if index_path: logger.info(f"Catalog built: {index_path}") else: logger.warning("No processed tile found — catalog not built") return if args.quality_backfill: from .dtm import parse_ign_classes from .index import build_index from .quality import backfill_quality n = backfill_quality(args.input, args.output, codes=tuple(parse_ign_classes(args.ign_classes))) logger.info(f"{n} quality sidecar(s) written") build_index(args.output, args.format) return # New run: reset the event log (generation queue). Done here, before # downloading — download events survive the start of processing # (process_all no longer truncates the log). from .progress import reset_events reset_events(args.output) quality = 100 if args.lossless else args.quality # Parse --only and --skip: accept comma-separated values only_viz = None if args.only: only_viz = [v.strip() for item in args.only for v in item.split(',')] skip_viz = None if args.skip: skip_viz = [v.strip() for item in args.skip for v in item.split(',')] resolutions = [float(r.strip()) for r in args.resolution.split(',') if r.strip()] # Download missing IGN tiles before processing. # only_viz/resolutions: a tile is skipped only if it already has all # the requested visualizations — an incomplete tile is # (re)downloaded to generate its missing visualizations. if args.fetch_tiles: from .fetch_ign import fetch_tiles, parse_tile_specs try: specs = parse_tile_specs(args.fetch_tiles) except ValueError as e: logger.error(str(e)) return logger.info(f"Downloading {len(specs)} LiDAR HD tile(s) from IGN...") fetched = fetch_tiles(args.input, specs, args.output, only_viz=only_viz, resolutions=resolutions, force=args.force or args.force_classification) if fetched: logger.info(f"{len(fetched)} tile(s) downloaded — processing...") else: logger.warning("No tile downloaded (already present or not found)") pipeline = LidarArchaeoPipeline( input_dir=args.input, output_dir=args.output, resolution=args.resolution, workers=resolve_workers(args.workers), force=args.force, ground_method=args.ground_classification, ign_classes=args.ign_classes, force_classify=args.force_classification, keep_tif=args.keep_tif, strip_align=not args.no_strip_align, openness_downsample=args.openness_downsample, edge_buffer=args.edge_buffer, quality=quality, only_viz=only_viz, skip_viz=skip_viz, output_format=args.format, gpu_ids=gpu_ids, no_index=args.no_index, incremental_index=args.incremental_index, ) # If --file is specified, process only matching files if args.file: from pathlib import Path input_dir = Path(args.input) # Each pattern is the full filename without extension (e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69.copc) # Also supports bare name without .copc (e.g. LHD_FXX_1000_6882_PTS_LAMB93_IGN69) selected_files = [] for pattern in args.file: # Try exact filename first (e.g. LHD_FXX_...copc.laz) exact_match = input_dir / pattern if exact_match.exists() and exact_match.is_file(): matches = [exact_match] else: # Try with added extensions (e.g. pattern=LHD_FXX_...IGN69) matches = (list(input_dir.glob(f"{pattern}.laz")) + list(input_dir.glob(f"{pattern}.las")) + list(input_dir.glob(f"{pattern}.copc.laz")) + list(input_dir.glob(f"{pattern}.copc.las"))) # Remove duplicates matches = list(dict.fromkeys(matches)) if not matches: logger.warning(f"No file found for: {pattern}") continue selected_files.extend(matches) # Remove duplicates across patterns seen = set() unique_files = [] for f in selected_files: if f not in seen: seen.add(f) unique_files.append(f) if not unique_files: logger.error("No file found for the given patterns") sys.exit(1) logger.info(f"Processing {len(unique_files)} selected file(s)") for laz_file in unique_files: logger.info(f" → {laz_file.name}") # Reuse process_all: parallel workers, summary, index, cleanup pipeline.process_all(files=unique_files) else: pipeline.process_all() except Exception as e: logger.error(f"Fatal error: {e}", exc_info=True) sys.exit(1) def _kill_orphan_pdal(signum=None, frame=None): """Kill orphan PDAL processes of THIS run on interrupt or termination.""" import subprocess # Belt-and-suspenders: os.killpg below is the primary mechanism. # This handles the edge case where a child escapes the process group. # Pattern scoped to our output (see main) so the PDAL processes of other # runs on the machine are not killed. if _pdal_kill_pattern: try: import re subprocess.run(["pkill", "-9", "-f", re.escape(_pdal_kill_pattern)], capture_output=True, timeout=3) except Exception: pass if signum is not None: logger.info("Interrupted — cleaning up processes") # Force-kill all child processes immediately try: import os os.killpg(os.getpgrp(), signal.SIGKILL) except Exception: pass sys.exit(130)