Files
lidar_rendu/docker-compose.worker.yml
Jacquin Antoine d0dc8d90e9 Make the worker batch timeout configurable via LIDAR_BATCH_TIMEOUT
The pool of tile workers used to cancel every remaining tile after a
hardcoded 2-hour wall clock, silently truncating large batches (a
670-tile completion run lost its last 348 tiles that way). The timeout
now defaults to unlimited and can be capped per deployment with the
LIDAR_BATCH_TIMEOUT environment variable (seconds); the local worker
compose sets it to 6 hours.

💘 Generated with Crush

Assisted-by: Crush:glm-5.2
2026-09-28 00:03:52 +02:00

76 lines
3.1 KiB
YAML

# Processing machine - tile generator (see docs/DEPLOY_WEBAPP.md).
#
# Runs on the powerful machine (GPU + PDAL) and exposes the API that remote
# maps (Raspberry Pi, docker-compose.maps.yml + LIDAR_GENERATION_URL) call:
# drawing an area -> IGN download + GPU processing here; source tiles and
# XYZ tiles served to the lightweight machines (/api/tiles + static files,
# /tiles/XYZ).
#
# docker compose -f docker-compose.worker.yml up -d --build
# docker compose -f docker-compose.worker.yml logs -f worker
# docker compose -f docker-compose.worker.yml down
#
# One-off batch processing of the tiles present in input/:
# docker compose -f docker-compose.worker.yml run --rm --build process
# (any argument replaces the default command: repeat
# python3 -m lidar_pipeline /data/input -o /data/output -r 0.2 -g all ... in full)
services:
# Tile generator: XYZ map + generation API (full image, GPU)
worker:
build: .
image: lidar-lidar
container_name: lidar-worker
init: true
user: "1000:1000"
gpus: all # remove this line on a machine without a GPU
ports:
- "8973:8973" # reachable by the remote maps (LIDAR_GENERATION_URL
# and LIDAR_SOURCE_URL point here)
volumes:
# input/ writable: the API downloads missing IGN tiles into it
- ./input:/data/input
- ./output:/data/output
environment:
- TZ=Europe/Paris
- LIDAR_INPUT_DIR=/data/input
- LIDAR_OUTPUT_DIR=/data/output
- LIDAR_PORT=8973
# Generations started from a remote map use the GPU
- LIDAR_GPU=1
- LIDAR_WORKERS=auto
# Wall-clock cap for one batch of tiles, in seconds (unset or 0 = unlimited)
- LIDAR_BATCH_TIMEOUT=21600
# Workers per GPU capped by the free VRAM when the run starts
# ((free - reserve) / per-worker peak); the excess runs on the CPU.
# Peak estimated at 2048 MiB: adjust after measuring (nvidia-smi during a run).
# - LIDAR_GPU_WORKER_MIB=2048
# - LIDAR_GPU_RESERVE_MIB=512
# One thread per worker: single-threaded BLAS/OpenMP, otherwise 12 workers x N
# threads swamp the 14 cores (load 68+ observed during runs)
- OMP_NUM_THREADS=1
- OPENBLAS_NUM_THREADS=1
- MKL_NUM_THREADS=1
- NUMEXPR_NUM_THREADS=1
# Protect the API if the network is not trusted: same value as
# LIDAR_REMOTE_TOKEN on every remote map (otherwise leave commented out)
# - LIDAR_API_TOKEN=change-me
command: python3 -m uvicorn lidar_pipeline.mapserve:app --host 0.0.0.0 --port 8973
restart: unless-stopped
# One-off processing of the input/ tiles (one pass, then exit)
process:
build: .
image: lidar-lidar
container_name: lidar-process
init: true
user: "1000:1000"
gpus: all # remove this line on a machine without a GPU
volumes:
- ./input:/data/input
- ./output:/data/output
environment:
- TZ=Europe/Paris
command: ["python3", "-m", "lidar_pipeline", "/data/input", "-o", "/data/output", "-r", "0.2", "-g", "all"]
profiles:
- process