Rasterisation GPU du DTM (bin_mean_2d) + GPU fallback renforcé et workers mono-thread

Rastérisation des points sol via gpu.bin_mean_2d (bincount 2D CuPy, sémantique
identique au repli scipy binned_statistic_2d) : ~x7 plus rapide sur cette
étape, logs de durée par phase dans create_dtm_fast. safe_gpu_call retente en
CPU sur toute erreur GPU (pas seulement les messages CUDA) : un transfert
échoué en cours de run mélangeait types numpy/cupy et faisait échouer la
visualisation entière. Workers en mono-thread BLAS/OpenMP sur la machine de
traitement (plus de saturation des cœurs pendant les runs).
This commit is contained in:
Antoine Jacquin
2026-09-21 23:42:55 +02:00
parent 2d5a9b2a46
commit 7c10ae3e18
4 changed files with 219 additions and 15 deletions

View File

@ -419,6 +419,64 @@ def xp_maximum_filter(arr, footprint=None, size=None):
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
# ---------------------------------------------------------------------------
# Rasterisation MNT — moyenne z par cellule (bincount 2D)
# ---------------------------------------------------------------------------
def _bin_mean_core(lib, xs, ys, zs, width, height, x_range, y_range):
"""Moyenne z par cellule d'une grille régulière, backend-agnostique.
`lib` = numpy ou cupy (mêmes primitives). Sémantique calquée sur
scipy.stats.binned_statistic_2d(statistic='mean', bins=[width, height],
range=[[xmin, xmax], [ymin, ymax]]) : points hors emprise ignorés, valeur
exactement sur le bord droit/haut rangée dans la dernière cellule.
Retourne (height, width) float64, NaN sur les cellules vides.
float64 obligatoire pour x/y : à des coordonnées Lambert 93 (~1e6 m) la
résolution float32 est ~6 cm — grossière devant un pixel de 0,2 m.
"""
(xmin, xmax), (ymin, ymax) = x_range, y_range
x = lib.asarray(xs, dtype=lib.float64)
y = lib.asarray(ys, dtype=lib.float64)
z = lib.asarray(zs, dtype=lib.float64)
inside = (x >= xmin) & (x <= xmax) & (y >= ymin) & (y <= ymax)
x, y, z = x[inside], y[inside], z[inside]
# floor((v - min) / pas) ; les valeurs intérieures sont >= 0 donc la
# troncature astype == floor. Le clip range le bord droit (et l'arrondi
# float adjacent) dans la dernière cellule, comme scipy.
ix = ((x - xmin) * (float(width) / (xmax - xmin))).astype(lib.int64)
iy = ((y - ymin) * (float(height) / (ymax - ymin))).astype(lib.int64)
ix = lib.clip(ix, 0, width - 1)
iy = lib.clip(iy, 0, height - 1)
n = int(width) * int(height)
idx = iy * int(width) + ix
sums = lib.bincount(idx, weights=z, minlength=n)
counts = lib.bincount(idx, minlength=n)
mean = sums / lib.where(counts == 0, 1, counts)
return lib.where(counts == 0, float("nan"), mean).reshape(int(height), int(width))
def bin_mean_2d(xs, ys, zs, width, height, x_range, y_range):
"""Rasterisation « moyenne par cellule » sur GPU (appelant : dtm.create_dtm_fast).
Retourne un tableau numpy (height, width) float64 (NaN = cellule vide),
ou None si le GPU est indisponible ou échoue (OOM le plus souvent) —
l'appelant retombe alors sur scipy. Un échec ici n'appelle PAS
disable_gpu() : la rastérisation est ponctuelle, les visualisations qui
suivent doivent garder leur accélérateur.
"""
if not _gpu_available():
return None
try:
result = _bin_mean_core(_cp, xs, ys, zs, width, height,
x_range, y_range)
return to_cpu(result)
except Exception as e:
logger.warning(f"Rasterisation GPU échouée ({e}) — repli scipy")
gpu_cleanup()
return None
# ---------------------------------------------------------------------------
# Misc
# ---------------------------------------------------------------------------
@ -458,9 +516,14 @@ def safe_gpu_call(func, *args, **kwargs):
try:
return func(*args, **kwargs)
except Exception as e:
err_msg = str(e)
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg):
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
# GPU actif : on TOUTE erreur (OOM, types numpy/cupy mêlés après un
# échec de transfert en cours de run, ...) le GPU est désactivé et le
# calcul est retranché en CPU — une panne partielle ne doit pas
# faire échouer la visualisation entière. En mode CPU, on relance
# l'erreur d'origine (déjà en CPU, rien à retrancher).
if _cp is not None:
logger.warning(f"Erreur GPU ({e.__class__.__name__}: {e}), "
f"retry en CPU...")
disable_gpu()
cpu_args = tuple(to_cpu(a) for a in args)
cpu_kwargs = {k: to_cpu(v) for k, v in kwargs.items()}