Fix GPU bugs, restore aspect viz, fix anomaly mask, revert flow_acc to vectorized D8
GPU: num_gpus() returns real count via _gpu_candidates (was always 0/1), available_gpu_ids() added. Pipeline: round-robin on real GPU host indices instead of file enumerate index. _process_file_standalone signature simplified. Restore generate_aspect using SharedDEM gradient (dy, dx). Colormap twilight 0-360 fixed range. VIZ_STEPS back to 16. Flow accumulation: revert to vectorized numpy D8 direction + module-level numba accumulator (cached, top-down sort) with Python fallback. Priority-flood NaN-aware. Log1p transform. Anomaly mask: replace RMS+fixed 2sigma threshold (was blank) with weighted sum of |z-score| + adaptive percentile threshold. Absolute z-score captures both positive and negative deviations. 6% signal detected vs 0% before.
This commit is contained in:
@ -24,6 +24,12 @@ _gpu_mem_gb = 0
|
||||
_best_gpu_id: int | None = None
|
||||
_gpu_reason = None
|
||||
|
||||
# GPU restriction from -g flag (host-level indices)
|
||||
_restricted_gpu_ids: list[int] | None = None
|
||||
|
||||
# Discovered GPU candidates (populated by _pick_gpu)
|
||||
_gpu_candidates: list[dict] = []
|
||||
|
||||
|
||||
def _pick_gpu() -> list:
|
||||
"""List all GPUs from the system, sorted by compute capability (highest first)."""
|
||||
@ -75,41 +81,104 @@ _cp_ndimage = None
|
||||
_gpu_initialized = False
|
||||
|
||||
|
||||
def _filter_candidates(gpus: list) -> list:
|
||||
"""Filter GPU candidates by CUDA_VISIBLE_DEVICES and _restricted_gpu_ids.
|
||||
|
||||
nvidia-smi lists ALL GPUs even when CUDA_VISIBLE_DEVICES is set
|
||||
(driver 580.x behavior), so we must filter manually.
|
||||
"""
|
||||
# Filter by CUDA_VISIBLE_DEVICES if set
|
||||
cuda_visible = os.environ.get('CUDA_VISIBLE_DEVICES')
|
||||
if cuda_visible is not None:
|
||||
try:
|
||||
visible = {int(i.strip()) for i in cuda_visible.split(',')}
|
||||
gpus = [g for g in gpus if g[0] in visible]
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Filter by programmatic restriction (-g 0, -g 0,2)
|
||||
if _restricted_gpu_ids is not None:
|
||||
allowed = set(_restricted_gpu_ids)
|
||||
gpus = [g for g in gpus if g[0] in allowed]
|
||||
|
||||
return gpus
|
||||
|
||||
|
||||
def _init_gpu():
|
||||
"""Lazily initialize CuPy on first GPU use.
|
||||
|
||||
Uses a subprocess to test each GPU (highest compute capability first).
|
||||
The subprocess sets CUDA_VISIBLE_DEVICES before importing CuPy and
|
||||
runs a warm-up kernel. First GPU that passes wins.
|
||||
|
||||
This is necessary because CUDA_VISIBLE_DEVICES must be set in the
|
||||
process environment BEFORE CuPy imports, not via os.environ in Python.
|
||||
1. Filters candidates by CUDA_VISIBLE_DEVICES and _restricted_gpu_ids
|
||||
2. If CUDA_VISIBLE_DEVICES is already set (e.g. run.sh -g 0):
|
||||
import CuPy directly (no subprocess test needed)
|
||||
3. Otherwise: test each GPU in a subprocess, pick the first that works
|
||||
"""
|
||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb
|
||||
if _gpu_initialized:
|
||||
return
|
||||
_gpu_initialized = True
|
||||
|
||||
if not _candidate_gpus:
|
||||
candidates = _filter_candidates(_candidate_gpus)
|
||||
if not candidates:
|
||||
logger.info("Pas de GPU utilisable — mode CPU uniquement")
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
HAS_GPU = False
|
||||
return
|
||||
|
||||
# Test each GPU in a subprocess
|
||||
cuda_visible = os.environ.get('CUDA_VISIBLE_DEVICES')
|
||||
|
||||
if cuda_visible is not None:
|
||||
# CUDA_VISIBLE_DEVICES already set (e.g. by run.sh -g 0).
|
||||
# Pick the best visible GPU and import CuPy directly.
|
||||
idx, name, cap_str, mem_mi, score, major = candidates[0]
|
||||
try:
|
||||
import cupy as _real_cupy
|
||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||
|
||||
# Warm-up kernel to verify GPU works
|
||||
x = _real_cupy.array([1.0, 2.0], dtype=_real_cupy.float32)
|
||||
s = _real_cupy.sum(x).get()
|
||||
if s != 3.0:
|
||||
raise RuntimeError("GPU warm-up failed")
|
||||
|
||||
_best_gpu_id = idx
|
||||
_gpu_name = name
|
||||
_gpu_mem_gb = mem_mi // 1024
|
||||
HAS_GPU = True
|
||||
_xp = _real_cupy
|
||||
_cp = _real_cupy
|
||||
_cp_ndimage = _real_cupy_ndimage
|
||||
return
|
||||
except Exception as e:
|
||||
logger.warning(f"GPU indisponible (CUDA_VISIBLE_DEVICES={cuda_visible}): {e}")
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
HAS_GPU = False
|
||||
return
|
||||
|
||||
# No CUDA_VISIBLE_DEVICES set — test each GPU in subprocess
|
||||
import subprocess
|
||||
_working_gpu = None
|
||||
for idx, name, cap_str, mem_mi, score, major in _candidate_gpus:
|
||||
for idx, name, cap_str, mem_mi, score, major in candidates:
|
||||
result = subprocess.run(
|
||||
['python3', '-c',
|
||||
'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); print(cupy.sum(a).get())'],
|
||||
'import cupy; a=cupy.array([1.0,2.0],dtype=cupy.float32); '
|
||||
'dev=cupy.cuda.runtime.getDevice(); '
|
||||
'print(f"OK:{dev}:{cupy.sum(a).get()}")'],
|
||||
capture_output=True, text=True, timeout=120,
|
||||
env={**os.environ, 'CUDA_VISIBLE_DEVICES': str(idx)},
|
||||
)
|
||||
if result.returncode == 0 and '3.0' in result.stdout:
|
||||
_working_gpu = (idx, name, mem_mi)
|
||||
break
|
||||
stdout = result.stdout.strip()
|
||||
if result.returncode == 0 and stdout.startswith('OK:') and '3.0' in stdout:
|
||||
# Verify the device actually used is device 0 (the GPU we targeted)
|
||||
parts = stdout.split(':')
|
||||
if len(parts) >= 2 and parts[1] == '0':
|
||||
_working_gpu = (idx, name, mem_mi)
|
||||
break
|
||||
else:
|
||||
logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) faux positif CPU fallback")
|
||||
logger.warning(f"GPU {idx} ({name}, sm_{cap_str}) non compatible: "
|
||||
f"{result.stderr.strip().splitlines()[-1] if result.stderr else 'inconnue'}")
|
||||
|
||||
@ -141,27 +210,38 @@ def _init_gpu():
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def num_gpus():
|
||||
"""Return 1 if GPU is active, 0 otherwise."""
|
||||
return 1 if HAS_GPU else 0
|
||||
"""Return the number of available GPUs (after restrict_gpus filtering)."""
|
||||
return len(_gpu_candidates)
|
||||
|
||||
|
||||
def available_gpu_ids():
|
||||
"""Return list of host-level GPU indices available for processing.
|
||||
|
||||
Respects any prior restrict_gpus() call.
|
||||
"""
|
||||
return [c['id'] for c in _gpu_candidates]
|
||||
|
||||
|
||||
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
||||
"""No-op — GPU is auto-selected at import time."""
|
||||
pass
|
||||
"""Restrict GPU selection to specific host-level indices.
|
||||
|
||||
Stores the restriction to be applied during _init_gpu().
|
||||
"""
|
||||
global _restricted_gpu_ids
|
||||
_restricted_gpu_ids = gpu_ids
|
||||
|
||||
|
||||
def set_active_gpu(gpu_id):
|
||||
"""No-op — GPU is auto-selected at import time."""
|
||||
pass
|
||||
"""Restrict to a single GPU by host-level index."""
|
||||
global _restricted_gpu_ids
|
||||
_restricted_gpu_ids = [gpu_id]
|
||||
|
||||
|
||||
def _gpu_available():
|
||||
"""Check if GPU is usable right now."""
|
||||
if not HAS_GPU:
|
||||
return False
|
||||
try:
|
||||
_init_gpu()
|
||||
return _cp is not None
|
||||
return HAS_GPU and _cp is not None
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@ -212,7 +292,7 @@ def to_gpu(arr):
|
||||
try:
|
||||
return _cp.asarray(arr.astype(np.float32))
|
||||
except Exception:
|
||||
pass
|
||||
disable_gpu()
|
||||
return arr.astype(np.float32)
|
||||
|
||||
|
||||
@ -302,7 +382,7 @@ def safe_gpu_call(func, *args, **kwargs):
|
||||
return func(*args, **kwargs)
|
||||
except Exception as e:
|
||||
err_msg = str(e)
|
||||
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
|
||||
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg):
|
||||
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
|
||||
disable_gpu()
|
||||
return func(*args, **kwargs)
|
||||
|
||||
Reference in New Issue
Block a user