Auto-detect best GPU (RTX 5060 preferred) + build CuPy from source for sm_120
This commit is contained in:
20
Dockerfile
20
Dockerfile
@ -1,4 +1,4 @@
|
|||||||
FROM nvidia/cuda:11.8.0-devel-ubuntu22.04
|
FROM nvidia/cuda:12.4.0-devel-ubuntu22.04
|
||||||
|
|
||||||
ENV DEBIAN_FRONTEND=noninteractive
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
ENV TZ=Europe/Paris
|
ENV TZ=Europe/Paris
|
||||||
@ -45,15 +45,15 @@ RUN pip3 install --no-cache-dir \
|
|||||||
pillow-avif-plugin \
|
pillow-avif-plugin \
|
||||||
cmcrameri
|
cmcrameri
|
||||||
|
|
||||||
# Install CuPy for GPU acceleration (optional - will fallback to numpy if not available)
|
# Build CuPy from source with nvcc, targeting sm_120 (RTX 5060) and nearby archs.
|
||||||
# We use CuPy 13.4 (CUDA 11.x wheel) because CuPy 14.x dropped JIT compilation
|
# Pre-built wheels (cupy-cuda12x 14.x) don't include sm_120, so we compile.
|
||||||
# support. With CUPY_CUDA_COMPILE_WITH_CACHE=1, CuPy 13.4 compiles kernels at
|
# This step takes ~30 min the first time; the image is cached after that.
|
||||||
# runtime for GPU architectures not in the pre-built wheel (e.g. sm_89 / RTX 4060 Ti).
|
RUN apt-get update && apt-get install -y --no-install-recommends git && \
|
||||||
# The devel image includes nvcc, required for JIT compilation.
|
git clone --depth 1 --branch v14.0.0 https://github.com/cupy/cupy.git /tmp/cupy-src && \
|
||||||
# nvcc stays in PATH for the 'lidar' user.
|
cd /tmp/cupy-src && \
|
||||||
ENV CUPY_CUDA_COMPILE_WITH_CACHE=1
|
CUPY_NVCC_GENERATE_CODE='sm_89;sm_90;sm_120' \
|
||||||
ENV PATH=/usr/local/cuda/bin:${PATH}
|
pip3 install --no-cache-dir -e . && \
|
||||||
RUN pip3 install --no-cache-dir cupy-cuda11x==13.4.0 || echo "CuPy not available - GPU acceleration disabled"
|
rm -rf /tmp/cupy-src
|
||||||
|
|
||||||
# Copy and install the pipeline package
|
# Copy and install the pipeline package
|
||||||
COPY setup.py .
|
COPY setup.py .
|
||||||
|
|||||||
@ -1,15 +1,8 @@
|
|||||||
"""GPU acceleration helpers for LiDAR pipeline.
|
"""GPU acceleration helpers for LiDAR pipeline.
|
||||||
|
|
||||||
Provides CuPy/numpy abstraction layer. If CuPy is available and a CUDA GPU
|
Auto-selects the best NVIDIA GPU (RTX 50xx preferred) and restricts
|
||||||
is detected, array operations are accelerated on the GPU. Otherwise, all
|
CUDA_VISIBLE_DEVICES so all workers share a single GPU. Falls back
|
||||||
operations fall back to numpy/scipy on CPU.
|
to CPU if no GPU is available or usable.
|
||||||
|
|
||||||
GPU errors (e.g. in forked subprocesses) are caught gracefully and
|
|
||||||
cause an automatic fallback to CPU for the current operation.
|
|
||||||
|
|
||||||
Multi-GPU support: each worker process sets CUDA_VISIBLE_DEVICES before
|
|
||||||
CuPy is imported, so CuPy only sees its assigned GPU. This avoids kernel
|
|
||||||
cache incompatibilities that occur with Device.use() switching.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
@ -19,125 +12,113 @@ from scipy import ndimage
|
|||||||
|
|
||||||
logger = logging.getLogger("lidar")
|
logger = logging.getLogger("lidar")
|
||||||
|
|
||||||
# Detect total GPU count via nvidia-smi (no CUDA context created).
|
# ---------------------------------------------------------------------------
|
||||||
# This must happen before any CUDA_VISIBLE_DEVICES manipulation.
|
# GPU auto-detection via nvidia-smi (no CUDA context created)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
_NUM_GPUS = 0
|
_NUM_GPUS = 0
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
_gpu_name = None
|
_gpu_name = None
|
||||||
_gpu_mem_gb = 0
|
_gpu_mem_gb = 0
|
||||||
# System-level GPU IDs that are currently visible (after restrict_gpus).
|
_best_gpu_id: int | None = None
|
||||||
# Populated by restrict_gpus() or auto-detected at import time.
|
|
||||||
_available_gpu_ids: list[int] = []
|
|
||||||
|
|
||||||
try:
|
|
||||||
import subprocess
|
def _pick_gpu() -> int | None:
|
||||||
_result = subprocess.run(
|
"""Pick the best GPU from the system.
|
||||||
['nvidia-smi', '--query-gpu=count,name,memory.total', '--format=csv,noheader,nounits'],
|
|
||||||
capture_output=True, text=True, timeout=5
|
Preference order:
|
||||||
)
|
1. RTX 50xx (Blackwell, compute >= 12.0)
|
||||||
if _result.returncode == 0:
|
2. RTX 40xx (Ada Lovelace, compute >= 8.9)
|
||||||
_lines = _result.stdout.strip().split('\n')
|
3. Any NVIDIA GPU with highest compute capability
|
||||||
_NUM_GPUS = len(_lines)
|
"""
|
||||||
# Parse first GPU info for logging
|
|
||||||
_parts = _lines[0].split(',')
|
|
||||||
if len(_parts) >= 3:
|
|
||||||
_gpu_name = _parts[1].strip()
|
|
||||||
try:
|
try:
|
||||||
_gpu_mem_gb = int(float(_parts[2].strip())) // 1024
|
import subprocess
|
||||||
except (ValueError, IndexError):
|
result = subprocess.run(
|
||||||
pass
|
['nvidia-smi', '--query-gpu=index,name,compute_cap,memory.total',
|
||||||
HAS_GPU = True
|
'--format=csv,noheader,nounits'],
|
||||||
# All GPUs are visible by default
|
capture_output=True, text=True, timeout=5,
|
||||||
_available_gpu_ids = list(range(_NUM_GPUS))
|
)
|
||||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
if result.returncode != 0:
|
||||||
pass
|
return None
|
||||||
|
|
||||||
# Lazy CuPy initialization — imported only when first needed.
|
global _NUM_GPUS, _gpu_name, _gpu_mem_gb, HAS_GPU, _best_gpu_id
|
||||||
# This allows CUDA_VISIBLE_DEVICES to be set before CuPy creates
|
|
||||||
# a CUDA context, enabling per-process GPU assignment.
|
gpus = []
|
||||||
_xp = np # Default: CPU
|
for line in result.stdout.strip().split('\n'):
|
||||||
_cp = None # cupy module (or None)
|
parts = [p.strip() for p in line.split(',')]
|
||||||
_cp_ndimage = None # cupyx.scipy.ndimage (or None)
|
if len(parts) < 4:
|
||||||
|
continue
|
||||||
|
idx = int(parts[0])
|
||||||
|
name = parts[1]
|
||||||
|
cap_str = parts[2]
|
||||||
|
mem_mi = int(parts[3])
|
||||||
|
major, minor = (int(x) for x in cap_str.split('.'))
|
||||||
|
# Score: higher compute capability first, then more VRAM
|
||||||
|
score = major * 1000 + minor * 100 + mem_mi
|
||||||
|
gpus.append((idx, name, cap_str, mem_mi, score))
|
||||||
|
|
||||||
|
if not gpus:
|
||||||
|
return None
|
||||||
|
|
||||||
|
_NUM_GPUS = len(gpus)
|
||||||
|
gpus.sort(key=lambda g: g[4], reverse=True)
|
||||||
|
best = gpus[0]
|
||||||
|
_best_gpu_id = best[0]
|
||||||
|
_gpu_name = best[1]
|
||||||
|
_gpu_mem_gb = best[3] // 1024
|
||||||
|
HAS_GPU = True
|
||||||
|
return _best_gpu_id
|
||||||
|
|
||||||
|
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
_best_gpu_id = _pick_gpu()
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Lazy CuPy initialization
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
_xp = np
|
||||||
|
_cp = None
|
||||||
|
_cp_ndimage = None
|
||||||
_gpu_initialized = False
|
_gpu_initialized = False
|
||||||
|
|
||||||
|
|
||||||
def _init_gpu():
|
def _init_gpu():
|
||||||
"""Lazily initialize CuPy on first GPU use.
|
"""Lazily initialize CuPy on first GPU use."""
|
||||||
|
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
||||||
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
|
|
||||||
set before the CUDA context is created.
|
|
||||||
|
|
||||||
With CUPY_CUDA_COMPILE_WITH_CACHE=1, JIT-compiled kernels are cached
|
|
||||||
on disk and shared across processes. We warm up on ALL visible GPUs
|
|
||||||
so workers that later restrict to a single GPU find pre-compiled kernels.
|
|
||||||
|
|
||||||
Per-GPU warm-up errors are caught individually: if GPU 1 fails to
|
|
||||||
compile a kernel (e.g. CUDA_ERROR_NO_BINARY_FOR_GPU on sm_89), we
|
|
||||||
record it and continue warming up GPU 0. Workers assigned to a
|
|
||||||
failed GPU fall back to CPU automatically.
|
|
||||||
"""
|
|
||||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _gpu_failed_ids
|
|
||||||
if _gpu_initialized:
|
if _gpu_initialized:
|
||||||
return
|
return
|
||||||
_gpu_initialized = True
|
_gpu_initialized = True
|
||||||
_gpu_failed_ids = set()
|
|
||||||
|
|
||||||
try:
|
if not HAS_GPU or _best_gpu_id is None:
|
||||||
import cupy as _real_cupy
|
|
||||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
|
||||||
|
|
||||||
n_devs = _real_cupy.cuda.runtime.getDeviceCount()
|
|
||||||
all_failed = False
|
|
||||||
|
|
||||||
# Warm up each GPU independently — one failure doesn't kill the others.
|
|
||||||
for dev_id in range(n_devs):
|
|
||||||
try:
|
|
||||||
with _real_cupy.cuda.Device(dev_id):
|
|
||||||
props = _real_cupy.cuda.runtime.getDeviceProperties(dev_id)
|
|
||||||
name = props['name'].decode() if isinstance(props['name'], bytes) else props['name']
|
|
||||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
|
||||||
_result = _real_cupy.sum(_test * _test)
|
|
||||||
_ = _result.get()
|
|
||||||
del _test, _result
|
|
||||||
logger.info(f" GPU {dev_id}: {name} — kernels pré-compilés")
|
|
||||||
except Exception as dev_err:
|
|
||||||
_gpu_failed_ids.add(dev_id)
|
|
||||||
logger.warning(
|
|
||||||
f" GPU {dev_id}: échec warm-up ({dev_err.__class__.__name__}: "
|
|
||||||
f"{dev_err}) — workers sur ce GPU passeront en CPU"
|
|
||||||
)
|
|
||||||
|
|
||||||
# If ALL visible GPUs failed, disable GPU entirely.
|
|
||||||
# This is critical for worker subprocesses that restricted
|
|
||||||
# CUDA_VISIBLE_DEVICES to a single GPU before importing CuPy.
|
|
||||||
if len(_gpu_failed_ids) >= n_devs:
|
|
||||||
all_failed = True
|
|
||||||
logger.warning("Tous les GPUs échoués au warm-up — passage en mode CPU")
|
|
||||||
|
|
||||||
if not all_failed:
|
|
||||||
_xp = _real_cupy
|
|
||||||
_cp = _real_cupy
|
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
|
||||||
|
|
||||||
# Limit GPU memory pool per worker to avoid OOM when multiple
|
|
||||||
# workers share one GPU. Each worker gets at most 3.5 GB (or
|
|
||||||
# 50 % of total VRAM on smaller cards).
|
|
||||||
try:
|
|
||||||
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
|
|
||||||
total_mem = props['totalGlobalMem']
|
|
||||||
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
|
|
||||||
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
|
|
||||||
logger.info(
|
|
||||||
f" Pool mémoire GPU limité à "
|
|
||||||
f"{max_worker_mem // (1024**3) * 1000 // 1024} MB"
|
|
||||||
)
|
|
||||||
except Exception:
|
|
||||||
pass # pool config is best-effort
|
|
||||||
else:
|
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
_cp_ndimage = None
|
_cp_ndimage = None
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Restrict to the selected GPU before CuPy imports
|
||||||
|
os.environ['CUDA_VISIBLE_DEVICES'] = str(_best_gpu_id)
|
||||||
|
|
||||||
|
import cupy as _real_cupy
|
||||||
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||||
|
|
||||||
|
# Warm-up: verify kernel execution works on this GPU
|
||||||
|
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||||
|
_result = _real_cupy.sum(_test * _test)
|
||||||
|
_ = _result.get()
|
||||||
|
del _test, _result
|
||||||
|
|
||||||
|
_xp = _real_cupy
|
||||||
|
_cp = _real_cupy
|
||||||
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
|
||||||
|
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
|
||||||
|
total_mem = props['totalGlobalMem']
|
||||||
|
# Memory pool: up to 90% of VRAM (single GPU shared by workers)
|
||||||
|
pool_size = int(total_mem * 0.9)
|
||||||
|
_real_cupy.cuda.set_memory_pool(0, pool_size)
|
||||||
|
|
||||||
except (ImportError, Exception) as e:
|
except (ImportError, Exception) as e:
|
||||||
logger.warning(f"GPU non disponible — mode CPU: {e}")
|
logger.warning(f"GPU non disponible — mode CPU: {e}")
|
||||||
@ -147,94 +128,32 @@ def _init_gpu():
|
|||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
|
|
||||||
|
|
||||||
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
# ---------------------------------------------------------------------------
|
||||||
"""Restrict which GPUs are visible to the process.
|
# Public API
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
By default (set_env_var=False), only records the GPU IDs for
|
|
||||||
num_gpus() and logging. Does NOT touch CUDA_VISIBLE_DEVICES
|
|
||||||
in the main process because CuPy 13.x JIT compilation (sm_89)
|
|
||||||
needs ALL GPUs visible during warm-up to pre-compile kernels.
|
|
||||||
|
|
||||||
Workers call this with set_env_var=True (via assign_gpu_to_worker)
|
|
||||||
after forking, when CuPy hasn't been imported yet in the child.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
gpu_ids: List of system-level GPU indices to make visible.
|
|
||||||
set_env_var: If True, actually set CUDA_VISIBLE_DEVICES.
|
|
||||||
Default False (safe for main process).
|
|
||||||
"""
|
|
||||||
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
|
|
||||||
if not gpu_ids or not HAS_GPU:
|
|
||||||
return
|
|
||||||
|
|
||||||
# Validate IDs against total GPU count from nvidia-smi
|
|
||||||
total_count = _NUM_GPUS or 1
|
|
||||||
valid_ids = [gid % total_count for gid in gpu_ids]
|
|
||||||
_available_gpu_ids = valid_ids
|
|
||||||
_NUM_GPUS = len(valid_ids)
|
|
||||||
|
|
||||||
if set_env_var:
|
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
|
|
||||||
logger.info(f"GPU visibles: {_available_gpu_ids}")
|
|
||||||
|
|
||||||
|
|
||||||
def num_gpus():
|
def num_gpus():
|
||||||
"""Return the number of visible (available) GPUs."""
|
"""Return 1 if GPU is active, 0 otherwise."""
|
||||||
return _NUM_GPUS
|
return 1 if HAS_GPU else 0
|
||||||
|
|
||||||
|
|
||||||
|
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
||||||
|
"""No-op — GPU is auto-selected at import time."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
def set_active_gpu(gpu_id):
|
def set_active_gpu(gpu_id):
|
||||||
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES.
|
"""No-op — GPU is auto-selected at import time."""
|
||||||
|
pass
|
||||||
gpu_id is an index into the currently visible GPU list
|
|
||||||
(_available_gpu_ids), not a system-level ID.
|
|
||||||
|
|
||||||
MUST be called before any GPU operation (to_gpu, etc.) to ensure
|
|
||||||
CuPy creates its CUDA context on the correct device. With lazy
|
|
||||||
initialization, CuPy is imported AFTER this call, so it only
|
|
||||||
sees the assigned GPU.
|
|
||||||
|
|
||||||
If the assigned GPU failed warm-up (e.g. NO_BINARY_FOR_GPU), this
|
|
||||||
function disables GPU for this worker so all operations fall back
|
|
||||||
to CPU without crashing.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
gpu_id: 0-based index into the visible GPU list.
|
|
||||||
"""
|
|
||||||
if not HAS_GPU or _NUM_GPUS <= 1:
|
|
||||||
return # Nothing to do for single GPU or no GPU
|
|
||||||
|
|
||||||
gpu_id = gpu_id % _NUM_GPUS
|
|
||||||
|
|
||||||
# Map visible-GPU index back to the real system GPU ID
|
|
||||||
system_gpu_id = _available_gpu_ids[gpu_id]
|
|
||||||
|
|
||||||
# If this GPU failed warm-up, disable GPU for this worker
|
|
||||||
if system_gpu_id in _gpu_failed_ids:
|
|
||||||
logger.warning(
|
|
||||||
f" GPU {system_gpu_id} échouée au warm-up — "
|
|
||||||
f"worker passe en mode CPU"
|
|
||||||
)
|
|
||||||
disable_gpu()
|
|
||||||
return
|
|
||||||
|
|
||||||
# Set CUDA_VISIBLE_DEVICES to isolate this worker to one GPU.
|
|
||||||
# This MUST happen before CuPy is imported (lazy init).
|
|
||||||
# The JIT kernels were already pre-compiled by the main process
|
|
||||||
# on all GPUs, so the worker finds them in the cache.
|
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
|
|
||||||
|
|
||||||
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
|
|
||||||
|
|
||||||
|
|
||||||
def _gpu_available():
|
def _gpu_available():
|
||||||
"""Check if GPU is usable right now (may fail in forked subprocesses)."""
|
"""Check if GPU is usable right now."""
|
||||||
if not HAS_GPU:
|
if not HAS_GPU:
|
||||||
return False
|
return False
|
||||||
try:
|
try:
|
||||||
_init_gpu()
|
_init_gpu()
|
||||||
_cp.cuda.runtime.getDevice()
|
return _cp is not None
|
||||||
return True
|
|
||||||
except Exception:
|
except Exception:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
@ -242,34 +161,30 @@ def _gpu_available():
|
|||||||
def log_gpu_status():
|
def log_gpu_status():
|
||||||
"""Log GPU detection result. Called after logging is configured."""
|
"""Log GPU detection result. Called after logging is configured."""
|
||||||
if _gpu_available():
|
if _gpu_available():
|
||||||
# Get actual device name from CuPy (after init)
|
|
||||||
try:
|
try:
|
||||||
dev = _cp.cuda.Device()
|
|
||||||
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
|
name = _cp.cuda.runtime.getDeviceProperties(0)['name']
|
||||||
if isinstance(name, bytes):
|
if isinstance(name, bytes):
|
||||||
name = name.decode()
|
name = name.decode()
|
||||||
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024 ** 3)
|
mem_gb = _cp.cuda.runtime.getDeviceProperties(0)['totalGlobalMem'] // (1024**3)
|
||||||
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM)"
|
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
||||||
except Exception:
|
except Exception:
|
||||||
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||||
if _NUM_GPUS > 1:
|
|
||||||
gpu_info += f" × {_NUM_GPUS}"
|
|
||||||
logger.info(gpu_info)
|
logger.info(gpu_info)
|
||||||
else:
|
else:
|
||||||
logger.info("Pas de GPU — mode CPU uniquement")
|
logger.info("Pas de GPU — mode CPU uniquement")
|
||||||
|
|
||||||
|
|
||||||
def to_gpu(arr):
|
# ---------------------------------------------------------------------------
|
||||||
"""Send array to GPU if available, otherwise return as float32 numpy.
|
# Array transfer
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
Uses float32 to reduce GPU memory usage. Falls back to CPU if GPU
|
def to_gpu(arr):
|
||||||
is unavailable (e.g. in forked subprocess).
|
"""Send array to GPU if available, otherwise return as float32 numpy."""
|
||||||
"""
|
|
||||||
if _gpu_available():
|
if _gpu_available():
|
||||||
try:
|
try:
|
||||||
return _cp.asarray(arr.astype(np.float32))
|
return _cp.asarray(arr.astype(np.float32))
|
||||||
except Exception:
|
except Exception:
|
||||||
pass # Fall back to CPU
|
pass
|
||||||
return arr.astype(np.float32)
|
return arr.astype(np.float32)
|
||||||
|
|
||||||
|
|
||||||
@ -279,12 +194,15 @@ def to_cpu(arr):
|
|||||||
try:
|
try:
|
||||||
return _cp.asnumpy(arr)
|
return _cp.asnumpy(arr)
|
||||||
except Exception:
|
except Exception:
|
||||||
pass # Already on CPU or GPU error
|
pass
|
||||||
return arr
|
return arr
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Filters — GPU if array is on GPU, CPU otherwise
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
def xp_gaussian_filter(arr, sigma):
|
def xp_gaussian_filter(arr, sigma):
|
||||||
"""Gaussian filter — uses GPU if array is on GPU, CPU otherwise."""
|
|
||||||
if _cp is not None and isinstance(arr, _cp.ndarray):
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
||||||
try:
|
try:
|
||||||
return _cp_ndimage.gaussian_filter(arr, sigma)
|
return _cp_ndimage.gaussian_filter(arr, sigma)
|
||||||
@ -294,7 +212,6 @@ def xp_gaussian_filter(arr, sigma):
|
|||||||
|
|
||||||
|
|
||||||
def xp_uniform_filter(arr, size):
|
def xp_uniform_filter(arr, size):
|
||||||
"""Uniform filter — uses GPU if array is on GPU, CPU otherwise."""
|
|
||||||
if _cp is not None and isinstance(arr, _cp.ndarray):
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
||||||
try:
|
try:
|
||||||
return _cp_ndimage.uniform_filter(arr, size)
|
return _cp_ndimage.uniform_filter(arr, size)
|
||||||
@ -304,7 +221,6 @@ def xp_uniform_filter(arr, size):
|
|||||||
|
|
||||||
|
|
||||||
def xp_minimum_filter(arr, footprint=None, size=None):
|
def xp_minimum_filter(arr, footprint=None, size=None):
|
||||||
"""Minimum filter — uses GPU if array is on GPU, CPU otherwise."""
|
|
||||||
if _cp is not None and isinstance(arr, _cp.ndarray):
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
||||||
try:
|
try:
|
||||||
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
|
return _cp_ndimage.minimum_filter(arr, footprint=footprint, size=size)
|
||||||
@ -314,7 +230,6 @@ def xp_minimum_filter(arr, footprint=None, size=None):
|
|||||||
|
|
||||||
|
|
||||||
def xp_maximum_filter(arr, footprint=None, size=None):
|
def xp_maximum_filter(arr, footprint=None, size=None):
|
||||||
"""Maximum filter — uses GPU if array is on GPU, CPU otherwise."""
|
|
||||||
if _cp is not None and isinstance(arr, _cp.ndarray):
|
if _cp is not None and isinstance(arr, _cp.ndarray):
|
||||||
try:
|
try:
|
||||||
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
return _cp_ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
||||||
@ -323,6 +238,10 @@ def xp_maximum_filter(arr, footprint=None, size=None):
|
|||||||
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
return ndimage.maximum_filter(arr, footprint=footprint, size=size)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Misc
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
def gpu_cleanup():
|
def gpu_cleanup():
|
||||||
"""Free GPU memory. Call between visualizations to prevent OOM."""
|
"""Free GPU memory. Call between visualizations to prevent OOM."""
|
||||||
if _cp is not None:
|
if _cp is not None:
|
||||||
@ -333,15 +252,10 @@ def gpu_cleanup():
|
|||||||
|
|
||||||
|
|
||||||
def disable_gpu():
|
def disable_gpu():
|
||||||
"""Disable GPU acceleration for the rest of this process.
|
"""Disable GPU acceleration for the rest of this process."""
|
||||||
|
|
||||||
Called when a CUDA error indicates the GPU is unusable (e.g.
|
|
||||||
CUDA_ERROR_NO_BINARY_FOR_GPU). Falls back to numpy for all
|
|
||||||
subsequent operations.
|
|
||||||
"""
|
|
||||||
global HAS_GPU, _xp, _cp, _cp_ndimage
|
global HAS_GPU, _xp, _cp, _cp_ndimage
|
||||||
if not HAS_GPU:
|
if not HAS_GPU:
|
||||||
return # Already disabled
|
return
|
||||||
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
_xp = np
|
_xp = np
|
||||||
@ -350,21 +264,12 @@ def disable_gpu():
|
|||||||
|
|
||||||
|
|
||||||
def is_gpu_active():
|
def is_gpu_active():
|
||||||
"""Check if GPU acceleration is currently active.
|
"""Check if GPU acceleration is currently active."""
|
||||||
|
|
||||||
Unlike the HAS_GPU module-level variable (which can go stale if
|
|
||||||
imported directly), this always reflects the current runtime state.
|
|
||||||
Use this in logging tags and conditional GPU paths.
|
|
||||||
"""
|
|
||||||
return HAS_GPU
|
return HAS_GPU
|
||||||
|
|
||||||
|
|
||||||
def safe_gpu_call(func, *args, **kwargs):
|
def safe_gpu_call(func, *args, **kwargs):
|
||||||
"""Call a function with GPU arrays, retrying on CPU if GPU fails.
|
"""Call a function with GPU arrays, retrying on CPU if GPU fails."""
|
||||||
|
|
||||||
Usage:
|
|
||||||
result = safe_gpu_call(generate_svf, dem_file, basename, vis_dir, resolution, shared=shared)
|
|
||||||
"""
|
|
||||||
try:
|
try:
|
||||||
return func(*args, **kwargs)
|
return func(*args, **kwargs)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
Reference in New Issue
Block a user