Add --gpu flag to select specific GPU(s) for processing
This commit is contained in:
@ -25,6 +25,9 @@ _NUM_GPUS = 0
|
||||
HAS_GPU = False
|
||||
_gpu_name = None
|
||||
_gpu_mem_gb = 0
|
||||
# System-level GPU IDs that are currently visible (after restrict_gpus).
|
||||
# Populated by restrict_gpus() or auto-detected at import time.
|
||||
_available_gpu_ids: list[int] = []
|
||||
|
||||
try:
|
||||
import subprocess
|
||||
@ -44,6 +47,8 @@ try:
|
||||
except (ValueError, IndexError):
|
||||
pass
|
||||
HAS_GPU = True
|
||||
# All GPUs are visible by default
|
||||
_available_gpu_ids = list(range(_NUM_GPUS))
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, Exception):
|
||||
pass
|
||||
|
||||
@ -61,8 +66,12 @@ def _init_gpu():
|
||||
|
||||
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
|
||||
set before the CUDA context is created.
|
||||
|
||||
Validates that the GPU can actually execute kernels by running
|
||||
a small computation. This catches CUDA_ERROR_NO_BINARY_FOR_GPU
|
||||
and other compute capability mismatches before they crash visualizations.
|
||||
"""
|
||||
global _xp, _cp, _cp_ndimage, _gpu_initialized
|
||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
||||
if _gpu_initialized:
|
||||
return
|
||||
_gpu_initialized = True
|
||||
@ -71,41 +80,80 @@ def _init_gpu():
|
||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||
# Verify GPU is actually accessible
|
||||
_real_cupy.cuda.runtime.getDevice()
|
||||
# Warm-up: run a small computation to verify kernel execution works.
|
||||
# This catches CUDA_ERROR_NO_BINARY_FOR_GPU (compute capability
|
||||
# mismatch) and driver errors before we commit to GPU mode.
|
||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||
_result = _real_cupy.sum(_test * _test)
|
||||
# Force execution (CuPy is lazy — .get() ensures the kernel ran)
|
||||
_ = _result.get()
|
||||
del _test, _result
|
||||
_xp = _real_cupy
|
||||
_cp = _real_cupy
|
||||
_cp_ndimage = _real_cupy_ndimage
|
||||
except (ImportError, Exception) as e:
|
||||
logger.debug(f"CuPy non disponible: {e}")
|
||||
logger.warning(f"GPU non disponible — mode CPU: {e}")
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
HAS_GPU = False
|
||||
|
||||
|
||||
def restrict_gpus(gpu_ids: list[int]):
|
||||
"""Restrict which GPUs are visible to the process.
|
||||
|
||||
Sets CUDA_VISIBLE_DEVICES so only the specified system GPU IDs
|
||||
are accessible. Also updates _available_gpu_ids and _NUM_GPUS.
|
||||
Must be called before any GPU operation.
|
||||
|
||||
Args:
|
||||
gpu_ids: List of system-level GPU indices to make visible.
|
||||
"""
|
||||
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
|
||||
if not gpu_ids or not HAS_GPU:
|
||||
return
|
||||
|
||||
# Validate IDs against total GPU count from nvidia-smi
|
||||
total_count = _NUM_GPUS or 1
|
||||
valid_ids = [gid % total_count for gid in gpu_ids]
|
||||
_available_gpu_ids = valid_ids
|
||||
_NUM_GPUS = len(valid_ids)
|
||||
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
|
||||
logger.info(f"GPU visibles: {_available_gpu_ids}")
|
||||
|
||||
|
||||
def num_gpus():
|
||||
"""Return the total number of CUDA GPUs in the system."""
|
||||
"""Return the number of visible (available) GPUs."""
|
||||
return _NUM_GPUS
|
||||
|
||||
|
||||
def set_active_gpu(gpu_id):
|
||||
"""Set the active GPU for the current process via CUDA_VISIBLE_DEVICES.
|
||||
|
||||
gpu_id is an index into the currently visible GPU list
|
||||
(_available_gpu_ids), not a system-level ID.
|
||||
|
||||
MUST be called before any GPU operation (to_gpu, etc.) to ensure
|
||||
CuPy creates its CUDA context on the correct device. With lazy
|
||||
initialization, CuPy is imported AFTER this call, so it only
|
||||
sees the assigned GPU.
|
||||
|
||||
Args:
|
||||
gpu_id: 0-based GPU index (referring to the system GPU numbering).
|
||||
gpu_id: 0-based index into the visible GPU list.
|
||||
"""
|
||||
if not HAS_GPU or _NUM_GPUS <= 1:
|
||||
return # Nothing to do for single GPU or no GPU
|
||||
|
||||
gpu_id = gpu_id % _NUM_GPUS
|
||||
|
||||
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(gpu_id)
|
||||
# Map visible-GPU index back to the real system GPU ID
|
||||
system_gpu_id = _available_gpu_ids[gpu_id]
|
||||
|
||||
logger.info(f" GPU {gpu_id} sélectionnée pour ce worker")
|
||||
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
|
||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
|
||||
|
||||
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
|
||||
|
||||
|
||||
def _gpu_available():
|
||||
@ -210,4 +258,48 @@ def gpu_cleanup():
|
||||
try:
|
||||
_cp.get_default_memory_pool().free_all_blocks()
|
||||
except Exception:
|
||||
pass
|
||||
pass
|
||||
|
||||
|
||||
def disable_gpu():
|
||||
"""Disable GPU acceleration for the rest of this process.
|
||||
|
||||
Called when a CUDA error indicates the GPU is unusable (e.g.
|
||||
CUDA_ERROR_NO_BINARY_FOR_GPU). Falls back to numpy for all
|
||||
subsequent operations.
|
||||
"""
|
||||
global HAS_GPU, _xp, _cp, _cp_ndimage
|
||||
if not HAS_GPU:
|
||||
return # Already disabled
|
||||
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
||||
HAS_GPU = False
|
||||
_xp = np
|
||||
_cp = None
|
||||
_cp_ndimage = None
|
||||
|
||||
|
||||
def is_gpu_active():
|
||||
"""Check if GPU acceleration is currently active.
|
||||
|
||||
Unlike the HAS_GPU module-level variable (which can go stale if
|
||||
imported directly), this always reflects the current runtime state.
|
||||
Use this in logging tags and conditional GPU paths.
|
||||
"""
|
||||
return HAS_GPU
|
||||
|
||||
|
||||
def safe_gpu_call(func, *args, **kwargs):
|
||||
"""Call a function with GPU arrays, retrying on CPU if GPU fails.
|
||||
|
||||
Usage:
|
||||
result = safe_gpu_call(generate_svf, dem_file, basename, vis_dir, resolution, shared=shared)
|
||||
"""
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
except Exception as e:
|
||||
err_msg = str(e)
|
||||
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg):
|
||||
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
|
||||
disable_gpu()
|
||||
return func(*args, **kwargs)
|
||||
raise
|
||||
Reference in New Issue
Block a user