Per-GPU warm-up with fallback CPU on NO_BINARY + memory pool limit per worker

This commit is contained in:
Antoine Jacquin
2026-05-31 20:30:22 +02:00
parent a8fd8addb7
commit d2f382c94d

View File

@ -67,38 +67,78 @@ def _init_gpu():
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
set before the CUDA context is created. set before the CUDA context is created.
Validates that the GPU can actually execute kernels by running With CUPY_CUDA_COMPILE_WITH_CACHE=1, JIT-compiled kernels are cached
a small computation. This catches CUDA_ERROR_NO_BINARY_FOR_GPU on disk and shared across processes. We warm up on ALL visible GPUs
and other compute capability mismatches before they crash visualizations. so workers that later restrict to a single GPU find pre-compiled kernels.
Per-GPU warm-up errors are caught individually: if GPU 1 fails to
compile a kernel (e.g. CUDA_ERROR_NO_BINARY_FOR_GPU on sm_89), we
record it and continue warming up GPU 0. Workers assigned to a
failed GPU fall back to CPU automatically.
""" """
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _gpu_failed_ids
if _gpu_initialized: if _gpu_initialized:
return return
_gpu_initialized = True _gpu_initialized = True
_gpu_failed_ids = set()
try: try:
import cupy as _real_cupy import cupy as _real_cupy
import cupyx.scipy.ndimage as _real_cupy_ndimage import cupyx.scipy.ndimage as _real_cupy_ndimage
# Verify GPU is actually accessible
_real_cupy.cuda.runtime.getDevice() n_devs = _real_cupy.cuda.runtime.getDeviceCount()
# Warm-up: run a small computation to verify kernel execution works. all_failed = False
# This catches CUDA_ERROR_NO_BINARY_FOR_GPU (compute capability
# mismatch) and driver errors before we commit to GPU mode. # Warm up each GPU independently — one failure doesn't kill the others.
for dev_id in range(n_devs):
try:
with _real_cupy.cuda.Device(dev_id):
props = _real_cupy.cuda.runtime.getDeviceProperties(dev_id)
name = props['name'].decode() if isinstance(props['name'], bytes) else props['name']
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
_result = _real_cupy.sum(_test * _test) _result = _real_cupy.sum(_test * _test)
# Force execution (CuPy is lazy — .get() ensures the kernel ran)
_ = _result.get() _ = _result.get()
del _test, _result del _test, _result
logger.info(f" GPU {dev_id}: {name} — kernels pré-compilés")
except Exception as dev_err:
_gpu_failed_ids.add(dev_id)
logger.warning(
f" GPU {dev_id}: échec warm-up ({dev_err.__class__.__name__}: "
f"{dev_err}) — workers sur ce GPU passeront en CPU"
)
# If ALL visible GPUs failed, disable GPU entirely.
# This is critical for worker subprocesses that restricted
# CUDA_VISIBLE_DEVICES to a single GPU before importing CuPy.
if len(_gpu_failed_ids) >= n_devs:
all_failed = True
logger.warning("Tous les GPUs échoués au warm-up — passage en mode CPU")
if not all_failed:
_xp = _real_cupy _xp = _real_cupy
_cp = _real_cupy _cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage _cp_ndimage = _real_cupy_ndimage
# Limit GPU memory pool per worker to avoid OOM when multiple # Limit GPU memory pool per worker to avoid OOM when multiple
# workers share one GPU. Each worker gets at most 3.5 GB (or # workers share one GPU. Each worker gets at most 3.5 GB (or
# 50 % of total VRAM on smaller cards). # 50 % of total VRAM on smaller cards).
try:
props = _real_cupy.cuda.runtime.getDeviceProperties(0) props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem'] total_mem = props['totalGlobalMem']
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3) max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
_real_cupy.cuda.set_memory_pool(0, max_worker_mem) _real_cupy.cuda.set_memory_pool(0, max_worker_mem)
logger.info(f" Pool mémoire GPU limité à {max_worker_mem // (1024**3) * 1000 // 1024} MB") logger.info(
f" Pool mémoire GPU limité à "
f"{max_worker_mem // (1024**3) * 1000 // 1024} MB"
)
except Exception:
pass # pool config is best-effort
else:
_xp = np
_cp = None
_cp_ndimage = None
HAS_GPU = False
except (ImportError, Exception) as e: except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}") logger.warning(f"GPU non disponible — mode CPU: {e}")
_xp = np _xp = np
@ -107,15 +147,21 @@ def _init_gpu():
HAS_GPU = False HAS_GPU = False
def restrict_gpus(gpu_ids: list[int]): def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
"""Restrict which GPUs are visible to the process. """Restrict which GPUs are visible to the process.
Sets CUDA_VISIBLE_DEVICES so only the specified system GPU IDs By default (set_env_var=False), only records the GPU IDs for
are accessible. Also updates _available_gpu_ids and _NUM_GPUS. num_gpus() and logging. Does NOT touch CUDA_VISIBLE_DEVICES
Must be called before any GPU operation. in the main process because CuPy 13.x JIT compilation (sm_89)
needs ALL GPUs visible during warm-up to pre-compile kernels.
Workers call this with set_env_var=True (via assign_gpu_to_worker)
after forking, when CuPy hasn't been imported yet in the child.
Args: Args:
gpu_ids: List of system-level GPU indices to make visible. gpu_ids: List of system-level GPU indices to make visible.
set_env_var: If True, actually set CUDA_VISIBLE_DEVICES.
Default False (safe for main process).
""" """
global _NUM_GPUS, HAS_GPU, _available_gpu_ids global _NUM_GPUS, HAS_GPU, _available_gpu_ids
if not gpu_ids or not HAS_GPU: if not gpu_ids or not HAS_GPU:
@ -127,6 +173,7 @@ def restrict_gpus(gpu_ids: list[int]):
_available_gpu_ids = valid_ids _available_gpu_ids = valid_ids
_NUM_GPUS = len(valid_ids) _NUM_GPUS = len(valid_ids)
if set_env_var:
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids) os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
logger.info(f"GPU visibles: {_available_gpu_ids}") logger.info(f"GPU visibles: {_available_gpu_ids}")
@ -147,6 +194,10 @@ def set_active_gpu(gpu_id):
initialization, CuPy is imported AFTER this call, so it only initialization, CuPy is imported AFTER this call, so it only
sees the assigned GPU. sees the assigned GPU.
If the assigned GPU failed warm-up (e.g. NO_BINARY_FOR_GPU), this
function disables GPU for this worker so all operations fall back
to CPU without crashing.
Args: Args:
gpu_id: 0-based index into the visible GPU list. gpu_id: 0-based index into the visible GPU list.
""" """
@ -158,7 +209,19 @@ def set_active_gpu(gpu_id):
# Map visible-GPU index back to the real system GPU ID # Map visible-GPU index back to the real system GPU ID
system_gpu_id = _available_gpu_ids[gpu_id] system_gpu_id = _available_gpu_ids[gpu_id]
# Set CUDA_VISIBLE_DEVICES before CuPy context creation # If this GPU failed warm-up, disable GPU for this worker
if system_gpu_id in _gpu_failed_ids:
logger.warning(
f" GPU {system_gpu_id} échouée au warm-up — "
f"worker passe en mode CPU"
)
disable_gpu()
return
# Set CUDA_VISIBLE_DEVICES to isolate this worker to one GPU.
# This MUST happen before CuPy is imported (lazy init).
# The JIT kernels were already pre-compiled by the main process
# on all GPUs, so the worker finds them in the cache.
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id) os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker") logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")