Per-GPU warm-up with fallback CPU on NO_BINARY + memory pool limit per worker
This commit is contained in:
@ -67,38 +67,78 @@ def _init_gpu():
|
|||||||
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
|
Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be
|
||||||
set before the CUDA context is created.
|
set before the CUDA context is created.
|
||||||
|
|
||||||
Validates that the GPU can actually execute kernels by running
|
With CUPY_CUDA_COMPILE_WITH_CACHE=1, JIT-compiled kernels are cached
|
||||||
a small computation. This catches CUDA_ERROR_NO_BINARY_FOR_GPU
|
on disk and shared across processes. We warm up on ALL visible GPUs
|
||||||
and other compute capability mismatches before they crash visualizations.
|
so workers that later restrict to a single GPU find pre-compiled kernels.
|
||||||
|
|
||||||
|
Per-GPU warm-up errors are caught individually: if GPU 1 fails to
|
||||||
|
compile a kernel (e.g. CUDA_ERROR_NO_BINARY_FOR_GPU on sm_89), we
|
||||||
|
record it and continue warming up GPU 0. Workers assigned to a
|
||||||
|
failed GPU fall back to CPU automatically.
|
||||||
"""
|
"""
|
||||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU
|
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _gpu_failed_ids
|
||||||
if _gpu_initialized:
|
if _gpu_initialized:
|
||||||
return
|
return
|
||||||
_gpu_initialized = True
|
_gpu_initialized = True
|
||||||
|
_gpu_failed_ids = set()
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import cupy as _real_cupy
|
import cupy as _real_cupy
|
||||||
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
import cupyx.scipy.ndimage as _real_cupy_ndimage
|
||||||
# Verify GPU is actually accessible
|
|
||||||
_real_cupy.cuda.runtime.getDevice()
|
n_devs = _real_cupy.cuda.runtime.getDeviceCount()
|
||||||
# Warm-up: run a small computation to verify kernel execution works.
|
all_failed = False
|
||||||
# This catches CUDA_ERROR_NO_BINARY_FOR_GPU (compute capability
|
|
||||||
# mismatch) and driver errors before we commit to GPU mode.
|
# Warm up each GPU independently — one failure doesn't kill the others.
|
||||||
|
for dev_id in range(n_devs):
|
||||||
|
try:
|
||||||
|
with _real_cupy.cuda.Device(dev_id):
|
||||||
|
props = _real_cupy.cuda.runtime.getDeviceProperties(dev_id)
|
||||||
|
name = props['name'].decode() if isinstance(props['name'], bytes) else props['name']
|
||||||
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
_test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32)
|
||||||
_result = _real_cupy.sum(_test * _test)
|
_result = _real_cupy.sum(_test * _test)
|
||||||
# Force execution (CuPy is lazy — .get() ensures the kernel ran)
|
|
||||||
_ = _result.get()
|
_ = _result.get()
|
||||||
del _test, _result
|
del _test, _result
|
||||||
|
logger.info(f" GPU {dev_id}: {name} — kernels pré-compilés")
|
||||||
|
except Exception as dev_err:
|
||||||
|
_gpu_failed_ids.add(dev_id)
|
||||||
|
logger.warning(
|
||||||
|
f" GPU {dev_id}: échec warm-up ({dev_err.__class__.__name__}: "
|
||||||
|
f"{dev_err}) — workers sur ce GPU passeront en CPU"
|
||||||
|
)
|
||||||
|
|
||||||
|
# If ALL visible GPUs failed, disable GPU entirely.
|
||||||
|
# This is critical for worker subprocesses that restricted
|
||||||
|
# CUDA_VISIBLE_DEVICES to a single GPU before importing CuPy.
|
||||||
|
if len(_gpu_failed_ids) >= n_devs:
|
||||||
|
all_failed = True
|
||||||
|
logger.warning("Tous les GPUs échoués au warm-up — passage en mode CPU")
|
||||||
|
|
||||||
|
if not all_failed:
|
||||||
_xp = _real_cupy
|
_xp = _real_cupy
|
||||||
_cp = _real_cupy
|
_cp = _real_cupy
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
|
||||||
# Limit GPU memory pool per worker to avoid OOM when multiple
|
# Limit GPU memory pool per worker to avoid OOM when multiple
|
||||||
# workers share one GPU. Each worker gets at most 3.5 GB (or
|
# workers share one GPU. Each worker gets at most 3.5 GB (or
|
||||||
# 50 % of total VRAM on smaller cards).
|
# 50 % of total VRAM on smaller cards).
|
||||||
|
try:
|
||||||
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
|
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
|
||||||
total_mem = props['totalGlobalMem']
|
total_mem = props['totalGlobalMem']
|
||||||
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
|
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
|
||||||
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
|
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
|
||||||
logger.info(f" Pool mémoire GPU limité à {max_worker_mem // (1024**3) * 1000 // 1024} MB")
|
logger.info(
|
||||||
|
f" Pool mémoire GPU limité à "
|
||||||
|
f"{max_worker_mem // (1024**3) * 1000 // 1024} MB"
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
pass # pool config is best-effort
|
||||||
|
else:
|
||||||
|
_xp = np
|
||||||
|
_cp = None
|
||||||
|
_cp_ndimage = None
|
||||||
|
HAS_GPU = False
|
||||||
|
|
||||||
except (ImportError, Exception) as e:
|
except (ImportError, Exception) as e:
|
||||||
logger.warning(f"GPU non disponible — mode CPU: {e}")
|
logger.warning(f"GPU non disponible — mode CPU: {e}")
|
||||||
_xp = np
|
_xp = np
|
||||||
@ -107,15 +147,21 @@ def _init_gpu():
|
|||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
|
|
||||||
|
|
||||||
def restrict_gpus(gpu_ids: list[int]):
|
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
||||||
"""Restrict which GPUs are visible to the process.
|
"""Restrict which GPUs are visible to the process.
|
||||||
|
|
||||||
Sets CUDA_VISIBLE_DEVICES so only the specified system GPU IDs
|
By default (set_env_var=False), only records the GPU IDs for
|
||||||
are accessible. Also updates _available_gpu_ids and _NUM_GPUS.
|
num_gpus() and logging. Does NOT touch CUDA_VISIBLE_DEVICES
|
||||||
Must be called before any GPU operation.
|
in the main process because CuPy 13.x JIT compilation (sm_89)
|
||||||
|
needs ALL GPUs visible during warm-up to pre-compile kernels.
|
||||||
|
|
||||||
|
Workers call this with set_env_var=True (via assign_gpu_to_worker)
|
||||||
|
after forking, when CuPy hasn't been imported yet in the child.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
gpu_ids: List of system-level GPU indices to make visible.
|
gpu_ids: List of system-level GPU indices to make visible.
|
||||||
|
set_env_var: If True, actually set CUDA_VISIBLE_DEVICES.
|
||||||
|
Default False (safe for main process).
|
||||||
"""
|
"""
|
||||||
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
|
global _NUM_GPUS, HAS_GPU, _available_gpu_ids
|
||||||
if not gpu_ids or not HAS_GPU:
|
if not gpu_ids or not HAS_GPU:
|
||||||
@ -127,6 +173,7 @@ def restrict_gpus(gpu_ids: list[int]):
|
|||||||
_available_gpu_ids = valid_ids
|
_available_gpu_ids = valid_ids
|
||||||
_NUM_GPUS = len(valid_ids)
|
_NUM_GPUS = len(valid_ids)
|
||||||
|
|
||||||
|
if set_env_var:
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
|
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids)
|
||||||
logger.info(f"GPU visibles: {_available_gpu_ids}")
|
logger.info(f"GPU visibles: {_available_gpu_ids}")
|
||||||
|
|
||||||
@ -147,6 +194,10 @@ def set_active_gpu(gpu_id):
|
|||||||
initialization, CuPy is imported AFTER this call, so it only
|
initialization, CuPy is imported AFTER this call, so it only
|
||||||
sees the assigned GPU.
|
sees the assigned GPU.
|
||||||
|
|
||||||
|
If the assigned GPU failed warm-up (e.g. NO_BINARY_FOR_GPU), this
|
||||||
|
function disables GPU for this worker so all operations fall back
|
||||||
|
to CPU without crashing.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
gpu_id: 0-based index into the visible GPU list.
|
gpu_id: 0-based index into the visible GPU list.
|
||||||
"""
|
"""
|
||||||
@ -158,7 +209,19 @@ def set_active_gpu(gpu_id):
|
|||||||
# Map visible-GPU index back to the real system GPU ID
|
# Map visible-GPU index back to the real system GPU ID
|
||||||
system_gpu_id = _available_gpu_ids[gpu_id]
|
system_gpu_id = _available_gpu_ids[gpu_id]
|
||||||
|
|
||||||
# Set CUDA_VISIBLE_DEVICES before CuPy context creation
|
# If this GPU failed warm-up, disable GPU for this worker
|
||||||
|
if system_gpu_id in _gpu_failed_ids:
|
||||||
|
logger.warning(
|
||||||
|
f" GPU {system_gpu_id} échouée au warm-up — "
|
||||||
|
f"worker passe en mode CPU"
|
||||||
|
)
|
||||||
|
disable_gpu()
|
||||||
|
return
|
||||||
|
|
||||||
|
# Set CUDA_VISIBLE_DEVICES to isolate this worker to one GPU.
|
||||||
|
# This MUST happen before CuPy is imported (lazy init).
|
||||||
|
# The JIT kernels were already pre-compiled by the main process
|
||||||
|
# on all GPUs, so the worker finds them in the cache.
|
||||||
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
|
os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id)
|
||||||
|
|
||||||
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
|
logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")
|
||||||
|
|||||||
Reference in New Issue
Block a user