From d2f382c94d5b9b814008342f420d10e4fc609921 Mon Sep 17 00:00:00 2001 From: Antoine Jacquin Date: Sun, 31 May 2026 20:30:22 +0200 Subject: [PATCH] Per-GPU warm-up with fallback CPU on NO_BINARY + memory pool limit per worker --- lidar_pipeline/gpu.py | 125 +++++++++++++++++++++++++++++++----------- 1 file changed, 94 insertions(+), 31 deletions(-) diff --git a/lidar_pipeline/gpu.py b/lidar_pipeline/gpu.py index 5b3650b..5477566 100644 --- a/lidar_pipeline/gpu.py +++ b/lidar_pipeline/gpu.py @@ -67,38 +67,78 @@ def _init_gpu(): Import CuPy only when needed, so CUDA_VISIBLE_DEVICES can be set before the CUDA context is created. - Validates that the GPU can actually execute kernels by running - a small computation. This catches CUDA_ERROR_NO_BINARY_FOR_GPU - and other compute capability mismatches before they crash visualizations. + With CUPY_CUDA_COMPILE_WITH_CACHE=1, JIT-compiled kernels are cached + on disk and shared across processes. We warm up on ALL visible GPUs + so workers that later restrict to a single GPU find pre-compiled kernels. + + Per-GPU warm-up errors are caught individually: if GPU 1 fails to + compile a kernel (e.g. CUDA_ERROR_NO_BINARY_FOR_GPU on sm_89), we + record it and continue warming up GPU 0. Workers assigned to a + failed GPU fall back to CPU automatically. """ - global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU + global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _gpu_failed_ids if _gpu_initialized: return _gpu_initialized = True + _gpu_failed_ids = set() + try: import cupy as _real_cupy import cupyx.scipy.ndimage as _real_cupy_ndimage - # Verify GPU is actually accessible - _real_cupy.cuda.runtime.getDevice() - # Warm-up: run a small computation to verify kernel execution works. - # This catches CUDA_ERROR_NO_BINARY_FOR_GPU (compute capability - # mismatch) and driver errors before we commit to GPU mode. - _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) - _result = _real_cupy.sum(_test * _test) - # Force execution (CuPy is lazy — .get() ensures the kernel ran) - _ = _result.get() - del _test, _result - _xp = _real_cupy - _cp = _real_cupy - _cp_ndimage = _real_cupy_ndimage - # Limit GPU memory pool per worker to avoid OOM when multiple - # workers share one GPU. Each worker gets at most 3.5 GB (or - # 50 % of total VRAM on smaller cards). - props = _real_cupy.cuda.runtime.getDeviceProperties(0) - total_mem = props['totalGlobalMem'] - max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3) - _real_cupy.cuda.set_memory_pool(0, max_worker_mem) - logger.info(f" Pool mémoire GPU limité à {max_worker_mem // (1024**3) * 1000 // 1024} MB") + + n_devs = _real_cupy.cuda.runtime.getDeviceCount() + all_failed = False + + # Warm up each GPU independently — one failure doesn't kill the others. + for dev_id in range(n_devs): + try: + with _real_cupy.cuda.Device(dev_id): + props = _real_cupy.cuda.runtime.getDeviceProperties(dev_id) + name = props['name'].decode() if isinstance(props['name'], bytes) else props['name'] + _test = _real_cupy.array([1.0, 2.0, 3.0], dtype=_real_cupy.float32) + _result = _real_cupy.sum(_test * _test) + _ = _result.get() + del _test, _result + logger.info(f" GPU {dev_id}: {name} — kernels pré-compilés") + except Exception as dev_err: + _gpu_failed_ids.add(dev_id) + logger.warning( + f" GPU {dev_id}: échec warm-up ({dev_err.__class__.__name__}: " + f"{dev_err}) — workers sur ce GPU passeront en CPU" + ) + + # If ALL visible GPUs failed, disable GPU entirely. + # This is critical for worker subprocesses that restricted + # CUDA_VISIBLE_DEVICES to a single GPU before importing CuPy. + if len(_gpu_failed_ids) >= n_devs: + all_failed = True + logger.warning("Tous les GPUs échoués au warm-up — passage en mode CPU") + + if not all_failed: + _xp = _real_cupy + _cp = _real_cupy + _cp_ndimage = _real_cupy_ndimage + + # Limit GPU memory pool per worker to avoid OOM when multiple + # workers share one GPU. Each worker gets at most 3.5 GB (or + # 50 % of total VRAM on smaller cards). + try: + props = _real_cupy.cuda.runtime.getDeviceProperties(0) + total_mem = props['totalGlobalMem'] + max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3) + _real_cupy.cuda.set_memory_pool(0, max_worker_mem) + logger.info( + f" Pool mémoire GPU limité à " + f"{max_worker_mem // (1024**3) * 1000 // 1024} MB" + ) + except Exception: + pass # pool config is best-effort + else: + _xp = np + _cp = None + _cp_ndimage = None + HAS_GPU = False + except (ImportError, Exception) as e: logger.warning(f"GPU non disponible — mode CPU: {e}") _xp = np @@ -107,15 +147,21 @@ def _init_gpu(): HAS_GPU = False -def restrict_gpus(gpu_ids: list[int]): +def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False): """Restrict which GPUs are visible to the process. - Sets CUDA_VISIBLE_DEVICES so only the specified system GPU IDs - are accessible. Also updates _available_gpu_ids and _NUM_GPUS. - Must be called before any GPU operation. + By default (set_env_var=False), only records the GPU IDs for + num_gpus() and logging. Does NOT touch CUDA_VISIBLE_DEVICES + in the main process because CuPy 13.x JIT compilation (sm_89) + needs ALL GPUs visible during warm-up to pre-compile kernels. + + Workers call this with set_env_var=True (via assign_gpu_to_worker) + after forking, when CuPy hasn't been imported yet in the child. Args: gpu_ids: List of system-level GPU indices to make visible. + set_env_var: If True, actually set CUDA_VISIBLE_DEVICES. + Default False (safe for main process). """ global _NUM_GPUS, HAS_GPU, _available_gpu_ids if not gpu_ids or not HAS_GPU: @@ -127,7 +173,8 @@ def restrict_gpus(gpu_ids: list[int]): _available_gpu_ids = valid_ids _NUM_GPUS = len(valid_ids) - os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids) + if set_env_var: + os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(g) for g in valid_ids) logger.info(f"GPU visibles: {_available_gpu_ids}") @@ -147,6 +194,10 @@ def set_active_gpu(gpu_id): initialization, CuPy is imported AFTER this call, so it only sees the assigned GPU. + If the assigned GPU failed warm-up (e.g. NO_BINARY_FOR_GPU), this + function disables GPU for this worker so all operations fall back + to CPU without crashing. + Args: gpu_id: 0-based index into the visible GPU list. """ @@ -158,7 +209,19 @@ def set_active_gpu(gpu_id): # Map visible-GPU index back to the real system GPU ID system_gpu_id = _available_gpu_ids[gpu_id] - # Set CUDA_VISIBLE_DEVICES before CuPy context creation + # If this GPU failed warm-up, disable GPU for this worker + if system_gpu_id in _gpu_failed_ids: + logger.warning( + f" GPU {system_gpu_id} échouée au warm-up — " + f"worker passe en mode CPU" + ) + disable_gpu() + return + + # Set CUDA_VISIBLE_DEVICES to isolate this worker to one GPU. + # This MUST happen before CuPy is imported (lazy init). + # The JIT kernels were already pre-compiled by the main process + # on all GPUs, so the worker finds them in the cache. os.environ['CUDA_VISIBLE_DEVICES'] = str(system_gpu_id) logger.info(f" GPU {system_gpu_id} sélectionnée pour ce worker")