Limit GPU memory pool per worker to prevent OOM with multi-worker

This commit is contained in:
Antoine Jacquin
2026-05-31 19:51:32 +02:00
parent deed10ea62
commit a8fd8addb7

View File

@ -91,6 +91,14 @@ def _init_gpu():
_xp = _real_cupy
_cp = _real_cupy
_cp_ndimage = _real_cupy_ndimage
# Limit GPU memory pool per worker to avoid OOM when multiple
# workers share one GPU. Each worker gets at most 3.5 GB (or
# 50 % of total VRAM on smaller cards).
props = _real_cupy.cuda.runtime.getDeviceProperties(0)
total_mem = props['totalGlobalMem']
max_worker_mem = min(int(total_mem * 0.5), 3.5 * 1024**3)
_real_cupy.cuda.set_memory_pool(0, max_worker_mem)
logger.info(f" Pool mémoire GPU limité à {max_worker_mem // (1024**3) * 1000 // 1024} MB")
except (ImportError, Exception) as e:
logger.warning(f"GPU non disponible — mode CPU: {e}")
_xp = np