Fix multi-GPU detection, VRAM leak on disable, and orphaned GPU array handling
_gpu_candidates was never populated (typo: _candidate_gpus), breaking num_gpus()/available_gpu_ids() and the multi-GPU round-robin. available_gpu_ids also indexed tuples with dict syntax (TypeError). disable_gpu() now frees the memory pool before clearing refs (was leaking VRAM). to_cpu() and safe_gpu_call() survive disable_gpu() via a persistent _cupy_ndarray type reference so orphaned GPU arrays are recovered to CPU. log_gpu_status() now uses Device(0) instead of the host index (CuPy renumbers to 0 after CUDA_VISIBLE_DEVICES).
This commit is contained in:
@ -28,7 +28,7 @@ _gpu_reason = None
|
|||||||
_restricted_gpu_ids: list[int] | None = None
|
_restricted_gpu_ids: list[int] | None = None
|
||||||
|
|
||||||
# Discovered GPU candidates (populated by _pick_gpu)
|
# Discovered GPU candidates (populated by _pick_gpu)
|
||||||
_gpu_candidates: list[dict] = []
|
_gpu_candidates: list = []
|
||||||
|
|
||||||
|
|
||||||
def _pick_gpu() -> list:
|
def _pick_gpu() -> list:
|
||||||
@ -66,9 +66,8 @@ def _pick_gpu() -> list:
|
|||||||
return []
|
return []
|
||||||
|
|
||||||
|
|
||||||
_candidate_gpus: list = []
|
|
||||||
try:
|
try:
|
||||||
_candidate_gpus = _pick_gpu() or []
|
_gpu_candidates = _pick_gpu() or []
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
@ -78,6 +77,7 @@ except Exception:
|
|||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
_cp_ndimage = None
|
_cp_ndimage = None
|
||||||
|
_cupy_ndarray = None # type ref qui survit à disable_gpu() pour que to_cpu() récupère les tableaux orphelins
|
||||||
_gpu_initialized = False
|
_gpu_initialized = False
|
||||||
|
|
||||||
|
|
||||||
@ -112,12 +112,12 @@ def _init_gpu():
|
|||||||
import CuPy directly (no subprocess test needed)
|
import CuPy directly (no subprocess test needed)
|
||||||
3. Otherwise: test each GPU in a subprocess, pick the first that works
|
3. Otherwise: test each GPU in a subprocess, pick the first that works
|
||||||
"""
|
"""
|
||||||
global _xp, _cp, _cp_ndimage, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb
|
global _xp, _cp, _cp_ndimage, _cupy_ndarray, _gpu_initialized, HAS_GPU, _best_gpu_id, _gpu_name, _gpu_mem_gb
|
||||||
if _gpu_initialized:
|
if _gpu_initialized:
|
||||||
return
|
return
|
||||||
_gpu_initialized = True
|
_gpu_initialized = True
|
||||||
|
|
||||||
candidates = _filter_candidates(_candidate_gpus)
|
candidates = _filter_candidates(_gpu_candidates)
|
||||||
if not candidates:
|
if not candidates:
|
||||||
logger.info("Pas de GPU utilisable — mode CPU uniquement")
|
logger.info("Pas de GPU utilisable — mode CPU uniquement")
|
||||||
_xp = np
|
_xp = np
|
||||||
@ -149,6 +149,7 @@ def _init_gpu():
|
|||||||
_xp = _real_cupy
|
_xp = _real_cupy
|
||||||
_cp = _real_cupy
|
_cp = _real_cupy
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
_cupy_ndarray = _real_cupy.ndarray
|
||||||
return
|
return
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning(f"GPU indisponible (CUDA_VISIBLE_DEVICES={cuda_visible}): {e}")
|
logger.warning(f"GPU indisponible (CUDA_VISIBLE_DEVICES={cuda_visible}): {e}")
|
||||||
@ -203,6 +204,7 @@ def _init_gpu():
|
|||||||
_xp = _real_cupy
|
_xp = _real_cupy
|
||||||
_cp = _real_cupy
|
_cp = _real_cupy
|
||||||
_cp_ndimage = _real_cupy_ndimage
|
_cp_ndimage = _real_cupy_ndimage
|
||||||
|
_cupy_ndarray = _real_cupy.ndarray
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@ -211,7 +213,7 @@ def _init_gpu():
|
|||||||
|
|
||||||
def num_gpus():
|
def num_gpus():
|
||||||
"""Return the number of available GPUs (after restrict_gpus filtering)."""
|
"""Return the number of available GPUs (after restrict_gpus filtering)."""
|
||||||
return len(_gpu_candidates)
|
return len(_filter_candidates(_gpu_candidates))
|
||||||
|
|
||||||
|
|
||||||
def available_gpu_ids():
|
def available_gpu_ids():
|
||||||
@ -219,16 +221,20 @@ def available_gpu_ids():
|
|||||||
|
|
||||||
Respects any prior restrict_gpus() call.
|
Respects any prior restrict_gpus() call.
|
||||||
"""
|
"""
|
||||||
return [c['id'] for c in _gpu_candidates]
|
return [c[0] for c in _filter_candidates(_gpu_candidates)]
|
||||||
|
|
||||||
|
|
||||||
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
def restrict_gpus(gpu_ids: list[int], set_env_var: bool = False):
|
||||||
"""Restrict GPU selection to specific host-level indices.
|
"""Restrict GPU selection to specific host-level indices.
|
||||||
|
|
||||||
Stores the restriction to be applied during _init_gpu().
|
Stores the restriction to be applied during _init_gpu().
|
||||||
|
If set_env_var is True, also sets CUDA_VISIBLE_DEVICES immediately so
|
||||||
|
child processes inherit the restriction.
|
||||||
"""
|
"""
|
||||||
global _restricted_gpu_ids
|
global _restricted_gpu_ids
|
||||||
_restricted_gpu_ids = gpu_ids
|
_restricted_gpu_ids = gpu_ids
|
||||||
|
if set_env_var and gpu_ids:
|
||||||
|
os.environ['CUDA_VISIBLE_DEVICES'] = ','.join(str(i) for i in gpu_ids)
|
||||||
|
|
||||||
|
|
||||||
def set_active_gpu(gpu_id):
|
def set_active_gpu(gpu_id):
|
||||||
@ -250,11 +256,12 @@ def log_gpu_status():
|
|||||||
"""Log GPU detection result. Called after logging is configured."""
|
"""Log GPU detection result. Called after logging is configured."""
|
||||||
if _gpu_available():
|
if _gpu_available():
|
||||||
try:
|
try:
|
||||||
with _cp.cuda.Device(_best_gpu_id):
|
with _cp.cuda.Device(0):
|
||||||
name = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['name']
|
props = _cp.cuda.runtime.getDeviceProperties(0)
|
||||||
|
name = props['name']
|
||||||
if isinstance(name, bytes):
|
if isinstance(name, bytes):
|
||||||
name = name.decode()
|
name = name.decode()
|
||||||
mem_gb = _cp.cuda.runtime.getDeviceProperties(_best_gpu_id)['totalGlobalMem'] // (1024**3)
|
mem_gb = props['totalGlobalMem'] // (1024**3)
|
||||||
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
gpu_info = f"GPU: {name} ({mem_gb} Go VRAM) — ID {_best_gpu_id}"
|
||||||
except Exception:
|
except Exception:
|
||||||
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
gpu_info = f"GPU: {_gpu_name} ({_gpu_mem_gb} Go VRAM)"
|
||||||
@ -297,10 +304,18 @@ def to_gpu(arr):
|
|||||||
|
|
||||||
|
|
||||||
def to_cpu(arr):
|
def to_cpu(arr):
|
||||||
"""Bring array back to CPU (numpy). No-op if already on CPU."""
|
"""Bring array back to CPU (numpy). No-op if already on CPU.
|
||||||
if _cp is not None and isinstance(arr, _cp.ndarray):
|
|
||||||
|
Fonctionne même après disable_gpu() grâce à la réf de type _cupy_ndarray.
|
||||||
|
"""
|
||||||
|
if _cupy_ndarray is not None and isinstance(arr, _cupy_ndarray):
|
||||||
try:
|
try:
|
||||||
|
if _cp is not None:
|
||||||
return _cp.asnumpy(arr)
|
return _cp.asnumpy(arr)
|
||||||
|
return arr.get()
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
return np.asarray(arr)
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
return arr
|
return arr
|
||||||
@ -360,11 +375,16 @@ def gpu_cleanup():
|
|||||||
|
|
||||||
|
|
||||||
def disable_gpu():
|
def disable_gpu():
|
||||||
"""Disable GPU acceleration for the rest of this process."""
|
"""Disable GPU acceleration for the rest of this process.
|
||||||
|
|
||||||
|
Libère le memory pool GPU avant de nuller les références (anti-fuite VRAM).
|
||||||
|
Garde _cupy_ndarray vivant pour que to_cpu() récupère les tableaux orphelins.
|
||||||
|
"""
|
||||||
global HAS_GPU, _xp, _cp, _cp_ndimage
|
global HAS_GPU, _xp, _cp, _cp_ndimage
|
||||||
if not HAS_GPU:
|
if not HAS_GPU:
|
||||||
return
|
return
|
||||||
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
logger.warning("GPU désactivé — passage en mode CPU pour la suite du processus")
|
||||||
|
gpu_cleanup()
|
||||||
HAS_GPU = False
|
HAS_GPU = False
|
||||||
_xp = np
|
_xp = np
|
||||||
_cp = None
|
_cp = None
|
||||||
@ -385,5 +405,7 @@ def safe_gpu_call(func, *args, **kwargs):
|
|||||||
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg):
|
if _cp is not None and ('CUDA' in err_msg or 'cuda' in err_msg or 'GPU' in err_msg or 'Out of memory' in err_msg):
|
||||||
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
|
logger.warning(f"Erreur GPU ({e.__class__.__name__}), retry en CPU...")
|
||||||
disable_gpu()
|
disable_gpu()
|
||||||
return func(*args, **kwargs)
|
cpu_args = tuple(to_cpu(a) for a in args)
|
||||||
|
cpu_kwargs = {k: to_cpu(v) for k, v in kwargs.items()}
|
||||||
|
return func(*cpu_args, **cpu_kwargs)
|
||||||
raise
|
raise
|
||||||
|
|||||||
Reference in New Issue
Block a user