Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context
The health check now reports what unmanaged VRAM actually costs rather than just how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit within 15.42 GB but not the 14.60 GB actually available", naming them. Getting that arithmetic right took a correction. The first version subtracted only the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB -- the same model the service had just refused with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process lives, which a purge does not free, so it is not available either. The floor is taken from the minimum ComfyUI VRAM in recent telemetry rather than its current value, which could be a 7 GB checkpoint mid-generation. The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the 30 s idle window, since a large model takes longer than that to load and the purge was freeing ComfyUI mid-load, so the reclaim path was never reached. Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model too large to fit on the card at all is not blamed on the third-party process. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
72
health.py
72
health.py
@@ -10,6 +10,7 @@ when it is missing and how to fix it. A degraded dependency should be loud.
|
||||
"""
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
import os
|
||||
import time
|
||||
from typing import Any, Dict, List
|
||||
@@ -139,6 +140,75 @@ def _check_residency() -> Dict[str, Any]:
|
||||
cap.get("hint", ""))
|
||||
|
||||
|
||||
def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float:
|
||||
"""Lowest VRAM ComfyUI has been observed holding while alive.
|
||||
|
||||
A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred
|
||||
MB for as long as the process runs. The minimum seen in recent telemetry is a better
|
||||
estimate of that floor than whatever it happens to hold right now, which could be a
|
||||
7 GB checkpoint mid-generation.
|
||||
"""
|
||||
try:
|
||||
rows = telemetry_store._rows(
|
||||
"SELECT MIN(comfy_bytes) AS floor FROM telemetry "
|
||||
"WHERE ts > ? AND comfy_bytes > 0",
|
||||
(time.time() - days * 86400,))
|
||||
if rows and rows[0].get("floor"):
|
||||
return round(rows[0]["floor"] / (1024 ** 3), 2)
|
||||
except Exception as e:
|
||||
logger.debug(f"comfy floor lookup failed: {e}")
|
||||
return default
|
||||
|
||||
|
||||
def _check_unmanaged_vram() -> Dict[str, Any]:
|
||||
"""Report unreclaimable VRAM in terms of what it actually costs.
|
||||
|
||||
"0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your
|
||||
models can no longer fit" is something you can act on.
|
||||
"""
|
||||
stats = vram_arbitrator.get_gpu_hardware_stats()
|
||||
if not stats.get("available"):
|
||||
return _check("unmanaged VRAM", DEGRADED, "GPU unavailable")
|
||||
bd = stats.get("breakdown", {})
|
||||
unmanaged_gb = bd.get("unmanaged_gb", 0.0)
|
||||
procs = bd.get("unmanaged", [])
|
||||
if not procs:
|
||||
return _check("unmanaged VRAM", OK, "no third-party GPU processes")
|
||||
|
||||
total_gb = stats.get("vram_total_gb", 0)
|
||||
# What HyperSwap could offer at best. Three things are never available: the desktop,
|
||||
# processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge.
|
||||
# Omitting that last one made this check claim a 14.93 GB model would fit against a
|
||||
# real ceiling of 14.60 GB -- the model that had just returned 507.
|
||||
comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0))
|
||||
ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb
|
||||
try:
|
||||
blobs = ram_optimizer.find_ollama_model_files()
|
||||
except Exception:
|
||||
blobs = []
|
||||
# Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache
|
||||
# are allocated.
|
||||
VRAM_OVERHEAD = 1.16
|
||||
blocked = sorted(
|
||||
{b["model"]: b for b in blobs
|
||||
if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb
|
||||
and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(),
|
||||
key=lambda b: -b["size_gb"])
|
||||
|
||||
names = ", ".join(b["model"] for b in blocked[:3])
|
||||
who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2])
|
||||
if blocked:
|
||||
return _check("unmanaged VRAM", DEGRADED,
|
||||
f"{unmanaged_gb} GB held by {who}",
|
||||
f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB "
|
||||
f"but not the {ceiling_gb:.2f} GB actually available: {names}",
|
||||
"Stop that process to reclaim the difference, or accept that "
|
||||
"these models cannot load")
|
||||
return _check("unmanaged VRAM", OK,
|
||||
f"{unmanaged_gb} GB held by {who}; no model is blocked by it",
|
||||
"", "")
|
||||
|
||||
|
||||
def _check_model_dirs() -> Dict[str, Any]:
|
||||
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
|
||||
if not os.path.isdir(comfy_dir):
|
||||
@@ -158,7 +228,7 @@ async def run_health_checks() -> Dict[str, Any]:
|
||||
|
||||
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
|
||||
_check_profile_drift, _check_store, _check_residency,
|
||||
_check_model_dirs, _check_comfy_ws]
|
||||
_check_model_dirs, _check_comfy_ws, _check_unmanaged_vram]
|
||||
results: List[Dict[str, Any]] = []
|
||||
for fn in sync_checks:
|
||||
try:
|
||||
|
||||
@@ -105,7 +105,8 @@ class TestAggregation:
|
||||
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
|
||||
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
|
||||
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
|
||||
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
|
||||
"_check_residency", "_check_model_dirs", "_check_comfy_ws",
|
||||
"_check_unmanaged_vram"):
|
||||
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||
|
||||
async def fake_http(name, url, impact, fix):
|
||||
@@ -121,7 +122,7 @@ class TestAggregation:
|
||||
monkeypatch.setattr(health, "_check_nvml", boom)
|
||||
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
|
||||
"_check_store", "_check_residency", "_check_model_dirs",
|
||||
"_check_comfy_ws"):
|
||||
"_check_comfy_ws", "_check_unmanaged_vram"):
|
||||
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||
|
||||
async def fake_http(name, url, impact, fix):
|
||||
@@ -132,3 +133,84 @@ class TestAggregation:
|
||||
# A broken check must surface as failed, not take down the endpoint.
|
||||
assert res["status"] == health.FAILED
|
||||
assert any("exploded" in c["detail"] for c in res["checks"])
|
||||
|
||||
|
||||
class TestUnmanagedVramCheck:
|
||||
"""Turning an unreclaimable-VRAM number into something actionable.
|
||||
|
||||
The arithmetic here has to be right or the check is worse than useless. A first
|
||||
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
|
||||
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
|
||||
model the service had just refused with 507 Insufficient Storage.
|
||||
"""
|
||||
|
||||
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
|
||||
procs=None):
|
||||
return {
|
||||
"available": True,
|
||||
"vram_total_gb": total,
|
||||
"breakdown": {
|
||||
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
|
||||
"comfyui_gb": comfy_gb,
|
||||
"unmanaged": procs if procs is not None else
|
||||
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
|
||||
"cmdline": "stt_relay.py"}],
|
||||
},
|
||||
}
|
||||
|
||||
def _blobs(self, sizes):
|
||||
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
|
||||
|
||||
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
|
||||
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
|
||||
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
|
||||
# the observed 507.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([12.87]))
|
||||
res = health._check_unmanaged_vram()
|
||||
assert res["status"] == health.DEGRADED
|
||||
assert "1 model(s)" in res["impact"]
|
||||
|
||||
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
|
||||
# A tiny model fits either way, so the unmanaged process is not what blocks it.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([2.0]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
|
||||
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
|
||||
# process is why would send the user after the wrong thing.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([23.7]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
|
||||
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
|
||||
# survives a purge.
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows",
|
||||
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
|
||||
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
|
||||
|
||||
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
|
||||
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
|
||||
|
||||
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
|
||||
def boom(*a, **k):
|
||||
raise RuntimeError("db gone")
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
|
||||
assert health._comfy_vram_floor_gb(default=0.5) == 0.5
|
||||
|
||||
@@ -215,6 +215,16 @@ async def stage_reclaim(c: httpx.AsyncClient, model: str) -> bool:
|
||||
f"HyperSwap cannot free")
|
||||
return True
|
||||
|
||||
# Re-run a graph first. The idle purge fires 30 s after ComfyUI goes quiet, and a
|
||||
# large model takes longer than that to load -- so without resetting the timer the
|
||||
# purge frees ComfyUI mid-load and the reclaim path is never reached.
|
||||
sys.path.insert(0, "/home/drjones/unified-model-manager")
|
||||
import autotune # noqa: E402
|
||||
await autotune._diffusion_benchmark()
|
||||
gpu = await api(c, "GET", "/api/gpu")
|
||||
print(f" reset the idle window; ComfyUI holds "
|
||||
f"{gpu['breakdown']['comfyui_gb']} GB, {gpu['vram_free_gb']} GB free")
|
||||
|
||||
res = await api(c, "POST", "/api/switch-model", allow_error=True,
|
||||
json={"model": model, "keep_alive": "2m"}, timeout=600)
|
||||
if res.get("_status") == 507:
|
||||
|
||||
Reference in New Issue
Block a user