Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context

The health check now reports what unmanaged VRAM actually costs rather than just
how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit
within 15.42 GB but not the 14.60 GB actually available", naming them.

Getting that arithmetic right took a correction. The first version subtracted only
the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting
against a real ceiling of 14.60 GB -- the same model the service had just refused
with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process
lives, which a purge does not free, so it is not available either. The floor is taken
from the minimum ComfyUI VRAM in recent telemetry rather than its current value,
which could be a 7 GB checkpoint mid-generation.

The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the
30 s idle window, since a large model takes longer than that to load and the purge
was freeing ComfyUI mid-load, so the reclaim path was never reached.

Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model
too large to fit on the card at all is not blamed on the third-party process.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-06 17:44:21 -07:00
parent 81e5d88426
commit aeba1b47fd
3 changed files with 165 additions and 3 deletions

View File

@@ -105,7 +105,8 @@ class TestAggregation:
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
"_check_residency", "_check_model_dirs", "_check_comfy_ws",
"_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
@@ -121,7 +122,7 @@ class TestAggregation:
monkeypatch.setattr(health, "_check_nvml", boom)
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
"_check_store", "_check_residency", "_check_model_dirs",
"_check_comfy_ws"):
"_check_comfy_ws", "_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
@@ -132,3 +133,84 @@ class TestAggregation:
# A broken check must surface as failed, not take down the endpoint.
assert res["status"] == health.FAILED
assert any("exploded" in c["detail"] for c in res["checks"])
class TestUnmanagedVramCheck:
"""Turning an unreclaimable-VRAM number into something actionable.
The arithmetic here has to be right or the check is worse than useless. A first
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
model the service had just refused with 507 Insufficient Storage.
"""
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
procs=None):
return {
"available": True,
"vram_total_gb": total,
"breakdown": {
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
"comfyui_gb": comfy_gb,
"unmanaged": procs if procs is not None else
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
"cmdline": "stt_relay.py"}],
},
}
def _blobs(self, sizes):
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
# the observed 507.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([12.87]))
res = health._check_unmanaged_vram()
assert res["status"] == health.DEGRADED
assert "1 model(s)" in res["impact"]
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
# A tiny model fits either way, so the unmanaged process is not what blocks it.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([2.0]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
# process is why would send the user after the wrong thing.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([23.7]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
# survives a purge.
monkeypatch.setattr(health.telemetry_store, "_rows",
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
def boom(*a, **k):
raise RuntimeError("db gone")
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
assert health._comfy_vram_floor_gb(default=0.5) == 0.5