Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context
The health check now reports what unmanaged VRAM actually costs rather than just how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit within 15.42 GB but not the 14.60 GB actually available", naming them. Getting that arithmetic right took a correction. The first version subtracted only the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB -- the same model the service had just refused with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process lives, which a purge does not free, so it is not available either. The floor is taken from the minimum ComfyUI VRAM in recent telemetry rather than its current value, which could be a 7 GB checkpoint mid-generation. The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the 30 s idle window, since a large model takes longer than that to load and the purge was freeing ComfyUI mid-load, so the reclaim path was never reached. Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model too large to fit on the card at all is not blamed on the third-party process. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -105,7 +105,8 @@ class TestAggregation:
|
||||
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
|
||||
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
|
||||
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
|
||||
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
|
||||
"_check_residency", "_check_model_dirs", "_check_comfy_ws",
|
||||
"_check_unmanaged_vram"):
|
||||
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||
|
||||
async def fake_http(name, url, impact, fix):
|
||||
@@ -121,7 +122,7 @@ class TestAggregation:
|
||||
monkeypatch.setattr(health, "_check_nvml", boom)
|
||||
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
|
||||
"_check_store", "_check_residency", "_check_model_dirs",
|
||||
"_check_comfy_ws"):
|
||||
"_check_comfy_ws", "_check_unmanaged_vram"):
|
||||
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||
|
||||
async def fake_http(name, url, impact, fix):
|
||||
@@ -132,3 +133,84 @@ class TestAggregation:
|
||||
# A broken check must surface as failed, not take down the endpoint.
|
||||
assert res["status"] == health.FAILED
|
||||
assert any("exploded" in c["detail"] for c in res["checks"])
|
||||
|
||||
|
||||
class TestUnmanagedVramCheck:
|
||||
"""Turning an unreclaimable-VRAM number into something actionable.
|
||||
|
||||
The arithmetic here has to be right or the check is worse than useless. A first
|
||||
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
|
||||
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
|
||||
model the service had just refused with 507 Insufficient Storage.
|
||||
"""
|
||||
|
||||
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
|
||||
procs=None):
|
||||
return {
|
||||
"available": True,
|
||||
"vram_total_gb": total,
|
||||
"breakdown": {
|
||||
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
|
||||
"comfyui_gb": comfy_gb,
|
||||
"unmanaged": procs if procs is not None else
|
||||
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
|
||||
"cmdline": "stt_relay.py"}],
|
||||
},
|
||||
}
|
||||
|
||||
def _blobs(self, sizes):
|
||||
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
|
||||
|
||||
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
|
||||
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
|
||||
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
|
||||
# the observed 507.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([12.87]))
|
||||
res = health._check_unmanaged_vram()
|
||||
assert res["status"] == health.DEGRADED
|
||||
assert "1 model(s)" in res["impact"]
|
||||
|
||||
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
|
||||
# A tiny model fits either way, so the unmanaged process is not what blocks it.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([2.0]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
|
||||
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
|
||||
# process is why would send the user after the wrong thing.
|
||||
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
||||
lambda: self._gpu())
|
||||
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
||||
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
||||
lambda: self._blobs([23.7]))
|
||||
assert health._check_unmanaged_vram()["status"] == health.OK
|
||||
|
||||
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
|
||||
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
|
||||
# survives a purge.
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows",
|
||||
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
|
||||
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
|
||||
|
||||
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
|
||||
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
|
||||
|
||||
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
|
||||
def boom(*a, **k):
|
||||
raise RuntimeError("db gone")
|
||||
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
|
||||
assert health._comfy_vram_floor_gb(default=0.5) == 0.5
|
||||
|
||||
Reference in New Issue
Block a user