Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context

The health check now reports what unmanaged VRAM actually costs rather than just
how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit
within 15.42 GB but not the 14.60 GB actually available", naming them.

Getting that arithmetic right took a correction. The first version subtracted only
the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting
against a real ceiling of 14.60 GB -- the same model the service had just refused
with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process
lives, which a purge does not free, so it is not available either. The floor is taken
from the minimum ComfyUI VRAM in recent telemetry rather than its current value,
which could be a 7 GB checkpoint mid-generation.

The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the
30 s idle window, since a large model takes longer than that to load and the purge
was freeing ComfyUI mid-load, so the reclaim path was never reached.

Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model
too large to fit on the card at all is not blamed on the third-party process.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-06 17:44:21 -07:00
parent 81e5d88426
commit aeba1b47fd
3 changed files with 165 additions and 3 deletions

View File

@@ -10,6 +10,7 @@ when it is missing and how to fix it. A degraded dependency should be loud.
""" """
import asyncio import asyncio
import logging import logging
import time
import os import os
import time import time
from typing import Any, Dict, List from typing import Any, Dict, List
@@ -139,6 +140,75 @@ def _check_residency() -> Dict[str, Any]:
cap.get("hint", "")) cap.get("hint", ""))
def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float:
"""Lowest VRAM ComfyUI has been observed holding while alive.
A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred
MB for as long as the process runs. The minimum seen in recent telemetry is a better
estimate of that floor than whatever it happens to hold right now, which could be a
7 GB checkpoint mid-generation.
"""
try:
rows = telemetry_store._rows(
"SELECT MIN(comfy_bytes) AS floor FROM telemetry "
"WHERE ts > ? AND comfy_bytes > 0",
(time.time() - days * 86400,))
if rows and rows[0].get("floor"):
return round(rows[0]["floor"] / (1024 ** 3), 2)
except Exception as e:
logger.debug(f"comfy floor lookup failed: {e}")
return default
def _check_unmanaged_vram() -> Dict[str, Any]:
"""Report unreclaimable VRAM in terms of what it actually costs.
"0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your
models can no longer fit" is something you can act on.
"""
stats = vram_arbitrator.get_gpu_hardware_stats()
if not stats.get("available"):
return _check("unmanaged VRAM", DEGRADED, "GPU unavailable")
bd = stats.get("breakdown", {})
unmanaged_gb = bd.get("unmanaged_gb", 0.0)
procs = bd.get("unmanaged", [])
if not procs:
return _check("unmanaged VRAM", OK, "no third-party GPU processes")
total_gb = stats.get("vram_total_gb", 0)
# What HyperSwap could offer at best. Three things are never available: the desktop,
# processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge.
# Omitting that last one made this check claim a 14.93 GB model would fit against a
# real ceiling of 14.60 GB -- the model that had just returned 507.
comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0))
ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb
try:
blobs = ram_optimizer.find_ollama_model_files()
except Exception:
blobs = []
# Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache
# are allocated.
VRAM_OVERHEAD = 1.16
blocked = sorted(
{b["model"]: b for b in blobs
if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb
and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(),
key=lambda b: -b["size_gb"])
names = ", ".join(b["model"] for b in blocked[:3])
who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2])
if blocked:
return _check("unmanaged VRAM", DEGRADED,
f"{unmanaged_gb} GB held by {who}",
f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB "
f"but not the {ceiling_gb:.2f} GB actually available: {names}",
"Stop that process to reclaim the difference, or accept that "
"these models cannot load")
return _check("unmanaged VRAM", OK,
f"{unmanaged_gb} GB held by {who}; no model is blocked by it",
"", "")
def _check_model_dirs() -> Dict[str, Any]: def _check_model_dirs() -> Dict[str, Any]:
comfy_dir = ram_optimizer.COMFY_MODELS_DIR comfy_dir = ram_optimizer.COMFY_MODELS_DIR
if not os.path.isdir(comfy_dir): if not os.path.isdir(comfy_dir):
@@ -158,7 +228,7 @@ async def run_health_checks() -> Dict[str, Any]:
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control, sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
_check_profile_drift, _check_store, _check_residency, _check_profile_drift, _check_store, _check_residency,
_check_model_dirs, _check_comfy_ws] _check_model_dirs, _check_comfy_ws, _check_unmanaged_vram]
results: List[Dict[str, Any]] = [] results: List[Dict[str, Any]] = []
for fn in sync_checks: for fn in sync_checks:
try: try:

View File

@@ -105,7 +105,8 @@ class TestAggregation:
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0]) monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1]) monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store", for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
"_check_residency", "_check_model_dirs", "_check_comfy_ws"): "_check_residency", "_check_model_dirs", "_check_comfy_ws",
"_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix): async def fake_http(name, url, impact, fix):
@@ -121,7 +122,7 @@ class TestAggregation:
monkeypatch.setattr(health, "_check_nvml", boom) monkeypatch.setattr(health, "_check_nvml", boom)
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift", for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
"_check_store", "_check_residency", "_check_model_dirs", "_check_store", "_check_residency", "_check_model_dirs",
"_check_comfy_ws"): "_check_comfy_ws", "_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix): async def fake_http(name, url, impact, fix):
@@ -132,3 +133,84 @@ class TestAggregation:
# A broken check must surface as failed, not take down the endpoint. # A broken check must surface as failed, not take down the endpoint.
assert res["status"] == health.FAILED assert res["status"] == health.FAILED
assert any("exploded" in c["detail"] for c in res["checks"]) assert any("exploded" in c["detail"] for c in res["checks"])
class TestUnmanagedVramCheck:
"""Turning an unreclaimable-VRAM number into something actionable.
The arithmetic here has to be right or the check is worse than useless. A first
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
model the service had just refused with 507 Insufficient Storage.
"""
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
procs=None):
return {
"available": True,
"vram_total_gb": total,
"breakdown": {
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
"comfyui_gb": comfy_gb,
"unmanaged": procs if procs is not None else
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
"cmdline": "stt_relay.py"}],
},
}
def _blobs(self, sizes):
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
# the observed 507.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([12.87]))
res = health._check_unmanaged_vram()
assert res["status"] == health.DEGRADED
assert "1 model(s)" in res["impact"]
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
# A tiny model fits either way, so the unmanaged process is not what blocks it.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([2.0]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
# process is why would send the user after the wrong thing.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([23.7]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
# survives a purge.
monkeypatch.setattr(health.telemetry_store, "_rows",
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
def boom(*a, **k):
raise RuntimeError("db gone")
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
assert health._comfy_vram_floor_gb(default=0.5) == 0.5

View File

@@ -215,6 +215,16 @@ async def stage_reclaim(c: httpx.AsyncClient, model: str) -> bool:
f"HyperSwap cannot free") f"HyperSwap cannot free")
return True return True
# Re-run a graph first. The idle purge fires 30 s after ComfyUI goes quiet, and a
# large model takes longer than that to load -- so without resetting the timer the
# purge frees ComfyUI mid-load and the reclaim path is never reached.
sys.path.insert(0, "/home/drjones/unified-model-manager")
import autotune # noqa: E402
await autotune._diffusion_benchmark()
gpu = await api(c, "GET", "/api/gpu")
print(f" reset the idle window; ComfyUI holds "
f"{gpu['breakdown']['comfyui_gb']} GB, {gpu['vram_free_gb']} GB free")
res = await api(c, "POST", "/api/switch-model", allow_error=True, res = await api(c, "POST", "/api/switch-model", allow_error=True,
json={"model": model, "keep_alive": "2m"}, timeout=600) json={"model": model, "keep_alive": "2m"}, timeout=600)
if res.get("_status") == 507: if res.get("_status") == 507: