diff --git a/health.py b/health.py index edf1f91..47807fd 100644 --- a/health.py +++ b/health.py @@ -10,6 +10,7 @@ when it is missing and how to fix it. A degraded dependency should be loud. """ import asyncio import logging +import time import os import time from typing import Any, Dict, List @@ -139,6 +140,75 @@ def _check_residency() -> Dict[str, Any]: cap.get("hint", "")) +def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float: + """Lowest VRAM ComfyUI has been observed holding while alive. + + A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred + MB for as long as the process runs. The minimum seen in recent telemetry is a better + estimate of that floor than whatever it happens to hold right now, which could be a + 7 GB checkpoint mid-generation. + """ + try: + rows = telemetry_store._rows( + "SELECT MIN(comfy_bytes) AS floor FROM telemetry " + "WHERE ts > ? AND comfy_bytes > 0", + (time.time() - days * 86400,)) + if rows and rows[0].get("floor"): + return round(rows[0]["floor"] / (1024 ** 3), 2) + except Exception as e: + logger.debug(f"comfy floor lookup failed: {e}") + return default + + +def _check_unmanaged_vram() -> Dict[str, Any]: + """Report unreclaimable VRAM in terms of what it actually costs. + + "0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your + models can no longer fit" is something you can act on. + """ + stats = vram_arbitrator.get_gpu_hardware_stats() + if not stats.get("available"): + return _check("unmanaged VRAM", DEGRADED, "GPU unavailable") + bd = stats.get("breakdown", {}) + unmanaged_gb = bd.get("unmanaged_gb", 0.0) + procs = bd.get("unmanaged", []) + if not procs: + return _check("unmanaged VRAM", OK, "no third-party GPU processes") + + total_gb = stats.get("vram_total_gb", 0) + # What HyperSwap could offer at best. Three things are never available: the desktop, + # processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge. + # Omitting that last one made this check claim a 14.93 GB model would fit against a + # real ceiling of 14.60 GB -- the model that had just returned 507. + comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0)) + ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb + try: + blobs = ram_optimizer.find_ollama_model_files() + except Exception: + blobs = [] + # Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache + # are allocated. + VRAM_OVERHEAD = 1.16 + blocked = sorted( + {b["model"]: b for b in blobs + if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb + and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(), + key=lambda b: -b["size_gb"]) + + names = ", ".join(b["model"] for b in blocked[:3]) + who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2]) + if blocked: + return _check("unmanaged VRAM", DEGRADED, + f"{unmanaged_gb} GB held by {who}", + f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB " + f"but not the {ceiling_gb:.2f} GB actually available: {names}", + "Stop that process to reclaim the difference, or accept that " + "these models cannot load") + return _check("unmanaged VRAM", OK, + f"{unmanaged_gb} GB held by {who}; no model is blocked by it", + "", "") + + def _check_model_dirs() -> Dict[str, Any]: comfy_dir = ram_optimizer.COMFY_MODELS_DIR if not os.path.isdir(comfy_dir): @@ -158,7 +228,7 @@ async def run_health_checks() -> Dict[str, Any]: sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control, _check_profile_drift, _check_store, _check_residency, - _check_model_dirs, _check_comfy_ws] + _check_model_dirs, _check_comfy_ws, _check_unmanaged_vram] results: List[Dict[str, Any]] = [] for fn in sync_checks: try: diff --git a/tests/test_health.py b/tests/test_health.py index 63e6d2e..fe26bb0 100644 --- a/tests/test_health.py +++ b/tests/test_health.py @@ -105,7 +105,8 @@ class TestAggregation: monkeypatch.setattr(health, "_check_nvml", lambda: checks[0]) monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1]) for fn in ("_check_fan_control", "_check_profile_drift", "_check_store", - "_check_residency", "_check_model_dirs", "_check_comfy_ws"): + "_check_residency", "_check_model_dirs", "_check_comfy_ws", + "_check_unmanaged_vram"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): @@ -121,7 +122,7 @@ class TestAggregation: monkeypatch.setattr(health, "_check_nvml", boom) for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift", "_check_store", "_check_residency", "_check_model_dirs", - "_check_comfy_ws"): + "_check_comfy_ws", "_check_unmanaged_vram"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): @@ -132,3 +133,84 @@ class TestAggregation: # A broken check must surface as failed, not take down the endpoint. assert res["status"] == health.FAILED assert any("exploded" in c["detail"] for c in res["checks"]) + + +class TestUnmanagedVramCheck: + """Turning an unreclaimable-VRAM number into something actionable. + + The arithmetic here has to be right or the check is worse than useless. A first + version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported + a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very + model the service had just refused with 507 Insufficient Storage. + """ + + def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99, + procs=None): + return { + "available": True, + "vram_total_gb": total, + "breakdown": { + "unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb, + "comfyui_gb": comfy_gb, + "unmanaged": procs if procs is not None else + [{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024, + "cmdline": "stt_relay.py"}], + }, + } + + def _blobs(self, sizes): + return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)] + + def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch): + monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", + lambda: self._gpu(unmanaged_gb=0.0, procs=[])) + assert health._check_unmanaged_vram()["status"] == health.OK + + def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch): + # 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available. + # A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching + # the observed 507. + monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", + lambda: self._gpu()) + monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) + monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", + lambda: self._blobs([12.87])) + res = health._check_unmanaged_vram() + assert res["status"] == health.DEGRADED + assert "1 model(s)" in res["impact"] + + def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch): + # A tiny model fits either way, so the unmanaged process is not what blocks it. + monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", + lambda: self._gpu()) + monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) + monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", + lambda: self._blobs([2.0])) + assert health._check_unmanaged_vram()["status"] == health.OK + + def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch): + # A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB + # process is why would send the user after the wrong thing. + monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", + lambda: self._gpu()) + monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) + monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", + lambda: self._blobs([23.7])) + assert health._check_unmanaged_vram()["status"] == health.OK + + def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch): + # Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what + # survives a purge. + monkeypatch.setattr(health.telemetry_store, "_rows", + lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}]) + assert health._comfy_vram_floor_gb(default=7.0) == 0.24 + + def test_floor_falls_back_when_history_is_empty(self, monkeypatch): + monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: []) + assert health._comfy_vram_floor_gb(default=0.56) == 0.56 + + def test_floor_falls_back_rather_than_raising(self, monkeypatch): + def boom(*a, **k): + raise RuntimeError("db gone") + monkeypatch.setattr(health.telemetry_store, "_rows", boom) + assert health._comfy_vram_floor_gb(default=0.5) == 0.5 diff --git a/verify_arbitration.py b/verify_arbitration.py index 93f8fac..def0cc4 100755 --- a/verify_arbitration.py +++ b/verify_arbitration.py @@ -215,6 +215,16 @@ async def stage_reclaim(c: httpx.AsyncClient, model: str) -> bool: f"HyperSwap cannot free") return True + # Re-run a graph first. The idle purge fires 30 s after ComfyUI goes quiet, and a + # large model takes longer than that to load -- so without resetting the timer the + # purge frees ComfyUI mid-load and the reclaim path is never reached. + sys.path.insert(0, "/home/drjones/unified-model-manager") + import autotune # noqa: E402 + await autotune._diffusion_benchmark() + gpu = await api(c, "GET", "/api/gpu") + print(f" reset the idle window; ComfyUI holds " + f"{gpu['breakdown']['comfyui_gb']} GB, {gpu['vram_free_gb']} GB free") + res = await api(c, "POST", "/api/switch-model", allow_error=True, json={"model": model, "keep_alive": "2m"}, timeout=600) if res.get("_status") == 507: