diff --git a/README.md b/README.md index ef5de6f..be7ac9e 100644 --- a/README.md +++ b/README.md @@ -185,6 +185,33 @@ which contradicts the hardware fails loudly rather than silently: | Busy yield | VRAM held at ≥50% GPU utilisation | A mid-generation model is finishing, not failing | | Residency confidence | probe trusted only at 100% | A 12-window probe once cleared 90% on a mostly-cold file | +## 1b. End-to-End Verification + +```bash +python verify_arbitration.py # full cycle, a few minutes +python verify_arbitration.py --quick # skip the diffusion stages +``` + +The unit suite covers logic in isolation. This exercises the promise the service exists +to make — an LLM and a diffusion pipeline sharing one 16 GB card — against real hardware, +and reports what actually happened at each stage. It restores what it changes and refuses +to start if ComfyUI is busy. + +A representative run on this machine: + +| Stage | Result | +| :--- | :--- | +| LLM load, classified by achieved bandwidth | 1.96 GB in 1327 ms → 1.47 GB/s → Partial Cache | +| VRAM yield confirmed against NVML | released in 43 ms, 2.39 GB freed | +| Diffusion, cold (includes checkpoint load) | 17863 ms → 1.12 it/s | +| Diffusion, warm | 3645 ms → **5.49 it/s** | +| ComfyUI retains its checkpoint | 7.03 GB held through the idle window | +| VRAM attribution adds up | 15.58 GB attributed vs 15.80 GB NVML — Ollama 7.71 + ComfyUI 7.03 coexisting | +| Reported GPU state matches hardware | profile asks 320 W, card reports 320 W | + +The stages report warnings rather than passes when they did not actually prove anything — +a reclaim that was never needed is not evidence that reclaiming works. + ## 2. Architectural Overview ```mermaid diff --git a/verify_arbitration.py b/verify_arbitration.py index 933ee6f..93f8fac 100755 --- a/verify_arbitration.py +++ b/verify_arbitration.py @@ -39,8 +39,13 @@ def record(stage: str, status: str, detail: str, evidence: str = "") -> None: print(f" {evidence}") -async def api(client: httpx.AsyncClient, method: str, path: str, **kw) -> Any: +async def api(client: httpx.AsyncClient, method: str, path: str, + allow_error: bool = False, **kw) -> Any: r = await client.request(method, f"{BASE}{path}", **kw) + if allow_error: + # Some stages deliberately provoke a failure and need to read it. + body = r.json() if r.headers.get("content-type", "").startswith("application/json") else {} + return {"_status": r.status_code, **(body if isinstance(body, dict) else {})} r.raise_for_status() return r.json() @@ -112,12 +117,21 @@ async def stage_diffusion(c: httpx.AsyncClient) -> bool: sys.path.insert(0, "/home/drjones/unified-model-manager") import autotune, vram_arbitrator # noqa: E402 (imported late; needs the service's deps) - t0 = time.perf_counter() + # The first run loads the checkpoint from disk. Timing that and calling the result + # "it/s" understates throughput by roughly 10x -- 0.67 it/s against a steady-state + # 6.7 -- so the load is measured separately and reported as what it is. + first = await autotune._diffusion_benchmark() + if not first.get("ok"): + record("diffusion runs", FAIL, first.get("error", "")[:90]) + return False + record("diffusion runs (cold, includes checkpoint load)", PASS, + f"{first['exec_ms']} ms", f"{first['it_per_sec']} it/s including load") + res = await autotune._diffusion_benchmark() if not res.get("ok"): - record("diffusion runs", FAIL, res.get("error", "")[:90]) + record("diffusion runs (warm)", FAIL, res.get("error", "")[:90]) return False - record("diffusion runs", PASS, + record("diffusion throughput (warm)", PASS, f"SDXL 1024/20 steps in {res['exec_ms']} ms", f"{res['it_per_sec']} it/s") snap = vram_arbitrator.get_process_vram_bytes() @@ -129,39 +143,102 @@ async def stage_diffusion(c: httpx.AsyncClient) -> bool: async def stage_idle_purge(c: httpx.AsyncClient) -> bool: - stats = await api(c, "GET", "/api/stats") - arb = stats["arbitrator"] + # The completion event arrives over the ComfyUI websocket, so the flag is set a + # moment after the graph returns. Checking instantly raced it. + arb = {} + for _ in range(12): + arb = (await api(c, "GET", "/api/stats"))["arbitrator"] + if arb.get("pending_purge"): + break + await asyncio.sleep(0.5) if not arb.get("pending_purge"): record("purge is deferred, not immediate", WARN, - "no purge pending (ComfyUI may already be clean)") + "no purge pending after 6 s (ComfyUI may already be clean)") return True + idle_s = arb.get("comfy_idle_s") record("purge is deferred, not immediate", PASS, f"holding checkpoints for {arb.get('idle_purge_after_s')} s", - f"idle {arb.get('comfy_idle_s')} s so far") + f"idle {idle_s} s so far" if idle_s is not None + else "idle timer just started") return True async def stage_reclaim(c: httpx.AsyncClient, model: str) -> bool: - """The direction that used to fail outright: an LLM that will not fit.""" + """The direction that used to fail outright: an LLM that will not fit. + + This only proves anything if the chosen model genuinely cannot fit in what ComfyUI + has left free. A small model fits alongside the checkpoint and the stage passes + without exercising the reclaim path at all, so pick the largest model that will not + fit and say plainly when no such model exists. + """ gpu = await api(c, "GET", "/api/gpu") comfy_gb = gpu["breakdown"]["comfyui_gb"] + free_gb = gpu["vram_free_gb"] if comfy_gb < 0.5: record("reclaims VRAM for the LLM", SKIP, f"ComfyUI only holds {comfy_gb} GB; nothing to reclaim") return True - res = await api(c, "POST", "/api/switch-model", - json={"model": model, "keep_alive": "2m"}, timeout=300) - if not res.get("success"): - record("reclaims VRAM for the LLM", FAIL, res.get("error", "")[:90], - str(res.get("unmanaged_blockers", ""))[:120]) + + models = (await api(c, "GET", "/api/models"))["ollama_models"] + EMBED = {"bert", "nomic-bert", "gte", "jina-bert"} + usable = [m for m in models + if (m.get("details", {}).get("family") or "").lower() not in EMBED + and "embed" not in m["name"].lower()] + # On-disk weight size is not the VRAM footprint: measured on this box, a 12.87 GB + # blob occupies 14.9 GB once context and KV cache are allocated. Sizing the test off + # disk size picks a model that cannot fit even after a successful reclaim. + VRAM_OVERHEAD = 1.18 + HEADROOM_GB = 0.4 + + def vram_need(m): + return m.get("size", 0) / (1024 ** 3) * VRAM_OVERHEAD + + # Unmanaged VRAM never comes back, so it is not part of what a reclaim can offer. + # Ignoring it picked a model that failed even after a correct reclaim -- on this box + # an 842 MB third-party process is the difference between a 14.9 GB model fitting + # and not. + reclaimable_gb = free_gb + comfy_gb - HEADROOM_GB + too_big = [m for m in usable + if vram_need(m) > free_gb and vram_need(m) < reclaimable_gb] + if too_big: + target = max(too_big, key=lambda m: m.get("size", 0)) + model = target["name"] + print(f" using {model} ({target['size'] / (1024**3):.1f} GB on disk, " + f"~{vram_need(target):.1f} GB in VRAM) — will not fit in " + f"{free_gb:.1f} GB free, should fit after reclaiming {comfy_gb:.1f} GB") + else: + unmanaged = gpu["breakdown"].get("unmanaged_gb", 0) + record("reclaims VRAM for the LLM", SKIP, + f"no installed model needs between {free_gb:.1f} and " + f"{reclaimable_gb:.1f} GB of VRAM", + f"reclaimable ceiling excludes {unmanaged} GB held by processes " + f"HyperSwap cannot free") + return True + + res = await api(c, "POST", "/api/switch-model", allow_error=True, + json={"model": model, "keep_alive": "2m"}, timeout=600) + if res.get("_status") == 507: + record("reclaims VRAM for the LLM", FAIL, + "reclaim ran but the model still did not fit", + str(res.get("detail", ""))[:150]) return False + if res.get("_status", 200) >= 400 or not res.get("success"): + record("reclaims VRAM for the LLM", FAIL, + f"HTTP {res.get('_status')}", str(res.get("detail", ""))[:120]) + return False + if res.get("_status") and res.get("_status") != 200: + pass if res.get("reclaimed_from_comfyui_gb"): record("reclaims VRAM for the LLM", PASS, f"reclaimed {res['reclaimed_from_comfyui_gb']} GB and retried", - f"loaded at {res.get('load_gbps')} GB/s") + f"'{model}' then loaded at {res.get('load_gbps')} GB/s") else: - record("reclaims VRAM for the LLM", PASS, - "fit without needing a reclaim", f"{res.get('load_gbps')} GB/s") + # It fit anyway, so nothing was proven; do not report that as a pass. The usual + # cause is the idle purge firing during the load and freeing ComfyUI first. + record("reclaims VRAM for the LLM", WARN, + "model fit without a reclaim, so the path was not exercised", + "the idle purge most likely freed ComfyUI during the load; " + f"loaded at {res.get('load_gbps')} GB/s") return True