From c6455d7c6ec6d3978e1e288cb5b785299f437ee6 Mon Sep 17 00:00:00 2001 From: drjones Date: Mon, 7 Sep 2026 15:16:10 -0700 Subject: [PATCH] Remove the duplicated starvation path; show every tenant; cache-bust assets _check_ollama_starved and _arbitrate were solving the same problem, one of them hardcoded to two applications. The Ollama-specific version is gone and the watchdog calls only the generic loop. The OOM retry inside switch_ollama_model no longer purges ComfyUI by name either: it asks plan_release which tenant should give up memory, so a third application can be the one that yields, and when the reclaim is not enough the response names the blockers instead of implying ComfyUI was at fault. The dashboard showed exactly two engines, which no longer matched what the service does. A GPU Tenants panel lists every configured application ordered by priority -- VRAM held, whether it is working, whether it can be reclaimed at all, and how much it needs -- along with the last arbitration decision and why it could or could not be satisfied. That panel did not appear at first, and the reason is worth fixing rather than working around: the browser kept serving a cached app.js despite the ETag, so a reload ran the old dashboard against the new API. Assets are now stamped with their mtime, so a changed file is always a different URL. Anyone updating this service would have hit the same thing. Also verified along the way that an apparent horizontal-overflow regression was a measurement artifact from a zero-width browser pane, not a real layout fault. Co-Authored-By: Claude Opus 5 --- README.md | 4 +- server.py | 16 ++++++- static/app.js | 54 ++++++++++++++++++++++ static/index.html | 19 ++++++++ vram_arbitrator.py | 111 ++++++++++----------------------------------- 5 files changed, 115 insertions(+), 89 deletions(-) diff --git a/README.md b/README.md index 803a6c1..26fd597 100644 --- a/README.md +++ b/README.md @@ -215,7 +215,9 @@ acted on, which makes the decision testable and loggable. Ollama, ComfyUI and the desktop compositor ship as defaults, so behaviour is unchanged — but nothing in the arbitration logic knows their names, and three applications can -contend for the card as easily as two. Endpoints are generic: +contend for the card as easily as two. The dashboard's **GPU Tenants** panel lists all of +them ordered by the priority arbitration actually considers, with the last decision and +why it could or could not be satisfied. Endpoints are generic: `GET /api/tenants`, `GET /api/tenants/{name}`, `POST /api/tenants/{name}/release`. A tenant with `"release": {"type": "none"}` is still worth declaring. The 842 MB speech diff --git a/server.py b/server.py index 075d10f..d90a833 100644 --- a/server.py +++ b/server.py @@ -695,9 +695,23 @@ app.mount("/static", StaticFiles(directory=f"{BASE_DIR}/static"), name="static") @app.get("/", summary="Dashboard Web UI", tags=["UI"]) async def root_index(): + """Serve the dashboard with cache-busted asset URLs. + + StaticFiles sends an ETag, but browsers were still serving app.js from cache after + it changed, so a reload showed the old dashboard against the new API -- a panel that + had just been added simply never appeared. Stamping each asset with its mtime means + a changed file is always a different URL. + """ with open(f"{BASE_DIR}/static/index.html", "r") as f: content = f.read() - return HTMLResponse(content=content) + for asset in ("app.js", "styles.css"): + try: + stamp = int(os.path.getmtime(f"{BASE_DIR}/static/{asset}")) + except OSError: + continue + content = content.replace(f"/static/{asset}", f"/static/{asset}?v={stamp}") + return HTMLResponse(content=content, + headers={"Cache-Control": "no-cache, must-revalidate"}) if __name__ == "__main__": import uvicorn diff --git a/static/app.js b/static/app.js index 70c519f..c008843 100644 --- a/static/app.js +++ b/static/app.js @@ -36,6 +36,7 @@ function updateDashboard(data) { // Governor and arbitration state ride along in the shared snapshot. if (data.governor) renderGovernor(data.governor); if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu); + if (data.arbitrator) renderTenants(data.arbitrator, data.gpu); // 1. GPU VRAM Stats const gpu = data.gpu || {}; @@ -1135,3 +1136,56 @@ function watchChartSize() { } document.addEventListener('DOMContentLoaded', () => setTimeout(watchChartSize, 200)); + + +// ---------------------------------------------------------------- tenants + +function renderTenants(arb, gpu) { + const body = document.getElementById('tenants-body'); + if (!body) return; + const state = arb && arb.tenant_state; + if (!state || !state.tenants) return; + + const total = (gpu && gpu.vram_total_gb) || 16; + document.getElementById('tenants-free').textContent = `${state.free_gb} GB free`; + + // Sorted by priority, the order arbitration actually considers them in. + const rows = [...state.tenants].sort((a, b) => b.priority - a.priority); + body.innerHTML = rows.map(t => { + const pct = Math.min((t.vram_gb / total) * 100, 100); + const bar = t.busy ? 'bg-emerald-500' + : t.reclaimable ? 'bg-cyan-600' : 'bg-amber-600'; + const badge = t.busy + ? 'working' + : t.reclaimable + ? 'idle · reclaimable' + : 'cannot be reclaimed'; + return `
+
+ ${t.name} + p${t.priority} + ${t.vram_gb.toFixed(2)} GB · ${badge} +
+
+
+
+
${t.reason || ''}${ + t.needs_vram_gb ? ` · needs ${t.needs_vram_gb} GB to work` : ''}
+
`; + }).join(''); + + // The most recent arbitration decision, including why it could not be satisfied. + const dec = document.getElementById('tenants-decision'); + const a = arb.last_arbitration; + if (!a) { + dec.innerHTML = 'No contention — nothing has needed to be released.'; + return; + } + const when = new Date(a.ts * 1000).toLocaleTimeString(); + const blockers = (a.blockers || []) + .map(b => `${b.name} (${b.vram_gb} GB, ${b.why})`).join(', '); + dec.innerHTML = + `${when} · ` + + `${a.demanding} short by ${a.shortfall_gb} GB — ${a.reason}` + + (blockers ? `
blocked by: ${blockers}
` : ''); +} diff --git a/static/index.html b/static/index.html index 07391c8..4da909c 100644 --- a/static/index.html +++ b/static/index.html @@ -680,6 +680,25 @@
+ + +
+
+
+
+ +
+
+

GPU Tenants

+

Every application contending for the card, from tenants.json

+
+
+ — +
+
+
+
+
diff --git a/vram_arbitrator.py b/vram_arbitrator.py index 9fdd965..fc5be6d 100644 --- a/vram_arbitrator.py +++ b/vram_arbitrator.py @@ -73,7 +73,8 @@ RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2 # depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU # and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and # returns a hard CUDA OOM instead. Both are handled -- the spill by -# AutoArbitrator._check_ollama_starved, the hard failure by the retry below. +# AutoArbitrator._arbitrate (generically, from the tenant registry), the hard failure by +# the retry below. OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate", "failed to allocate", "cuda error") @@ -860,33 +861,41 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m", # idle ComfyUI and try once more. body = resp.text if looks_like_vram_oom(body) and not _retrying: - snap = get_process_vram_bytes() - if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES: + # Which application should give up memory is a question for the registry, + # not something to answer by purging ComfyUI by name. Any reclaimable idle + # tenant below Ollama in priority is a candidate. + state = await arbitrator._tenant_state() + free_gb = arbitrator._last_tenant_state["free_gb"] + size_gb = _model_size_bytes(target_model) / (1024**3) + needed = size_gb * 1.16 if size_gb else free_gb + 1.0 + plan = tenants_mod.plan_release("ollama", state, free_gb, needed) + if plan["release"]: logger.warning( - f"Ollama could not fit '{target_model}' with ComfyUI holding " - f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying") - purge = await instant_free_comfyui_vram() + f"Ollama could not fit '{target_model}' — {plan['reason']}") + freed_before = free_gb + for victim in plan["release"]: + await arbitrator._release_tenant( + victim, f"Ollama could not load '{target_model}'") arbitrator.stats["reclaims_for_ollama"] += 1 arbitrator.last_action = ( - f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so " - f"'{target_model}' could load") + f"Released {', '.join(plan['release'])} so '{target_model}' could load") _record({ "event_type": "VRAM Reclaim for Ollama", - "source": "ComfyUI Pipeline", + "source": ", ".join(plan["release"]), "target": target_model, - "duration_ms": purge.get("duration_ms"), "cache_status": "Reclaimed", "detail": f"Ollama OOM: {body[:160]}", }) await asyncio.sleep(0.3) retry = await switch_ollama_model(target_model, keep_alive, _retrying=True) - retry["reclaimed_from_comfyui_gb"] = round( - snap["comfyui_bytes"] / (1024**3), 2) + retry["released_tenants"] = plan["release"] + retry["would_free_gb"] = plan.get("would_free_gb") retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM" if not retry.get("success"): - # Be specific about why the reclaim was not enough. Blaming ComfyUI - # when a third-party process is holding the memory sends the user - # looking in the wrong place. + # Be specific about why the reclaim was not enough. Blaming a tenant + # when a process nobody can release is holding the memory sends the + # user looking in the wrong place. + retry["blockers"] = plan.get("blockers") retry["unmanaged_blockers"] = describe_unmanaged() return retry return {"success": False, "error": f"HTTP {resp.status_code}: {body}", @@ -942,7 +951,6 @@ class AutoArbitrator: self._yield_backoff_until: Dict[str, float] = {} self._yield_busy_streak: Dict[str, int] = {} self.last_reclaim_time = 0.0 - self.last_starvation_check: Optional[Dict[str, Any]] = None self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0, "bad_status": 0, "error": 0} self._running_id: Optional[str] = None @@ -1301,75 +1309,6 @@ class AutoArbitrator: f"{int(now - since)}s") return - async def _check_ollama_starved(self) -> None: - """The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU. - - Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the - README calling the arbitration bidirectional. When Ollama cannot fit a model it - does not fail, it silently places layers on the CPU and runs about an order of - magnitude slower -- so this is the failure mode a user is least likely to notice - and most likely to feel. - - If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached - checkpoints are the thing to give up. - """ - # Every bail-out is recorded. This check silently did nothing while a model sat - # at 29% on the GPU, and with four separate early returns there was no way to - # tell which one had fired without guessing. - now = time.time() - def bail(reason: str, **extra): - self.last_starvation_check = {"ts": now, "acted": False, - "reason": reason, **extra} - - if self.comfy_was_active: - return bail("ComfyUI is active; it needs the VRAM itself") - if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S: - return bail("within reclaim cooldown", - seconds_left=round(self.RECLAIM_COOLDOWN_S - - (now - self.last_reclaim_time), 1)) - - ollama = await get_ollama_live_state() - if not ollama.get("partially_offloaded"): - return bail("LLM is not spilling to CPU", - gpu_fraction=ollama.get("gpu_fraction"), - model=ollama.get("active_model_name")) - - snap = get_process_vram_bytes() - if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES: - # ComfyUI is not the one holding the memory; nothing we can do here. - return bail("LLM is spilling but ComfyUI holds too little to help", - cpu_offload_pct=ollama.get("cpu_offload_pct"), - comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2)) - - self.last_starvation_check = {"ts": now, "acted": True, - "reason": "reclaiming for the LLM", - "cpu_offload_pct": ollama.get("cpu_offload_pct")} - - self.last_reclaim_time = now - model = ollama.get("active_model_name") - offload = ollama.get("cpu_offload_pct") - logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds " - f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM") - await self._purge_comfy_now(f"LLM spilling {offload}% to CPU") - self.stats["reclaims_for_ollama"] += 1 - - # Freeing VRAM does not move layers back; only a reload re-places the model. Do - # that only when the model is idle, never mid-generation. - after = get_process_vram_bytes() - if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model: - logger.info(f"Reloading '{model}' to place it fully on the GPU...") - await instant_free_ollama_vram(model, confirm=True) - res = await switch_ollama_model(model, keep_alive="30m") - recheck = await get_ollama_live_state() - self.last_action = ( - f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and " - f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU" - if res.get("success") else - f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}") - else: - self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will " - f"stay partly on CPU until its next load") - async def _poll_watchdog(self): """Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz. @@ -1392,7 +1331,6 @@ class AutoArbitrator: await self.trigger_comfy_completed() else: self.watchdog_branches["idle_check"] += 1 - await self._check_ollama_starved() await self._arbitrate() else: self.watchdog_branches["bad_status"] += 1 @@ -1440,7 +1378,6 @@ class AutoArbitrator: "idle_purge_after_s": self.COMFY_IDLE_PURGE_S, "oc_profile": self.oc_profile, "counters": dict(self.stats), - "last_starvation_check": self.last_starvation_check, "comfy_stale_job": self.comfy_stale_job, "last_arbitration": self.last_arbitration, "tenant_state": self._last_tenant_state,