diff --git a/README.md b/README.md index 803a6c1..26fd597 100644 --- a/README.md +++ b/README.md @@ -215,7 +215,9 @@ acted on, which makes the decision testable and loggable. Ollama, ComfyUI and the desktop compositor ship as defaults, so behaviour is unchanged — but nothing in the arbitration logic knows their names, and three applications can -contend for the card as easily as two. Endpoints are generic: +contend for the card as easily as two. The dashboard's **GPU Tenants** panel lists all of +them ordered by the priority arbitration actually considers, with the last decision and +why it could or could not be satisfied. Endpoints are generic: `GET /api/tenants`, `GET /api/tenants/{name}`, `POST /api/tenants/{name}/release`. A tenant with `"release": {"type": "none"}` is still worth declaring. The 842 MB speech diff --git a/server.py b/server.py index 075d10f..d90a833 100644 --- a/server.py +++ b/server.py @@ -695,9 +695,23 @@ app.mount("/static", StaticFiles(directory=f"{BASE_DIR}/static"), name="static") @app.get("/", summary="Dashboard Web UI", tags=["UI"]) async def root_index(): + """Serve the dashboard with cache-busted asset URLs. + + StaticFiles sends an ETag, but browsers were still serving app.js from cache after + it changed, so a reload showed the old dashboard against the new API -- a panel that + had just been added simply never appeared. Stamping each asset with its mtime means + a changed file is always a different URL. + """ with open(f"{BASE_DIR}/static/index.html", "r") as f: content = f.read() - return HTMLResponse(content=content) + for asset in ("app.js", "styles.css"): + try: + stamp = int(os.path.getmtime(f"{BASE_DIR}/static/{asset}")) + except OSError: + continue + content = content.replace(f"/static/{asset}", f"/static/{asset}?v={stamp}") + return HTMLResponse(content=content, + headers={"Cache-Control": "no-cache, must-revalidate"}) if __name__ == "__main__": import uvicorn diff --git a/static/app.js b/static/app.js index 70c519f..c008843 100644 --- a/static/app.js +++ b/static/app.js @@ -36,6 +36,7 @@ function updateDashboard(data) { // Governor and arbitration state ride along in the shared snapshot. if (data.governor) renderGovernor(data.governor); if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu); + if (data.arbitrator) renderTenants(data.arbitrator, data.gpu); // 1. GPU VRAM Stats const gpu = data.gpu || {}; @@ -1135,3 +1136,56 @@ function watchChartSize() { } document.addEventListener('DOMContentLoaded', () => setTimeout(watchChartSize, 200)); + + +// ---------------------------------------------------------------- tenants + +function renderTenants(arb, gpu) { + const body = document.getElementById('tenants-body'); + if (!body) return; + const state = arb && arb.tenant_state; + if (!state || !state.tenants) return; + + const total = (gpu && gpu.vram_total_gb) || 16; + document.getElementById('tenants-free').textContent = `${state.free_gb} GB free`; + + // Sorted by priority, the order arbitration actually considers them in. + const rows = [...state.tenants].sort((a, b) => b.priority - a.priority); + body.innerHTML = rows.map(t => { + const pct = Math.min((t.vram_gb / total) * 100, 100); + const bar = t.busy ? 'bg-emerald-500' + : t.reclaimable ? 'bg-cyan-600' : 'bg-amber-600'; + const badge = t.busy + ? 'working' + : t.reclaimable + ? 'idle · reclaimable' + : 'cannot be reclaimed'; + return `
+
+ ${t.name} + p${t.priority} + ${t.vram_gb.toFixed(2)} GB · ${badge} +
+
+
+
+
${t.reason || ''}${ + t.needs_vram_gb ? ` · needs ${t.needs_vram_gb} GB to work` : ''}
+
`; + }).join(''); + + // The most recent arbitration decision, including why it could not be satisfied. + const dec = document.getElementById('tenants-decision'); + const a = arb.last_arbitration; + if (!a) { + dec.innerHTML = 'No contention — nothing has needed to be released.'; + return; + } + const when = new Date(a.ts * 1000).toLocaleTimeString(); + const blockers = (a.blockers || []) + .map(b => `${b.name} (${b.vram_gb} GB, ${b.why})`).join(', '); + dec.innerHTML = + `${when} · ` + + `${a.demanding} short by ${a.shortfall_gb} GB — ${a.reason}` + + (blockers ? `
blocked by: ${blockers}
` : ''); +} diff --git a/static/index.html b/static/index.html index 07391c8..4da909c 100644 --- a/static/index.html +++ b/static/index.html @@ -680,6 +680,25 @@
+ + +
+
+
+
+ +
+
+

GPU Tenants

+

Every application contending for the card, from tenants.json

+
+
+ — +
+
+
+
+
diff --git a/vram_arbitrator.py b/vram_arbitrator.py index 9fdd965..fc5be6d 100644 --- a/vram_arbitrator.py +++ b/vram_arbitrator.py @@ -73,7 +73,8 @@ RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2 # depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU # and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and # returns a hard CUDA OOM instead. Both are handled -- the spill by -# AutoArbitrator._check_ollama_starved, the hard failure by the retry below. +# AutoArbitrator._arbitrate (generically, from the tenant registry), the hard failure by +# the retry below. OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate", "failed to allocate", "cuda error") @@ -860,33 +861,41 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m", # idle ComfyUI and try once more. body = resp.text if looks_like_vram_oom(body) and not _retrying: - snap = get_process_vram_bytes() - if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES: + # Which application should give up memory is a question for the registry, + # not something to answer by purging ComfyUI by name. Any reclaimable idle + # tenant below Ollama in priority is a candidate. + state = await arbitrator._tenant_state() + free_gb = arbitrator._last_tenant_state["free_gb"] + size_gb = _model_size_bytes(target_model) / (1024**3) + needed = size_gb * 1.16 if size_gb else free_gb + 1.0 + plan = tenants_mod.plan_release("ollama", state, free_gb, needed) + if plan["release"]: logger.warning( - f"Ollama could not fit '{target_model}' with ComfyUI holding " - f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying") - purge = await instant_free_comfyui_vram() + f"Ollama could not fit '{target_model}' — {plan['reason']}") + freed_before = free_gb + for victim in plan["release"]: + await arbitrator._release_tenant( + victim, f"Ollama could not load '{target_model}'") arbitrator.stats["reclaims_for_ollama"] += 1 arbitrator.last_action = ( - f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so " - f"'{target_model}' could load") + f"Released {', '.join(plan['release'])} so '{target_model}' could load") _record({ "event_type": "VRAM Reclaim for Ollama", - "source": "ComfyUI Pipeline", + "source": ", ".join(plan["release"]), "target": target_model, - "duration_ms": purge.get("duration_ms"), "cache_status": "Reclaimed", "detail": f"Ollama OOM: {body[:160]}", }) await asyncio.sleep(0.3) retry = await switch_ollama_model(target_model, keep_alive, _retrying=True) - retry["reclaimed_from_comfyui_gb"] = round( - snap["comfyui_bytes"] / (1024**3), 2) + retry["released_tenants"] = plan["release"] + retry["would_free_gb"] = plan.get("would_free_gb") retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM" if not retry.get("success"): - # Be specific about why the reclaim was not enough. Blaming ComfyUI - # when a third-party process is holding the memory sends the user - # looking in the wrong place. + # Be specific about why the reclaim was not enough. Blaming a tenant + # when a process nobody can release is holding the memory sends the + # user looking in the wrong place. + retry["blockers"] = plan.get("blockers") retry["unmanaged_blockers"] = describe_unmanaged() return retry return {"success": False, "error": f"HTTP {resp.status_code}: {body}", @@ -942,7 +951,6 @@ class AutoArbitrator: self._yield_backoff_until: Dict[str, float] = {} self._yield_busy_streak: Dict[str, int] = {} self.last_reclaim_time = 0.0 - self.last_starvation_check: Optional[Dict[str, Any]] = None self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0, "bad_status": 0, "error": 0} self._running_id: Optional[str] = None @@ -1301,75 +1309,6 @@ class AutoArbitrator: f"{int(now - since)}s") return - async def _check_ollama_starved(self) -> None: - """The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU. - - Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the - README calling the arbitration bidirectional. When Ollama cannot fit a model it - does not fail, it silently places layers on the CPU and runs about an order of - magnitude slower -- so this is the failure mode a user is least likely to notice - and most likely to feel. - - If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached - checkpoints are the thing to give up. - """ - # Every bail-out is recorded. This check silently did nothing while a model sat - # at 29% on the GPU, and with four separate early returns there was no way to - # tell which one had fired without guessing. - now = time.time() - def bail(reason: str, **extra): - self.last_starvation_check = {"ts": now, "acted": False, - "reason": reason, **extra} - - if self.comfy_was_active: - return bail("ComfyUI is active; it needs the VRAM itself") - if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S: - return bail("within reclaim cooldown", - seconds_left=round(self.RECLAIM_COOLDOWN_S - - (now - self.last_reclaim_time), 1)) - - ollama = await get_ollama_live_state() - if not ollama.get("partially_offloaded"): - return bail("LLM is not spilling to CPU", - gpu_fraction=ollama.get("gpu_fraction"), - model=ollama.get("active_model_name")) - - snap = get_process_vram_bytes() - if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES: - # ComfyUI is not the one holding the memory; nothing we can do here. - return bail("LLM is spilling but ComfyUI holds too little to help", - cpu_offload_pct=ollama.get("cpu_offload_pct"), - comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2)) - - self.last_starvation_check = {"ts": now, "acted": True, - "reason": "reclaiming for the LLM", - "cpu_offload_pct": ollama.get("cpu_offload_pct")} - - self.last_reclaim_time = now - model = ollama.get("active_model_name") - offload = ollama.get("cpu_offload_pct") - logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds " - f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM") - await self._purge_comfy_now(f"LLM spilling {offload}% to CPU") - self.stats["reclaims_for_ollama"] += 1 - - # Freeing VRAM does not move layers back; only a reload re-places the model. Do - # that only when the model is idle, never mid-generation. - after = get_process_vram_bytes() - if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model: - logger.info(f"Reloading '{model}' to place it fully on the GPU...") - await instant_free_ollama_vram(model, confirm=True) - res = await switch_ollama_model(model, keep_alive="30m") - recheck = await get_ollama_live_state() - self.last_action = ( - f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and " - f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU" - if res.get("success") else - f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}") - else: - self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will " - f"stay partly on CPU until its next load") - async def _poll_watchdog(self): """Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz. @@ -1392,7 +1331,6 @@ class AutoArbitrator: await self.trigger_comfy_completed() else: self.watchdog_branches["idle_check"] += 1 - await self._check_ollama_starved() await self._arbitrate() else: self.watchdog_branches["bad_status"] += 1 @@ -1440,7 +1378,6 @@ class AutoArbitrator: "idle_purge_after_s": self.COMFY_IDLE_PURGE_S, "oc_profile": self.oc_profile, "counters": dict(self.stats), - "last_starvation_check": self.last_starvation_check, "comfy_stale_job": self.comfy_stale_job, "last_arbitration": self.last_arbitration, "tenant_state": self._last_tenant_state,