Remove the duplicated starvation path; show every tenant; cache-bust assets

_check_ollama_starved and _arbitrate were solving the same problem, one of them
hardcoded to two applications. The Ollama-specific version is gone and the watchdog
calls only the generic loop. The OOM retry inside switch_ollama_model no longer
purges ComfyUI by name either: it asks plan_release which tenant should give up
memory, so a third application can be the one that yields, and when the reclaim is
not enough the response names the blockers instead of implying ComfyUI was at fault.

The dashboard showed exactly two engines, which no longer matched what the service
does. A GPU Tenants panel lists every configured application ordered by priority --
VRAM held, whether it is working, whether it can be reclaimed at all, and how much it
needs -- along with the last arbitration decision and why it could or could not be
satisfied.

That panel did not appear at first, and the reason is worth fixing rather than
working around: the browser kept serving a cached app.js despite the ETag, so a
reload ran the old dashboard against the new API. Assets are now stamped with their
mtime, so a changed file is always a different URL. Anyone updating this service would
have hit the same thing.

Also verified along the way that an apparent horizontal-overflow regression was a
measurement artifact from a zero-width browser pane, not a real layout fault.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-07 15:16:10 -07:00
parent 2b00ab4e12
commit c6455d7c6e
5 changed files with 115 additions and 89 deletions

View File

@@ -215,7 +215,9 @@ acted on, which makes the decision testable and loggable.
Ollama, ComfyUI and the desktop compositor ship as defaults, so behaviour is unchanged — Ollama, ComfyUI and the desktop compositor ship as defaults, so behaviour is unchanged —
but nothing in the arbitration logic knows their names, and three applications can but nothing in the arbitration logic knows their names, and three applications can
contend for the card as easily as two. Endpoints are generic: contend for the card as easily as two. The dashboard's **GPU Tenants** panel lists all of
them ordered by the priority arbitration actually considers, with the last decision and
why it could or could not be satisfied. Endpoints are generic:
`GET /api/tenants`, `GET /api/tenants/{name}`, `POST /api/tenants/{name}/release`. `GET /api/tenants`, `GET /api/tenants/{name}`, `POST /api/tenants/{name}/release`.
A tenant with `"release": {"type": "none"}` is still worth declaring. The 842 MB speech A tenant with `"release": {"type": "none"}` is still worth declaring. The 842 MB speech

View File

@@ -695,9 +695,23 @@ app.mount("/static", StaticFiles(directory=f"{BASE_DIR}/static"), name="static")
@app.get("/", summary="Dashboard Web UI", tags=["UI"]) @app.get("/", summary="Dashboard Web UI", tags=["UI"])
async def root_index(): async def root_index():
"""Serve the dashboard with cache-busted asset URLs.
StaticFiles sends an ETag, but browsers were still serving app.js from cache after
it changed, so a reload showed the old dashboard against the new API -- a panel that
had just been added simply never appeared. Stamping each asset with its mtime means
a changed file is always a different URL.
"""
with open(f"{BASE_DIR}/static/index.html", "r") as f: with open(f"{BASE_DIR}/static/index.html", "r") as f:
content = f.read() content = f.read()
return HTMLResponse(content=content) for asset in ("app.js", "styles.css"):
try:
stamp = int(os.path.getmtime(f"{BASE_DIR}/static/{asset}"))
except OSError:
continue
content = content.replace(f"/static/{asset}", f"/static/{asset}?v={stamp}")
return HTMLResponse(content=content,
headers={"Cache-Control": "no-cache, must-revalidate"})
if __name__ == "__main__": if __name__ == "__main__":
import uvicorn import uvicorn

View File

@@ -36,6 +36,7 @@ function updateDashboard(data) {
// Governor and arbitration state ride along in the shared snapshot. // Governor and arbitration state ride along in the shared snapshot.
if (data.governor) renderGovernor(data.governor); if (data.governor) renderGovernor(data.governor);
if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu); if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu);
if (data.arbitrator) renderTenants(data.arbitrator, data.gpu);
// 1. GPU VRAM Stats // 1. GPU VRAM Stats
const gpu = data.gpu || {}; const gpu = data.gpu || {};
@@ -1135,3 +1136,56 @@ function watchChartSize() {
} }
document.addEventListener('DOMContentLoaded', () => setTimeout(watchChartSize, 200)); document.addEventListener('DOMContentLoaded', () => setTimeout(watchChartSize, 200));
// ---------------------------------------------------------------- tenants
function renderTenants(arb, gpu) {
const body = document.getElementById('tenants-body');
if (!body) return;
const state = arb && arb.tenant_state;
if (!state || !state.tenants) return;
const total = (gpu && gpu.vram_total_gb) || 16;
document.getElementById('tenants-free').textContent = `${state.free_gb} GB free`;
// Sorted by priority, the order arbitration actually considers them in.
const rows = [...state.tenants].sort((a, b) => b.priority - a.priority);
body.innerHTML = rows.map(t => {
const pct = Math.min((t.vram_gb / total) * 100, 100);
const bar = t.busy ? 'bg-emerald-500'
: t.reclaimable ? 'bg-cyan-600' : 'bg-amber-600';
const badge = t.busy
? '<span class="text-emerald-400">working</span>'
: t.reclaimable
? '<span class="text-slate-500">idle · reclaimable</span>'
: '<span class="text-amber-400">cannot be reclaimed</span>';
return `<div>
<div class="flex justify-between text-[11px] font-mono">
<span class="text-slate-200">${t.name}
<span class="text-slate-600">p${t.priority}</span></span>
<span class="text-slate-400">${t.vram_gb.toFixed(2)} GB · ${badge}</span>
</div>
<div class="w-full bg-slate-950 rounded-full h-1.5 mt-1 overflow-hidden border border-slate-800/60">
<div class="${bar} h-full transition-all duration-500" style="width:${pct}%"></div>
</div>
<div class="text-[10px] text-slate-600 mt-0.5">${t.reason || ''}${
t.needs_vram_gb ? ` · needs ${t.needs_vram_gb} GB to work` : ''}</div>
</div>`;
}).join('');
// The most recent arbitration decision, including why it could not be satisfied.
const dec = document.getElementById('tenants-decision');
const a = arb.last_arbitration;
if (!a) {
dec.innerHTML = '<span class="text-slate-600">No contention — nothing has needed to be released.</span>';
return;
}
const when = new Date(a.ts * 1000).toLocaleTimeString();
const blockers = (a.blockers || [])
.map(b => `${b.name} (${b.vram_gb} GB, ${b.why})`).join(', ');
dec.innerHTML =
`<span class="${a.possible ? 'text-cyan-400' : 'text-amber-400'}">${when} · ` +
`${a.demanding} short by ${a.shortfall_gb} GB</span> — ${a.reason}` +
(blockers ? `<div class="text-slate-600">blocked by: ${blockers}</div>` : '');
}

View File

@@ -680,6 +680,25 @@
<div class="grid grid-cols-1 xl:grid-cols-2 3xl:grid-cols-3 gap-5 mt-5"> <div class="grid grid-cols-1 xl:grid-cols-2 3xl:grid-cols-3 gap-5 mt-5">
<!-- All GPU tenants, however many are configured -->
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
<div class="flex items-center space-x-2">
<div class="p-2 rounded-lg bg-teal-950/80 border border-teal-800 text-teal-400">
<i class="fa-solid fa-layer-group text-sm"></i>
</div>
<div>
<h3 class="font-bold text-slate-100 text-sm">GPU Tenants</h3>
<p class="text-xs text-slate-400">Every application contending for the card, from <code class="text-teal-400">tenants.json</code></p>
</div>
</div>
<span id="tenants-free" class="text-xs font-mono text-slate-500">—</span>
</div>
<div id="tenants-body" class="mt-4 space-y-2"></div>
<div id="tenants-decision" class="mt-3 text-[11px] font-mono text-slate-400"></div>
</div>
<!-- System health: makes a silently-broken dependency loud --> <!-- System health: makes a silently-broken dependency loud -->
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2"> <div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
<div class="flex items-center justify-between pb-3 border-b border-slate-800"> <div class="flex items-center justify-between pb-3 border-b border-slate-800">

View File

@@ -73,7 +73,8 @@ RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2
# depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU # depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU
# and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and # and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and
# returns a hard CUDA OOM instead. Both are handled -- the spill by # returns a hard CUDA OOM instead. Both are handled -- the spill by
# AutoArbitrator._check_ollama_starved, the hard failure by the retry below. # AutoArbitrator._arbitrate (generically, from the tenant registry), the hard failure by
# the retry below.
OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate", OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
"failed to allocate", "cuda error") "failed to allocate", "cuda error")
@@ -860,33 +861,41 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
# idle ComfyUI and try once more. # idle ComfyUI and try once more.
body = resp.text body = resp.text
if looks_like_vram_oom(body) and not _retrying: if looks_like_vram_oom(body) and not _retrying:
snap = get_process_vram_bytes() # Which application should give up memory is a question for the registry,
if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES: # not something to answer by purging ComfyUI by name. Any reclaimable idle
# tenant below Ollama in priority is a candidate.
state = await arbitrator._tenant_state()
free_gb = arbitrator._last_tenant_state["free_gb"]
size_gb = _model_size_bytes(target_model) / (1024**3)
needed = size_gb * 1.16 if size_gb else free_gb + 1.0
plan = tenants_mod.plan_release("ollama", state, free_gb, needed)
if plan["release"]:
logger.warning( logger.warning(
f"Ollama could not fit '{target_model}' with ComfyUI holding " f"Ollama could not fit '{target_model}' — {plan['reason']}")
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying") freed_before = free_gb
purge = await instant_free_comfyui_vram() for victim in plan["release"]:
await arbitrator._release_tenant(
victim, f"Ollama could not load '{target_model}'")
arbitrator.stats["reclaims_for_ollama"] += 1 arbitrator.stats["reclaims_for_ollama"] += 1
arbitrator.last_action = ( arbitrator.last_action = (
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so " f"Released {', '.join(plan['release'])} so '{target_model}' could load")
f"'{target_model}' could load")
_record({ _record({
"event_type": "VRAM Reclaim for Ollama", "event_type": "VRAM Reclaim for Ollama",
"source": "ComfyUI Pipeline", "source": ", ".join(plan["release"]),
"target": target_model, "target": target_model,
"duration_ms": purge.get("duration_ms"),
"cache_status": "Reclaimed", "cache_status": "Reclaimed",
"detail": f"Ollama OOM: {body[:160]}", "detail": f"Ollama OOM: {body[:160]}",
}) })
await asyncio.sleep(0.3) await asyncio.sleep(0.3)
retry = await switch_ollama_model(target_model, keep_alive, _retrying=True) retry = await switch_ollama_model(target_model, keep_alive, _retrying=True)
retry["reclaimed_from_comfyui_gb"] = round( retry["released_tenants"] = plan["release"]
snap["comfyui_bytes"] / (1024**3), 2) retry["would_free_gb"] = plan.get("would_free_gb")
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM" retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
if not retry.get("success"): if not retry.get("success"):
# Be specific about why the reclaim was not enough. Blaming ComfyUI # Be specific about why the reclaim was not enough. Blaming a tenant
# when a third-party process is holding the memory sends the user # when a process nobody can release is holding the memory sends the
# looking in the wrong place. # user looking in the wrong place.
retry["blockers"] = plan.get("blockers")
retry["unmanaged_blockers"] = describe_unmanaged() retry["unmanaged_blockers"] = describe_unmanaged()
return retry return retry
return {"success": False, "error": f"HTTP {resp.status_code}: {body}", return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
@@ -942,7 +951,6 @@ class AutoArbitrator:
self._yield_backoff_until: Dict[str, float] = {} self._yield_backoff_until: Dict[str, float] = {}
self._yield_busy_streak: Dict[str, int] = {} self._yield_busy_streak: Dict[str, int] = {}
self.last_reclaim_time = 0.0 self.last_reclaim_time = 0.0
self.last_starvation_check: Optional[Dict[str, Any]] = None
self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0, self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0,
"bad_status": 0, "error": 0} "bad_status": 0, "error": 0}
self._running_id: Optional[str] = None self._running_id: Optional[str] = None
@@ -1301,75 +1309,6 @@ class AutoArbitrator:
f"{int(now - since)}s") f"{int(now - since)}s")
return return
async def _check_ollama_starved(self) -> None:
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the
README calling the arbitration bidirectional. When Ollama cannot fit a model it
does not fail, it silently places layers on the CPU and runs about an order of
magnitude slower -- so this is the failure mode a user is least likely to notice
and most likely to feel.
If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
checkpoints are the thing to give up.
"""
# Every bail-out is recorded. This check silently did nothing while a model sat
# at 29% on the GPU, and with four separate early returns there was no way to
# tell which one had fired without guessing.
now = time.time()
def bail(reason: str, **extra):
self.last_starvation_check = {"ts": now, "acted": False,
"reason": reason, **extra}
if self.comfy_was_active:
return bail("ComfyUI is active; it needs the VRAM itself")
if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
return bail("within reclaim cooldown",
seconds_left=round(self.RECLAIM_COOLDOWN_S
- (now - self.last_reclaim_time), 1))
ollama = await get_ollama_live_state()
if not ollama.get("partially_offloaded"):
return bail("LLM is not spilling to CPU",
gpu_fraction=ollama.get("gpu_fraction"),
model=ollama.get("active_model_name"))
snap = get_process_vram_bytes()
if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
# ComfyUI is not the one holding the memory; nothing we can do here.
return bail("LLM is spilling but ComfyUI holds too little to help",
cpu_offload_pct=ollama.get("cpu_offload_pct"),
comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2))
self.last_starvation_check = {"ts": now, "acted": True,
"reason": "reclaiming for the LLM",
"cpu_offload_pct": ollama.get("cpu_offload_pct")}
self.last_reclaim_time = now
model = ollama.get("active_model_name")
offload = ollama.get("cpu_offload_pct")
logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds "
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM")
await self._purge_comfy_now(f"LLM spilling {offload}% to CPU")
self.stats["reclaims_for_ollama"] += 1
# Freeing VRAM does not move layers back; only a reload re-places the model. Do
# that only when the model is idle, never mid-generation.
after = get_process_vram_bytes()
if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model:
logger.info(f"Reloading '{model}' to place it fully on the GPU...")
await instant_free_ollama_vram(model, confirm=True)
res = await switch_ollama_model(model, keep_alive="30m")
recheck = await get_ollama_live_state()
self.last_action = (
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and "
f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU"
if res.get("success") else
f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}")
else:
self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will "
f"stay partly on CPU until its next load")
async def _poll_watchdog(self): async def _poll_watchdog(self):
"""Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz. """Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz.
@@ -1392,7 +1331,6 @@ class AutoArbitrator:
await self.trigger_comfy_completed() await self.trigger_comfy_completed()
else: else:
self.watchdog_branches["idle_check"] += 1 self.watchdog_branches["idle_check"] += 1
await self._check_ollama_starved()
await self._arbitrate() await self._arbitrate()
else: else:
self.watchdog_branches["bad_status"] += 1 self.watchdog_branches["bad_status"] += 1
@@ -1440,7 +1378,6 @@ class AutoArbitrator:
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S, "idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
"oc_profile": self.oc_profile, "oc_profile": self.oc_profile,
"counters": dict(self.stats), "counters": dict(self.stats),
"last_starvation_check": self.last_starvation_check,
"comfy_stale_job": self.comfy_stale_job, "comfy_stale_job": self.comfy_stale_job,
"last_arbitration": self.last_arbitration, "last_arbitration": self.last_arbitration,
"tenant_state": self._last_tenant_state, "tenant_state": self._last_tenant_state,