Remove the duplicated starvation path; show every tenant; cache-bust assets
_check_ollama_starved and _arbitrate were solving the same problem, one of them hardcoded to two applications. The Ollama-specific version is gone and the watchdog calls only the generic loop. The OOM retry inside switch_ollama_model no longer purges ComfyUI by name either: it asks plan_release which tenant should give up memory, so a third application can be the one that yields, and when the reclaim is not enough the response names the blockers instead of implying ComfyUI was at fault. The dashboard showed exactly two engines, which no longer matched what the service does. A GPU Tenants panel lists every configured application ordered by priority -- VRAM held, whether it is working, whether it can be reclaimed at all, and how much it needs -- along with the last arbitration decision and why it could or could not be satisfied. That panel did not appear at first, and the reason is worth fixing rather than working around: the browser kept serving a cached app.js despite the ETag, so a reload ran the old dashboard against the new API. Assets are now stamped with their mtime, so a changed file is always a different URL. Anyone updating this service would have hit the same thing. Also verified along the way that an apparent horizontal-overflow regression was a measurement artifact from a zero-width browser pane, not a real layout fault. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -73,7 +73,8 @@ RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2
|
||||
# depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU
|
||||
# and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and
|
||||
# returns a hard CUDA OOM instead. Both are handled -- the spill by
|
||||
# AutoArbitrator._check_ollama_starved, the hard failure by the retry below.
|
||||
# AutoArbitrator._arbitrate (generically, from the tenant registry), the hard failure by
|
||||
# the retry below.
|
||||
OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
|
||||
"failed to allocate", "cuda error")
|
||||
|
||||
@@ -860,33 +861,41 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
|
||||
# idle ComfyUI and try once more.
|
||||
body = resp.text
|
||||
if looks_like_vram_oom(body) and not _retrying:
|
||||
snap = get_process_vram_bytes()
|
||||
if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES:
|
||||
# Which application should give up memory is a question for the registry,
|
||||
# not something to answer by purging ComfyUI by name. Any reclaimable idle
|
||||
# tenant below Ollama in priority is a candidate.
|
||||
state = await arbitrator._tenant_state()
|
||||
free_gb = arbitrator._last_tenant_state["free_gb"]
|
||||
size_gb = _model_size_bytes(target_model) / (1024**3)
|
||||
needed = size_gb * 1.16 if size_gb else free_gb + 1.0
|
||||
plan = tenants_mod.plan_release("ollama", state, free_gb, needed)
|
||||
if plan["release"]:
|
||||
logger.warning(
|
||||
f"Ollama could not fit '{target_model}' with ComfyUI holding "
|
||||
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying")
|
||||
purge = await instant_free_comfyui_vram()
|
||||
f"Ollama could not fit '{target_model}' — {plan['reason']}")
|
||||
freed_before = free_gb
|
||||
for victim in plan["release"]:
|
||||
await arbitrator._release_tenant(
|
||||
victim, f"Ollama could not load '{target_model}'")
|
||||
arbitrator.stats["reclaims_for_ollama"] += 1
|
||||
arbitrator.last_action = (
|
||||
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so "
|
||||
f"'{target_model}' could load")
|
||||
f"Released {', '.join(plan['release'])} so '{target_model}' could load")
|
||||
_record({
|
||||
"event_type": "VRAM Reclaim for Ollama",
|
||||
"source": "ComfyUI Pipeline",
|
||||
"source": ", ".join(plan["release"]),
|
||||
"target": target_model,
|
||||
"duration_ms": purge.get("duration_ms"),
|
||||
"cache_status": "Reclaimed",
|
||||
"detail": f"Ollama OOM: {body[:160]}",
|
||||
})
|
||||
await asyncio.sleep(0.3)
|
||||
retry = await switch_ollama_model(target_model, keep_alive, _retrying=True)
|
||||
retry["reclaimed_from_comfyui_gb"] = round(
|
||||
snap["comfyui_bytes"] / (1024**3), 2)
|
||||
retry["released_tenants"] = plan["release"]
|
||||
retry["would_free_gb"] = plan.get("would_free_gb")
|
||||
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
|
||||
if not retry.get("success"):
|
||||
# Be specific about why the reclaim was not enough. Blaming ComfyUI
|
||||
# when a third-party process is holding the memory sends the user
|
||||
# looking in the wrong place.
|
||||
# Be specific about why the reclaim was not enough. Blaming a tenant
|
||||
# when a process nobody can release is holding the memory sends the
|
||||
# user looking in the wrong place.
|
||||
retry["blockers"] = plan.get("blockers")
|
||||
retry["unmanaged_blockers"] = describe_unmanaged()
|
||||
return retry
|
||||
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
|
||||
@@ -942,7 +951,6 @@ class AutoArbitrator:
|
||||
self._yield_backoff_until: Dict[str, float] = {}
|
||||
self._yield_busy_streak: Dict[str, int] = {}
|
||||
self.last_reclaim_time = 0.0
|
||||
self.last_starvation_check: Optional[Dict[str, Any]] = None
|
||||
self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0,
|
||||
"bad_status": 0, "error": 0}
|
||||
self._running_id: Optional[str] = None
|
||||
@@ -1301,75 +1309,6 @@ class AutoArbitrator:
|
||||
f"{int(now - since)}s")
|
||||
return
|
||||
|
||||
async def _check_ollama_starved(self) -> None:
|
||||
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
|
||||
|
||||
Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the
|
||||
README calling the arbitration bidirectional. When Ollama cannot fit a model it
|
||||
does not fail, it silently places layers on the CPU and runs about an order of
|
||||
magnitude slower -- so this is the failure mode a user is least likely to notice
|
||||
and most likely to feel.
|
||||
|
||||
If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
|
||||
checkpoints are the thing to give up.
|
||||
"""
|
||||
# Every bail-out is recorded. This check silently did nothing while a model sat
|
||||
# at 29% on the GPU, and with four separate early returns there was no way to
|
||||
# tell which one had fired without guessing.
|
||||
now = time.time()
|
||||
def bail(reason: str, **extra):
|
||||
self.last_starvation_check = {"ts": now, "acted": False,
|
||||
"reason": reason, **extra}
|
||||
|
||||
if self.comfy_was_active:
|
||||
return bail("ComfyUI is active; it needs the VRAM itself")
|
||||
if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
|
||||
return bail("within reclaim cooldown",
|
||||
seconds_left=round(self.RECLAIM_COOLDOWN_S
|
||||
- (now - self.last_reclaim_time), 1))
|
||||
|
||||
ollama = await get_ollama_live_state()
|
||||
if not ollama.get("partially_offloaded"):
|
||||
return bail("LLM is not spilling to CPU",
|
||||
gpu_fraction=ollama.get("gpu_fraction"),
|
||||
model=ollama.get("active_model_name"))
|
||||
|
||||
snap = get_process_vram_bytes()
|
||||
if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
|
||||
# ComfyUI is not the one holding the memory; nothing we can do here.
|
||||
return bail("LLM is spilling but ComfyUI holds too little to help",
|
||||
cpu_offload_pct=ollama.get("cpu_offload_pct"),
|
||||
comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2))
|
||||
|
||||
self.last_starvation_check = {"ts": now, "acted": True,
|
||||
"reason": "reclaiming for the LLM",
|
||||
"cpu_offload_pct": ollama.get("cpu_offload_pct")}
|
||||
|
||||
self.last_reclaim_time = now
|
||||
model = ollama.get("active_model_name")
|
||||
offload = ollama.get("cpu_offload_pct")
|
||||
logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds "
|
||||
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM")
|
||||
await self._purge_comfy_now(f"LLM spilling {offload}% to CPU")
|
||||
self.stats["reclaims_for_ollama"] += 1
|
||||
|
||||
# Freeing VRAM does not move layers back; only a reload re-places the model. Do
|
||||
# that only when the model is idle, never mid-generation.
|
||||
after = get_process_vram_bytes()
|
||||
if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model:
|
||||
logger.info(f"Reloading '{model}' to place it fully on the GPU...")
|
||||
await instant_free_ollama_vram(model, confirm=True)
|
||||
res = await switch_ollama_model(model, keep_alive="30m")
|
||||
recheck = await get_ollama_live_state()
|
||||
self.last_action = (
|
||||
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and "
|
||||
f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU"
|
||||
if res.get("success") else
|
||||
f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}")
|
||||
else:
|
||||
self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will "
|
||||
f"stay partly on CPU until its next load")
|
||||
|
||||
async def _poll_watchdog(self):
|
||||
"""Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz.
|
||||
|
||||
@@ -1392,7 +1331,6 @@ class AutoArbitrator:
|
||||
await self.trigger_comfy_completed()
|
||||
else:
|
||||
self.watchdog_branches["idle_check"] += 1
|
||||
await self._check_ollama_starved()
|
||||
await self._arbitrate()
|
||||
else:
|
||||
self.watchdog_branches["bad_status"] += 1
|
||||
@@ -1440,7 +1378,6 @@ class AutoArbitrator:
|
||||
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
|
||||
"oc_profile": self.oc_profile,
|
||||
"counters": dict(self.stats),
|
||||
"last_starvation_check": self.last_starvation_check,
|
||||
"comfy_stale_job": self.comfy_stale_job,
|
||||
"last_arbitration": self.last_arbitration,
|
||||
"tenant_state": self._last_tenant_state,
|
||||
|
||||
Reference in New Issue
Block a user