diff --git a/vram_arbitrator.py b/vram_arbitrator.py
index 9fdd965..fc5be6d 100644
--- a/vram_arbitrator.py
+++ b/vram_arbitrator.py
@@ -73,7 +73,8 @@ RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2
# depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU
# and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and
# returns a hard CUDA OOM instead. Both are handled -- the spill by
-# AutoArbitrator._check_ollama_starved, the hard failure by the retry below.
+# AutoArbitrator._arbitrate (generically, from the tenant registry), the hard failure by
+# the retry below.
OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
"failed to allocate", "cuda error")
@@ -860,33 +861,41 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
# idle ComfyUI and try once more.
body = resp.text
if looks_like_vram_oom(body) and not _retrying:
- snap = get_process_vram_bytes()
- if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES:
+ # Which application should give up memory is a question for the registry,
+ # not something to answer by purging ComfyUI by name. Any reclaimable idle
+ # tenant below Ollama in priority is a candidate.
+ state = await arbitrator._tenant_state()
+ free_gb = arbitrator._last_tenant_state["free_gb"]
+ size_gb = _model_size_bytes(target_model) / (1024**3)
+ needed = size_gb * 1.16 if size_gb else free_gb + 1.0
+ plan = tenants_mod.plan_release("ollama", state, free_gb, needed)
+ if plan["release"]:
logger.warning(
- f"Ollama could not fit '{target_model}' with ComfyUI holding "
- f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying")
- purge = await instant_free_comfyui_vram()
+ f"Ollama could not fit '{target_model}' — {plan['reason']}")
+ freed_before = free_gb
+ for victim in plan["release"]:
+ await arbitrator._release_tenant(
+ victim, f"Ollama could not load '{target_model}'")
arbitrator.stats["reclaims_for_ollama"] += 1
arbitrator.last_action = (
- f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so "
- f"'{target_model}' could load")
+ f"Released {', '.join(plan['release'])} so '{target_model}' could load")
_record({
"event_type": "VRAM Reclaim for Ollama",
- "source": "ComfyUI Pipeline",
+ "source": ", ".join(plan["release"]),
"target": target_model,
- "duration_ms": purge.get("duration_ms"),
"cache_status": "Reclaimed",
"detail": f"Ollama OOM: {body[:160]}",
})
await asyncio.sleep(0.3)
retry = await switch_ollama_model(target_model, keep_alive, _retrying=True)
- retry["reclaimed_from_comfyui_gb"] = round(
- snap["comfyui_bytes"] / (1024**3), 2)
+ retry["released_tenants"] = plan["release"]
+ retry["would_free_gb"] = plan.get("would_free_gb")
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
if not retry.get("success"):
- # Be specific about why the reclaim was not enough. Blaming ComfyUI
- # when a third-party process is holding the memory sends the user
- # looking in the wrong place.
+ # Be specific about why the reclaim was not enough. Blaming a tenant
+ # when a process nobody can release is holding the memory sends the
+ # user looking in the wrong place.
+ retry["blockers"] = plan.get("blockers")
retry["unmanaged_blockers"] = describe_unmanaged()
return retry
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
@@ -942,7 +951,6 @@ class AutoArbitrator:
self._yield_backoff_until: Dict[str, float] = {}
self._yield_busy_streak: Dict[str, int] = {}
self.last_reclaim_time = 0.0
- self.last_starvation_check: Optional[Dict[str, Any]] = None
self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0,
"bad_status": 0, "error": 0}
self._running_id: Optional[str] = None
@@ -1301,75 +1309,6 @@ class AutoArbitrator:
f"{int(now - since)}s")
return
- async def _check_ollama_starved(self) -> None:
- """The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
-
- Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the
- README calling the arbitration bidirectional. When Ollama cannot fit a model it
- does not fail, it silently places layers on the CPU and runs about an order of
- magnitude slower -- so this is the failure mode a user is least likely to notice
- and most likely to feel.
-
- If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
- checkpoints are the thing to give up.
- """
- # Every bail-out is recorded. This check silently did nothing while a model sat
- # at 29% on the GPU, and with four separate early returns there was no way to
- # tell which one had fired without guessing.
- now = time.time()
- def bail(reason: str, **extra):
- self.last_starvation_check = {"ts": now, "acted": False,
- "reason": reason, **extra}
-
- if self.comfy_was_active:
- return bail("ComfyUI is active; it needs the VRAM itself")
- if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
- return bail("within reclaim cooldown",
- seconds_left=round(self.RECLAIM_COOLDOWN_S
- - (now - self.last_reclaim_time), 1))
-
- ollama = await get_ollama_live_state()
- if not ollama.get("partially_offloaded"):
- return bail("LLM is not spilling to CPU",
- gpu_fraction=ollama.get("gpu_fraction"),
- model=ollama.get("active_model_name"))
-
- snap = get_process_vram_bytes()
- if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
- # ComfyUI is not the one holding the memory; nothing we can do here.
- return bail("LLM is spilling but ComfyUI holds too little to help",
- cpu_offload_pct=ollama.get("cpu_offload_pct"),
- comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2))
-
- self.last_starvation_check = {"ts": now, "acted": True,
- "reason": "reclaiming for the LLM",
- "cpu_offload_pct": ollama.get("cpu_offload_pct")}
-
- self.last_reclaim_time = now
- model = ollama.get("active_model_name")
- offload = ollama.get("cpu_offload_pct")
- logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds "
- f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM")
- await self._purge_comfy_now(f"LLM spilling {offload}% to CPU")
- self.stats["reclaims_for_ollama"] += 1
-
- # Freeing VRAM does not move layers back; only a reload re-places the model. Do
- # that only when the model is idle, never mid-generation.
- after = get_process_vram_bytes()
- if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model:
- logger.info(f"Reloading '{model}' to place it fully on the GPU...")
- await instant_free_ollama_vram(model, confirm=True)
- res = await switch_ollama_model(model, keep_alive="30m")
- recheck = await get_ollama_live_state()
- self.last_action = (
- f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and "
- f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU"
- if res.get("success") else
- f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}")
- else:
- self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will "
- f"stay partly on CPU until its next load")
-
async def _poll_watchdog(self):
"""Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz.
@@ -1392,7 +1331,6 @@ class AutoArbitrator:
await self.trigger_comfy_completed()
else:
self.watchdog_branches["idle_check"] += 1
- await self._check_ollama_starved()
await self._arbitrate()
else:
self.watchdog_branches["bad_status"] += 1
@@ -1440,7 +1378,6 @@ class AutoArbitrator:
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
"oc_profile": self.oc_profile,
"counters": dict(self.stats),
- "last_starvation_check": self.last_starvation_check,
"comfy_stale_job": self.comfy_stale_job,
"last_arbitration": self.last_arbitration,
"tenant_state": self._last_tenant_state,