Stop a stale ComfyUI queue entry from disabling half the arbitration
Chasing why the reverse-direction reclaim never fired turned up something worse than the reclaim itself. The starvation check was never running. Instrumenting the watchdog showed busy=6, idle_check=0: every poll took the "ComfyUI is busy" branch. ComfyUI's /queue was reporting a WAN 2.1 i2v job in queue_running while the GPU sat at 0% and ComfyUI held 0.56 GB. The job was dead; ComfyUI had simply never cleared the row. Believing that flag meant this service thought ComfyUI was permanently busy, so it yielded the LLM's VRAM on every poll, never ran the idle purge, and never checked whether the LLM had been squeezed onto the CPU. One stale row disabled half of the arbitration, and it very likely explains the earlier burst of yields against a cron-driven model. A running entry is now corroborated before it is believed. The first attempt used GPU utilisation, which does not work: utilisation is shared with Ollama and with the third-party process on this box, so peak utilisation stayed above any sensible threshold and a stuck entry never looked stale. ComfyUI's own VRAM is the right signal -- a real diffusion job loads gigabytes of checkpoint, a dead one holds only its CUDA context. After the fix the same watchdog reports busy=3, idle_check=32. Every early return in the starvation check now records why it bailed, because with four of them there was no way to tell which had fired. /api/health reports a stale queue entry with its impact and how to clear it. Also confirmed, contradicting an earlier conclusion in this branch: Ollama on this box *does* spill to the CPU. smtek/Qwen3.8-27B:Q2_K_XL held steady at 29.2% on GPU (size=15.59 GB, size_vram=4.56 GB) across twelve seconds of polling -- a stable placement, not a progressive load. Both failure modes are real; which one occurs depends on the model. Tests: 206 (was 199). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -931,6 +931,14 @@ class AutoArbitrator:
|
||||
self._yield_backoff_until: Dict[str, float] = {}
|
||||
self._yield_busy_streak: Dict[str, int] = {}
|
||||
self.last_reclaim_time = 0.0
|
||||
self.last_starvation_check: Optional[Dict[str, Any]] = None
|
||||
self.watchdog_branches = {"busy": 0, "completed": 0, "idle_check": 0,
|
||||
"bad_status": 0, "error": 0}
|
||||
self._running_id: Optional[str] = None
|
||||
self._running_since: Optional[float] = None
|
||||
self._peak_comfy_bytes = 0
|
||||
self.comfy_stale_job: Optional[str] = None
|
||||
self.last_watchdog_error: Optional[str] = None
|
||||
self.stats = {
|
||||
"yields": 0, # release confirmed
|
||||
"yield_deferred_busy": 0, # model mid-generation; unload queued behind it
|
||||
@@ -1122,6 +1130,62 @@ class AutoArbitrator:
|
||||
backoff = min(backoff * 1.5, 15.0)
|
||||
|
||||
RECLAIM_COOLDOWN_S = 30.0
|
||||
# A queue entry that has claimed to be running this long without the GPU ever going
|
||||
# busy is stale, not slow.
|
||||
STALE_RUNNING_S = 90.0
|
||||
# ComfyUI's own VRAM, not GPU utilisation, is what distinguishes a real job from a
|
||||
# stale row. Utilisation is shared: Ollama and any third-party process drive it too,
|
||||
# so peak utilisation stayed above any sensible threshold and a stuck entry never
|
||||
# looked stale. A real diffusion job loads gigabytes of checkpoint; a dead one holds
|
||||
# only the CUDA context.
|
||||
STALE_COMFY_BYTES = 1.5 * 1024 ** 3
|
||||
|
||||
def _comfy_genuinely_busy(self, queue: Dict[str, Any]) -> bool:
|
||||
"""Decide whether ComfyUI is really working, not just claiming to be.
|
||||
|
||||
ComfyUI can leave an entry in queue_running after a job dies -- observed here as
|
||||
a WAN 2.1 i2v entry that sat there with the GPU at 0% and ComfyUI holding 0.56 GB.
|
||||
Trusting that flag alone made this service believe ComfyUI was permanently busy,
|
||||
which meant it evicted the LLM on every poll, never ran the idle purge, and never
|
||||
checked whether the LLM had been squeezed onto the CPU. Half the arbitration was
|
||||
disabled by one stale row.
|
||||
|
||||
A running entry is corroborated against GPU utilisation before it is believed.
|
||||
"""
|
||||
running = queue.get("queue_running") or []
|
||||
pending = queue.get("queue_pending") or []
|
||||
if pending:
|
||||
self._running_since = None
|
||||
self._running_id = None
|
||||
return True
|
||||
if not running:
|
||||
self._running_since = None
|
||||
self._running_id = None
|
||||
self.comfy_stale_job = None
|
||||
return False
|
||||
|
||||
entry = running[0]
|
||||
prompt_id = entry[1] if isinstance(entry, (list, tuple)) and len(entry) > 1 else str(entry)
|
||||
now = time.time()
|
||||
if prompt_id != self._running_id:
|
||||
self._running_id = prompt_id
|
||||
self._running_since = now
|
||||
self._peak_comfy_bytes = 0
|
||||
|
||||
snap = get_process_vram_bytes()
|
||||
self._peak_comfy_bytes = max(self._peak_comfy_bytes, snap.get("comfyui_bytes", 0))
|
||||
|
||||
elapsed = now - (self._running_since or now)
|
||||
if elapsed > self.STALE_RUNNING_S and self._peak_comfy_bytes < self.STALE_COMFY_BYTES:
|
||||
if self.comfy_stale_job != prompt_id:
|
||||
logger.warning(
|
||||
f"ComfyUI reports prompt {prompt_id} running for {int(elapsed)}s while "
|
||||
f"holding only {self._peak_comfy_bytes / (1024**3):.2f} GB — no checkpoint "
|
||||
f"is loaded, so the queue entry is stale. Ignoring it; otherwise ComfyUI "
|
||||
f"looks permanently busy and arbitration stops working.")
|
||||
self.comfy_stale_job = prompt_id
|
||||
return False
|
||||
return True
|
||||
|
||||
async def _check_ollama_starved(self) -> None:
|
||||
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
|
||||
@@ -1135,17 +1199,37 @@ class AutoArbitrator:
|
||||
If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
|
||||
checkpoints are the thing to give up.
|
||||
"""
|
||||
# Every bail-out is recorded. This check silently did nothing while a model sat
|
||||
# at 29% on the GPU, and with four separate early returns there was no way to
|
||||
# tell which one had fired without guessing.
|
||||
now = time.time()
|
||||
if self.comfy_was_active or (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
|
||||
return
|
||||
def bail(reason: str, **extra):
|
||||
self.last_starvation_check = {"ts": now, "acted": False,
|
||||
"reason": reason, **extra}
|
||||
|
||||
if self.comfy_was_active:
|
||||
return bail("ComfyUI is active; it needs the VRAM itself")
|
||||
if (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
|
||||
return bail("within reclaim cooldown",
|
||||
seconds_left=round(self.RECLAIM_COOLDOWN_S
|
||||
- (now - self.last_reclaim_time), 1))
|
||||
|
||||
ollama = await get_ollama_live_state()
|
||||
if not ollama.get("partially_offloaded"):
|
||||
return
|
||||
return bail("LLM is not spilling to CPU",
|
||||
gpu_fraction=ollama.get("gpu_fraction"),
|
||||
model=ollama.get("active_model_name"))
|
||||
|
||||
snap = get_process_vram_bytes()
|
||||
if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
|
||||
return # ComfyUI is not the one holding the memory; nothing we can do here
|
||||
# ComfyUI is not the one holding the memory; nothing we can do here.
|
||||
return bail("LLM is spilling but ComfyUI holds too little to help",
|
||||
cpu_offload_pct=ollama.get("cpu_offload_pct"),
|
||||
comfy_gb=round(snap["comfyui_bytes"] / (1024**3), 2))
|
||||
|
||||
self.last_starvation_check = {"ts": now, "acted": True,
|
||||
"reason": "reclaiming for the LLM",
|
||||
"cpu_offload_pct": ollama.get("cpu_offload_pct")}
|
||||
|
||||
self.last_reclaim_time = now
|
||||
model = ollama.get("active_model_name")
|
||||
@@ -1185,15 +1269,25 @@ class AutoArbitrator:
|
||||
resp = await client.get("/queue")
|
||||
if resp.status_code == 200:
|
||||
q = resp.json()
|
||||
busy = len(q.get("queue_running", [])) > 0 or len(q.get("queue_pending", [])) > 0
|
||||
busy = self._comfy_genuinely_busy(q)
|
||||
if busy:
|
||||
self.watchdog_branches["busy"] += 1
|
||||
await self.trigger_comfy_priority("Watchdog saw an active queue")
|
||||
elif self.comfy_was_active:
|
||||
self.watchdog_branches["completed"] += 1
|
||||
await self.trigger_comfy_completed()
|
||||
else:
|
||||
self.watchdog_branches["idle_check"] += 1
|
||||
await self._check_ollama_starved()
|
||||
except Exception:
|
||||
pass
|
||||
else:
|
||||
self.watchdog_branches["bad_status"] += 1
|
||||
except Exception as e:
|
||||
# This used to swallow everything silently, including anything raised by
|
||||
# the starvation check, which is why that check could appear to run and
|
||||
# do nothing.
|
||||
self.watchdog_branches["error"] += 1
|
||||
self.last_watchdog_error = str(e)[:200]
|
||||
logger.debug(f"watchdog poll error: {e}")
|
||||
await asyncio.sleep(interval)
|
||||
|
||||
def suspend_oc(self, reason: str = "tuning sweep") -> None:
|
||||
@@ -1231,6 +1325,10 @@ class AutoArbitrator:
|
||||
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
|
||||
"oc_profile": self.oc_profile,
|
||||
"counters": dict(self.stats),
|
||||
"last_starvation_check": self.last_starvation_check,
|
||||
"comfy_stale_job": self.comfy_stale_job,
|
||||
"watchdog_branches": dict(self.watchdog_branches),
|
||||
"last_watchdog_error": self.last_watchdog_error,
|
||||
"yield_backoff": {m: round(max(t - time.time(), 0), 1)
|
||||
for m, t in self._yield_backoff_until.items()
|
||||
if t > time.time()},
|
||||
|
||||
Reference in New Issue
Block a user