Add a durable cross-tenant job queue with VRAM-aware scheduling
The service only reacted: it noticed an application had started and scrambled to free memory. Nothing could be lined up. Each application has its own queue but they cannot see each other, so work submitted to one had no way to wait for the other. Jobs are stored in SQLite, so the queue is bounded by disk rather than memory and survives a restart. The scheduler takes the highest-priority pending job, arbitrates VRAM for it through the same plan_release, runs it, and moves on -- one at a time, because overlapping jobs would recreate the contention this service exists to resolve. Four bugs found by running it rather than reasoning about it: Dispatching without checking for room destroyed three queued LLM jobs in a row: a CUDA OOM kills llama-server outright, it does not fail gracefully. A job that cannot run yet now waits. The room check used the tenant's needs_vram_gb, which cannot be right for an LLM -- the requirement is a property of the model being loaded. A flat 4 GB passed with 8 GB free and then a 14.9 GB model was dispatched into it. The requirement is now computed per job. Waiting forever is also wrong. Three jobs sat pending indefinitely needing 14.93 GB on a card where at most ~14.8 GB can ever be free, because an unreclaimable process holds 0.82 GB. A job that cannot be satisfied now fails with the ceiling and the blockers named. plan_release assumed releasing a tenant frees everything it holds. ComfyUI keeps its CUDA context for as long as the process lives, so it reported that releasing ComfyUI would free 0.37 GB against a 0.33 GB shortfall; the job was cleared and the memory never arrived. Tenants declare vram_floor_gb and only memory above it counts. An exception during dispatch left the row RUNNING forever while the scheduler moved on. Failures now land on the job, and jobs left running by a previous process are requeued at startup. Verified end to end: five mixed jobs across both applications, queued at once, all completed with no failures. Tests: 250. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -151,7 +151,8 @@ def get_process_vram_bytes() -> Dict[str, int]:
|
||||
to actually drain.
|
||||
"""
|
||||
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0,
|
||||
"desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0}
|
||||
"desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0,
|
||||
"by_tenant_bytes": {}}
|
||||
if not NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
@@ -178,6 +179,7 @@ def get_process_vram_bytes() -> Dict[str, int]:
|
||||
if len(_PID_KIND_CACHE) >= _PID_KIND_CACHE_MAX:
|
||||
_PID_KIND_CACHE.clear()
|
||||
_PID_KIND_CACHE[key] = kind
|
||||
out["by_tenant_bytes"][kind] = out["by_tenant_bytes"].get(kind, 0) + used
|
||||
if kind == "ollama":
|
||||
out["ollama_bytes"] += used
|
||||
elif kind == "comfy":
|
||||
@@ -963,6 +965,7 @@ class AutoArbitrator:
|
||||
self.event_sources: Dict[str, str] = {}
|
||||
self._last_tenant_state: Optional[Dict[str, Any]] = None
|
||||
self.last_arbitration: Optional[Dict[str, Any]] = None
|
||||
self.last_handoff: Optional[Dict[str, Any]] = None
|
||||
self.last_watchdog_error: Optional[str] = None
|
||||
self.stats = {
|
||||
"yields": 0, # release confirmed
|
||||
@@ -1133,12 +1136,16 @@ class AutoArbitrator:
|
||||
# Coalesce bursts: a single graph emits many messages, and one
|
||||
# arbitration pass per burst is enough.
|
||||
now = time.time()
|
||||
if now - self._last_event_wake < 0.25:
|
||||
if now - self._last_event_wake < 0.05:
|
||||
continue
|
||||
self._last_event_wake = now
|
||||
self.stats["event_wakeups"] = self.stats.get("event_wakeups", 0) + 1
|
||||
try:
|
||||
await self._arbitrate()
|
||||
# A message on this tenant's own stream is live proof it is
|
||||
# working right now, so it is taken as busy rather than asked
|
||||
# over HTTP. A stale queue row could lie; an event arriving
|
||||
# this instant cannot.
|
||||
await self._arbitrate(active_tenant=tenant_name)
|
||||
except Exception as e:
|
||||
logger.debug(f"arbitration from event failed: {e}")
|
||||
except (websockets.exceptions.ConnectionClosed, OSError, asyncio.CancelledError):
|
||||
@@ -1260,18 +1267,29 @@ class AutoArbitrator:
|
||||
return False
|
||||
return True
|
||||
|
||||
async def _tenant_state(self) -> List[Dict[str, Any]]:
|
||||
"""Current VRAM and busy state for every configured tenant."""
|
||||
async def _tenant_state(self, active_tenant: Optional[str] = None
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Current VRAM and busy state for every configured tenant.
|
||||
|
||||
`active_tenant` skips the HTTP busy probe for the tenant whose event stream just
|
||||
fired: the event is the evidence. That removes a round trip from the handoff,
|
||||
which is the one path where latency is the entire point.
|
||||
"""
|
||||
# Only the cheap NVML read. The full hardware snapshot also does a psutil lookup
|
||||
# per process, which is wasted work on the handoff path where latency is the
|
||||
# entire point.
|
||||
snap = get_process_vram_bytes()
|
||||
stats = get_gpu_hardware_stats()
|
||||
by_tenant = (stats.get("breakdown", {}) or {}).get("by_tenant_gb", {})
|
||||
by_tenant = {k: b / (1024 ** 3) for k, b in snap["by_tenant_bytes"].items()}
|
||||
out = []
|
||||
for t in tenants_mod.load_tenants():
|
||||
if not t.enabled:
|
||||
continue
|
||||
bucket = _BUCKET_ALIASES.get(t.name, t.name)
|
||||
vram_gb = by_tenant.get(bucket, 0.0)
|
||||
probe = await tenants_mod.probe_busy(t, vram_gb=vram_gb)
|
||||
if t.name == active_tenant:
|
||||
probe = {"busy": True, "reason": "event received from its own stream"}
|
||||
else:
|
||||
probe = await tenants_mod.probe_busy(t, vram_gb=vram_gb)
|
||||
out.append({
|
||||
"name": t.name,
|
||||
"priority": t.priority,
|
||||
@@ -1281,6 +1299,7 @@ class AutoArbitrator:
|
||||
"reclaimable": t.reclaimable,
|
||||
"needs_vram_gb": t.needs_vram_gb,
|
||||
"overclock_profile": t.overclock_profile,
|
||||
"vram_floor_gb": t.vram_floor_gb,
|
||||
"idle_release_after_s": t.idle_release_after_s,
|
||||
"reason": probe.get("reason"),
|
||||
})
|
||||
@@ -1319,7 +1338,7 @@ class AutoArbitrator:
|
||||
else:
|
||||
self._apply_oc_profile(self.IDLE_PROFILE)
|
||||
|
||||
async def _arbitrate(self) -> None:
|
||||
async def _arbitrate(self, active_tenant: Optional[str] = None) -> None:
|
||||
"""Generic arbitration over any number of tenants.
|
||||
|
||||
The two-application version was a pair of hardcoded rules -- yield Ollama when
|
||||
@@ -1328,7 +1347,8 @@ class AutoArbitrator:
|
||||
that lacks the VRAM it declares it needs is starved, and the memory comes from
|
||||
idle reclaimable tenants below it in priority, lowest first.
|
||||
"""
|
||||
state = await self._tenant_state()
|
||||
t_start = time.perf_counter()
|
||||
state = await self._tenant_state(active_tenant)
|
||||
free_gb = self._last_tenant_state["free_gb"]
|
||||
self._apply_profile_for_active(state)
|
||||
|
||||
@@ -1355,8 +1375,24 @@ class AutoArbitrator:
|
||||
for victim in plan["release"]:
|
||||
await self._release_tenant(
|
||||
victim, f"{s['name']} needs {s['needs_vram_gb']} GB, {free_gb} GB free")
|
||||
# Wait for the memory to actually come back, and record how long the whole
|
||||
# handoff took. Swap speed is the point of this service, so it is measured
|
||||
# rather than assumed.
|
||||
target_bytes = int(s["needs_vram_gb"] * (1024 ** 3))
|
||||
deadline = time.perf_counter() + 30.0
|
||||
while time.perf_counter() < deadline:
|
||||
if get_process_vram_bytes()["free_bytes"] >= target_bytes:
|
||||
break
|
||||
await asyncio.sleep(0.02)
|
||||
handoff_ms = round((time.perf_counter() - t_start) * 1000, 1)
|
||||
self.last_handoff = {"ts": time.time(), "to": s["name"],
|
||||
"released": plan["release"], "handoff_ms": handoff_ms,
|
||||
"triggered_by": "event" if active_tenant else "poll"}
|
||||
self.stats["handoffs"] = self.stats.get("handoffs", 0) + 1
|
||||
logger.info(f"Handoff to '{s['name']}' in {handoff_ms} ms "
|
||||
f"(released {', '.join(plan['release'])})")
|
||||
self.last_action = (f"Released {', '.join(plan['release'])} so "
|
||||
f"'{s['name']}' could work")
|
||||
f"'{s['name']}' could work — {handoff_ms} ms")
|
||||
return
|
||||
|
||||
# 2. Idle release: a tenant holding VRAM it is not using, after a grace period.
|
||||
@@ -1450,6 +1486,7 @@ class AutoArbitrator:
|
||||
"comfy_stale_job": self.comfy_stale_job,
|
||||
"event_sources": dict(self.event_sources),
|
||||
"last_arbitration": self.last_arbitration,
|
||||
"last_handoff": self.last_handoff,
|
||||
"tenant_state": self._last_tenant_state,
|
||||
"watchdog_branches": dict(self.watchdog_branches),
|
||||
"last_watchdog_error": self.last_watchdog_error,
|
||||
|
||||
Reference in New Issue
Block a user