Arbitrate over any number of tenants by priority
Completes the generalisation. Classification and release were already data; the decision loop was still two hardcoded rules -- yield Ollama when ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express a third participant. plan_release() works from the registry instead. A busy tenant that cannot reach its declared needs_vram_gb is starved, and the memory comes from idle reclaimable tenants below it in priority, lowest first, stopping once enough is freed. Tenants that cannot be released are named as blockers rather than passed over, so an impossible plan says which process is in the way. The plan is returned before being acted on, so the decision is testable and is logged before anything is released. Idle release is now per-tenant too, replacing the ComfyUI-specific purge timer. Two bugs found by running it against the live machine rather than only in tests: Starvation was measured against free VRAM alone, so a tenant working perfectly well on 13 GB was flagged as demanding simply because little was left over -- which is the normal state of a busy GPU, and would have caused pointless releases from everything else. A tenant is starved only if it cannot reach what it needs counting what it already holds. Fields added to the tenant schema were silently absent from the config already written to disk, so needs_vram_gb defaulted to 0 and starvation could never trigger for the two tenants that mattered. Shipped defaults are now merged into an existing config on load, with explicit user values still winning. Tests: 242 (was 231), including a three-application contention case -- the property the hardcoded pair of rules could not express. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -949,6 +949,9 @@ class AutoArbitrator:
|
||||
self._running_since: Optional[float] = None
|
||||
self._peak_comfy_bytes = 0
|
||||
self.comfy_stale_job: Optional[str] = None
|
||||
self._idle_since: Dict[str, float] = {}
|
||||
self._last_tenant_state: Optional[Dict[str, Any]] = None
|
||||
self.last_arbitration: Optional[Dict[str, Any]] = None
|
||||
self.last_watchdog_error: Optional[str] = None
|
||||
self.stats = {
|
||||
"yields": 0, # release confirmed
|
||||
@@ -1198,6 +1201,106 @@ class AutoArbitrator:
|
||||
return False
|
||||
return True
|
||||
|
||||
async def _tenant_state(self) -> List[Dict[str, Any]]:
|
||||
"""Current VRAM and busy state for every configured tenant."""
|
||||
snap = get_process_vram_bytes()
|
||||
stats = get_gpu_hardware_stats()
|
||||
by_tenant = (stats.get("breakdown", {}) or {}).get("by_tenant_gb", {})
|
||||
out = []
|
||||
for t in tenants_mod.load_tenants():
|
||||
if not t.enabled:
|
||||
continue
|
||||
bucket = _BUCKET_ALIASES.get(t.name, t.name)
|
||||
vram_gb = by_tenant.get(bucket, 0.0)
|
||||
probe = await tenants_mod.probe_busy(t, vram_gb=vram_gb)
|
||||
out.append({
|
||||
"name": t.name,
|
||||
"priority": t.priority,
|
||||
"vram_gb": vram_gb,
|
||||
"busy": bool(probe.get("busy")),
|
||||
"below_floor": bool(probe.get("below_floor")),
|
||||
"reclaimable": t.reclaimable,
|
||||
"needs_vram_gb": t.needs_vram_gb,
|
||||
"idle_release_after_s": t.idle_release_after_s,
|
||||
"reason": probe.get("reason"),
|
||||
})
|
||||
self._last_tenant_state = {"ts": time.time(), "free_gb":
|
||||
round(snap["free_bytes"] / (1024**3), 2),
|
||||
"tenants": out}
|
||||
return out
|
||||
|
||||
async def _release_tenant(self, name: str, reason: str) -> Dict[str, Any]:
|
||||
"""Release one tenant's VRAM by whatever mechanism it declares."""
|
||||
t = tenants_mod.get_tenant(name)
|
||||
if not t or not t.reclaimable:
|
||||
return {"success": False, "reason": "not reclaimable"}
|
||||
models = None
|
||||
if t.release.per_model:
|
||||
state = await get_ollama_live_state()
|
||||
models = [m.get("name") for m in state.get("loaded_models", []) if m.get("name")]
|
||||
logger.info(f"Releasing VRAM from '{name}': {reason}")
|
||||
res = await tenants_mod.release_vram(t, models=models)
|
||||
self.stats["tenant_releases"] = self.stats.get("tenant_releases", 0) + 1
|
||||
return res
|
||||
|
||||
async def _arbitrate(self) -> None:
|
||||
"""Generic arbitration over any number of tenants.
|
||||
|
||||
The two-application version was a pair of hardcoded rules -- yield Ollama when
|
||||
ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express
|
||||
a third participant at all. This works from the registry instead: a busy tenant
|
||||
that lacks the VRAM it declares it needs is starved, and the memory comes from
|
||||
idle reclaimable tenants below it in priority, lowest first.
|
||||
"""
|
||||
state = await self._tenant_state()
|
||||
free_gb = self._last_tenant_state["free_gb"]
|
||||
|
||||
# 1. Starvation: highest-priority demanding tenant first.
|
||||
for s in sorted(state, key=lambda x: -x["priority"]):
|
||||
if not s["busy"] or not s["needs_vram_gb"]:
|
||||
continue
|
||||
# Starved means it cannot reach what it needs even counting what it already
|
||||
# holds. Comparing free VRAM alone flagged a tenant that was working
|
||||
# perfectly well on 13 GB as demanding, purely because little was left over
|
||||
# -- which is the normal state of a busy GPU, and would have caused
|
||||
# pointless releases from everyone else.
|
||||
if s["vram_gb"] + free_gb >= s["needs_vram_gb"]:
|
||||
continue
|
||||
plan = tenants_mod.plan_release(s["name"], state, free_gb, s["needs_vram_gb"])
|
||||
self.last_arbitration = {"ts": time.time(), "demanding": s["name"],
|
||||
"free_gb": free_gb, **plan}
|
||||
if not plan["release"]:
|
||||
logger.debug(f"'{s['name']}' is short of VRAM but {plan['reason']}")
|
||||
return
|
||||
if time.time() - self.last_reclaim_time < self.RECLAIM_COOLDOWN_S:
|
||||
return
|
||||
self.last_reclaim_time = time.time()
|
||||
for victim in plan["release"]:
|
||||
await self._release_tenant(
|
||||
victim, f"{s['name']} needs {s['needs_vram_gb']} GB, {free_gb} GB free")
|
||||
self.last_action = (f"Released {', '.join(plan['release'])} so "
|
||||
f"'{s['name']}' could work")
|
||||
return
|
||||
|
||||
# 2. Idle release: a tenant holding VRAM it is not using, after a grace period.
|
||||
now = time.time()
|
||||
for s in state:
|
||||
if not s["reclaimable"] or s["vram_gb"] <= 0.25:
|
||||
self._idle_since.pop(s["name"], None)
|
||||
continue
|
||||
if s["busy"]:
|
||||
self._idle_since.pop(s["name"], None)
|
||||
continue
|
||||
since = self._idle_since.setdefault(s["name"], now)
|
||||
grace = s["idle_release_after_s"]
|
||||
if grace and (now - since) >= grace:
|
||||
self._idle_since.pop(s["name"], None)
|
||||
await self._release_tenant(
|
||||
s["name"], f"idle {int(now - since)}s holding {s['vram_gb']} GB")
|
||||
self.last_action = (f"Released idle '{s['name']}' after "
|
||||
f"{int(now - since)}s")
|
||||
return
|
||||
|
||||
async def _check_ollama_starved(self) -> None:
|
||||
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
|
||||
|
||||
@@ -1290,6 +1393,7 @@ class AutoArbitrator:
|
||||
else:
|
||||
self.watchdog_branches["idle_check"] += 1
|
||||
await self._check_ollama_starved()
|
||||
await self._arbitrate()
|
||||
else:
|
||||
self.watchdog_branches["bad_status"] += 1
|
||||
except Exception as e:
|
||||
@@ -1338,6 +1442,8 @@ class AutoArbitrator:
|
||||
"counters": dict(self.stats),
|
||||
"last_starvation_check": self.last_starvation_check,
|
||||
"comfy_stale_job": self.comfy_stale_job,
|
||||
"last_arbitration": self.last_arbitration,
|
||||
"tenant_state": self._last_tenant_state,
|
||||
"watchdog_branches": dict(self.watchdog_branches),
|
||||
"last_watchdog_error": self.last_watchdog_error,
|
||||
"yield_backoff": {m: round(max(t - time.time(), 0), 1)
|
||||
|
||||
Reference in New Issue
Block a user