Let any tenant declare its GPU profile and event source; fix priority semantics
Two remaining pieces of the two-application coupling are gone. Overclock profiles were switched by naming 'comfy' and 'ollama' directly, so a third application could never get tuned clocks. A tenant declares overclock_profile and the arbitrator applies whichever the highest-priority *working* tenant asks for, falling back to the idle profile when nothing is running. The websocket listener parsed ComfyUI's message schema -- status, execution_start, executing, execution_success -- which tied the fast path to one application. An event source is now declarative and the messages are not parsed at all: any message means "look now", and the tenant's own busy probe decides what is true. That gives the same sub-second reaction to any application that emits anything on state change, with no knowledge of what it emits. Generalising this exposed a design error in the priority rule I had introduced. plan_release excluded candidates ranking above the demander, which broke both directions in turn. With the LLM at priority 60 and diffusion at 50, ComfyUI could never reclaim from Ollama -- the premise the whole service is built on, and preserved until now only by the ComfyUI-specific trigger that was about to be removed. Swapping the ranks then broke the reverse: a starved Ollama could no longer reclaim from an idle ComfyUI. Priority now orders rather than vetoes. Any idle reclaimable tenant is a candidate, because an idle tenant is not using its VRAM; priority decides who is asked first, and busy tenants are never interrupted whatever their rank. Diffusion outranks the LLM, whose weights reload from page cache in seconds. All three cases are pinned by tests, including that busy work is never interrupted even by a far higher-priority demander. Tests: 244 (was 242). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -932,6 +932,7 @@ class AutoArbitrator:
|
||||
def __init__(self):
|
||||
self.running = False
|
||||
self.ws_task: Optional[asyncio.Task] = None
|
||||
self.event_tasks: List[asyncio.Task] = []
|
||||
self.poll_task: Optional[asyncio.Task] = None
|
||||
self.idle_task: Optional[asyncio.Task] = None
|
||||
self.last_yield_time = 0.0
|
||||
@@ -958,6 +959,8 @@ class AutoArbitrator:
|
||||
self._peak_comfy_bytes = 0
|
||||
self.comfy_stale_job: Optional[str] = None
|
||||
self._idle_since: Dict[str, float] = {}
|
||||
self._last_event_wake = 0.0
|
||||
self.event_sources: Dict[str, str] = {}
|
||||
self._last_tenant_state: Optional[Dict[str, Any]] = None
|
||||
self.last_arbitration: Optional[Dict[str, Any]] = None
|
||||
self.last_watchdog_error: Optional[str] = None
|
||||
@@ -976,6 +979,12 @@ class AutoArbitrator:
|
||||
return
|
||||
self.running = True
|
||||
self.ws_task = asyncio.create_task(self._ws_listener())
|
||||
for t in tenants_mod.load_tenants():
|
||||
if t.enabled and t.events.type == "websocket" and t.events.url:
|
||||
self.event_tasks.append(asyncio.create_task(
|
||||
self._event_listener(t.name, t.events.url,
|
||||
t.events.reconnect_backoff_s,
|
||||
t.events.max_backoff_s)))
|
||||
self.poll_task = asyncio.create_task(self._poll_watchdog())
|
||||
self.idle_task = asyncio.create_task(self._idle_purge_loop())
|
||||
logger.info("AutoArbitrator background engine started (Bidirectional).")
|
||||
@@ -988,9 +997,10 @@ class AutoArbitrator:
|
||||
|
||||
async def stop(self):
|
||||
self.running = False
|
||||
for task in (self.ws_task, self.poll_task, self.idle_task):
|
||||
for task in [self.ws_task, self.poll_task, self.idle_task, *self.event_tasks]:
|
||||
if task:
|
||||
task.cancel()
|
||||
self.event_tasks.clear()
|
||||
await close_clients()
|
||||
logger.info("AutoArbitrator background engine stopped.")
|
||||
|
||||
@@ -1100,6 +1110,47 @@ class AutoArbitrator:
|
||||
return {"purged": True, "free_gb": round(snap["free_bytes"] / (1024**3), 2)}
|
||||
return {"purged": False, "free_gb": round(free_gb, 2), "reason": "ComfyUI holds no VRAM"}
|
||||
|
||||
async def _event_listener(self, tenant_name: str, url: str,
|
||||
backoff_s: float, max_backoff_s: float) -> None:
|
||||
"""Wake on a tenant's event stream instead of waiting for the next poll.
|
||||
|
||||
Deliberately does not parse the messages. The previous listener understood
|
||||
ComfyUI's schema -- status/execution_start/executing/execution_success -- which
|
||||
tied the fast path to one application. Treating any message as "look now" and
|
||||
letting the tenant's own busy probe decide gives the same sub-second reaction
|
||||
for any application that emits anything on state change.
|
||||
"""
|
||||
backoff = backoff_s
|
||||
while self.running:
|
||||
try:
|
||||
async with websockets.connect(url, ping_interval=10, ping_timeout=10) as ws:
|
||||
self.event_sources[tenant_name] = "connected"
|
||||
self.connected_ws = True
|
||||
backoff = backoff_s
|
||||
logger.info(f"Event source connected for '{tenant_name}': {url}")
|
||||
while self.running:
|
||||
await ws.recv()
|
||||
# Coalesce bursts: a single graph emits many messages, and one
|
||||
# arbitration pass per burst is enough.
|
||||
now = time.time()
|
||||
if now - self._last_event_wake < 0.25:
|
||||
continue
|
||||
self._last_event_wake = now
|
||||
self.stats["event_wakeups"] = self.stats.get("event_wakeups", 0) + 1
|
||||
try:
|
||||
await self._arbitrate()
|
||||
except Exception as e:
|
||||
logger.debug(f"arbitration from event failed: {e}")
|
||||
except (websockets.exceptions.ConnectionClosed, OSError, asyncio.CancelledError):
|
||||
self.event_sources[tenant_name] = "disconnected"
|
||||
self.connected_ws = False
|
||||
except Exception as e:
|
||||
self.event_sources[tenant_name] = f"error: {str(e)[:60]}"
|
||||
self.connected_ws = False
|
||||
logger.debug(f"event source error for '{tenant_name}': {e}")
|
||||
await asyncio.sleep(backoff)
|
||||
backoff = min(backoff * 1.5, max_backoff_s)
|
||||
|
||||
async def _ws_listener(self):
|
||||
client_id = "hyperswap-arbitrator"
|
||||
ws_url = f"ws://127.0.0.1:8188/ws?clientId={client_id}"
|
||||
@@ -1229,6 +1280,7 @@ class AutoArbitrator:
|
||||
"below_floor": bool(probe.get("below_floor")),
|
||||
"reclaimable": t.reclaimable,
|
||||
"needs_vram_gb": t.needs_vram_gb,
|
||||
"overclock_profile": t.overclock_profile,
|
||||
"idle_release_after_s": t.idle_release_after_s,
|
||||
"reason": probe.get("reason"),
|
||||
})
|
||||
@@ -1251,6 +1303,22 @@ class AutoArbitrator:
|
||||
self.stats["tenant_releases"] = self.stats.get("tenant_releases", 0) + 1
|
||||
return res
|
||||
|
||||
IDLE_PROFILE = "balanced"
|
||||
|
||||
def _apply_profile_for_active(self, state: List[Dict[str, Any]]) -> None:
|
||||
"""Apply the GPU profile declared by whichever tenant is currently working.
|
||||
|
||||
This used to be two calls naming 'comfy' and 'ollama' directly, so a third
|
||||
application could never get tuned clocks. The highest-priority busy tenant wins;
|
||||
with nothing working the card returns to the idle profile.
|
||||
"""
|
||||
busy = [s for s in state if s["busy"] and s.get("overclock_profile")]
|
||||
if busy:
|
||||
busy.sort(key=lambda s: -s["priority"])
|
||||
self._apply_oc_profile(busy[0]["overclock_profile"])
|
||||
else:
|
||||
self._apply_oc_profile(self.IDLE_PROFILE)
|
||||
|
||||
async def _arbitrate(self) -> None:
|
||||
"""Generic arbitration over any number of tenants.
|
||||
|
||||
@@ -1262,6 +1330,7 @@ class AutoArbitrator:
|
||||
"""
|
||||
state = await self._tenant_state()
|
||||
free_gb = self._last_tenant_state["free_gb"]
|
||||
self._apply_profile_for_active(state)
|
||||
|
||||
# 1. Starvation: highest-priority demanding tenant first.
|
||||
for s in sorted(state, key=lambda x: -x["priority"]):
|
||||
@@ -1379,6 +1448,7 @@ class AutoArbitrator:
|
||||
"oc_profile": self.oc_profile,
|
||||
"counters": dict(self.stats),
|
||||
"comfy_stale_job": self.comfy_stale_job,
|
||||
"event_sources": dict(self.event_sources),
|
||||
"last_arbitration": self.last_arbitration,
|
||||
"tenant_state": self._last_tenant_state,
|
||||
"watchdog_branches": dict(self.watchdog_branches),
|
||||
|
||||
Reference in New Issue
Block a user