Let any tenant declare its GPU profile and event source; fix priority semantics

Two remaining pieces of the two-application coupling are gone.

Overclock profiles were switched by naming 'comfy' and 'ollama' directly, so a third
application could never get tuned clocks. A tenant declares overclock_profile and the
arbitrator applies whichever the highest-priority *working* tenant asks for, falling
back to the idle profile when nothing is running.

The websocket listener parsed ComfyUI's message schema -- status, execution_start,
executing, execution_success -- which tied the fast path to one application. An event
source is now declarative and the messages are not parsed at all: any message means
"look now", and the tenant's own busy probe decides what is true. That gives the same
sub-second reaction to any application that emits anything on state change, with no
knowledge of what it emits.

Generalising this exposed a design error in the priority rule I had introduced.
plan_release excluded candidates ranking above the demander, which broke both
directions in turn. With the LLM at priority 60 and diffusion at 50, ComfyUI could
never reclaim from Ollama -- the premise the whole service is built on, and preserved
until now only by the ComfyUI-specific trigger that was about to be removed. Swapping
the ranks then broke the reverse: a starved Ollama could no longer reclaim from an
idle ComfyUI.

Priority now orders rather than vetoes. Any idle reclaimable tenant is a candidate,
because an idle tenant is not using its VRAM; priority decides who is asked first, and
busy tenants are never interrupted whatever their rank. Diffusion outranks the LLM,
whose weights reload from page cache in seconds. All three cases are pinned by tests,
including that busy work is never interrupted even by a far higher-priority demander.

Tests: 244 (was 242).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-07 15:23:48 -07:00
parent c6455d7c6e
commit ca97f18be6
5 changed files with 161 additions and 12 deletions

View File

@@ -70,6 +70,22 @@ class BusyProbe:
stale_after_s: float = 90.0
@dataclass
class EventSource:
"""A stream that tells us *when* to look, not what to think.
ComfyUI publishes a websocket, and the original listener parsed its message types to
decide what was happening -- which meant understanding one application's schema. Any
message is instead treated purely as a wake-up: re-run this tenant's busy probe now
rather than waiting for the next poll. That gives sub-second reaction to any
application with an event stream, with no knowledge of what it emits.
"""
type: str = "none" # none | websocket
url: Optional[str] = None
reconnect_backoff_s: float = 2.0
max_backoff_s: float = 15.0
@dataclass
class ReleaseStrategy:
"""How to ask a tenant to give VRAM back."""
@@ -96,9 +112,14 @@ class GpuTenant:
# back. Iterating on a ComfyUI workflow should not pay a reload between every run,
# so this is deliberately not immediate.
idle_release_after_s: float = 30.0
# GPU profile to apply while this tenant is the active workload. Clock and power
# tuning is workload-specific -- diffusion is compute bound, LLM decode is bandwidth
# bound -- and that was previously switched by application name in the arbitrator.
overclock_profile: Optional[str] = None
match: ProcessMatch = field(default_factory=ProcessMatch)
busy: BusyProbe = field(default_factory=BusyProbe)
release: ReleaseStrategy = field(default_factory=ReleaseStrategy)
events: EventSource = field(default_factory=EventSource)
notes: str = ""
@property
@@ -116,10 +137,12 @@ def _tenant_from_dict(d: Dict[str, Any]) -> GpuTenant:
enabled=d.get("enabled", True),
priority=int(d.get("priority", 50)),
needs_vram_gb=float(d.get("needs_vram_gb", 0.0)),
overclock_profile=d.get("overclock_profile"),
idle_release_after_s=float(d.get("idle_release_after_s", 30.0)),
match=ProcessMatch(**(d.get("match") or {})),
busy=BusyProbe(**(d.get("busy") or {})),
release=ReleaseStrategy(**(d.get("release") or {})),
events=EventSource(**(d.get("events") or {})),
notes=d.get("notes", ""),
)
@@ -129,9 +152,14 @@ DEFAULT_TENANTS: List[Dict[str, Any]] = [
{
"name": "ollama",
"kind": KIND_LLM,
"priority": 60,
# Lower than ComfyUI on purpose: an interactive diffusion job preempts the LLM,
# whose weights stay in the page cache and reload in seconds. Getting this the
# wrong way round silently disabled the service's central behaviour -- ComfyUI
# could never reclaim from Ollama.
"priority": 50,
"needs_vram_gb": 4.0,
"idle_release_after_s": 0.0,
"overclock_profile": "ollama",
"match": {"names": ["ollama"], "cmdline": ["llama-server", "ollama"]},
"busy": {"type": "http_count", "url": "http://localhost:11434/api/ps",
"count_keys": ["models"]},
@@ -143,9 +171,10 @@ DEFAULT_TENANTS: List[Dict[str, Any]] = [
{
"name": "comfyui",
"kind": KIND_DIFFUSION,
"priority": 50,
"priority": 60,
"needs_vram_gb": 6.0,
"idle_release_after_s": 30.0,
"overclock_profile": "comfy",
"match": {"cmdline": ["comfyui", "comfy"], "cmdline_endswith": ["main.py"]},
"busy": {"type": "http_count", "url": "http://127.0.0.1:8188/queue",
"count_keys": ["queue_running", "queue_pending"],
@@ -153,6 +182,7 @@ DEFAULT_TENANTS: List[Dict[str, Any]] = [
"release": {"type": "http_post", "url": "http://127.0.0.1:8188/free",
"body": {"unload_models": True, "free_memory": True},
"timeout_s": 30.0},
"events": {"type": "websocket", "url": "ws://127.0.0.1:8188/ws?clientId=hyperswap"},
"notes": "Leaves dead jobs in queue_running; the queue flag is corroborated "
"against its own VRAM before being believed.",
},
@@ -210,7 +240,7 @@ def load_tenants(force: bool = False) -> List[GpuTenant]:
base = defaults_by_name.get(d.get("name"))
if base:
merged = {**base, **d}
for key in ("match", "busy", "release"):
for key in ("match", "busy", "release", "events"):
if isinstance(base.get(key), dict):
merged[key] = {**base[key], **(d.get(key) or {})}
d = merged
@@ -364,12 +394,19 @@ def plan_release(demanding: str, tenants_state: List[Dict[str, Any]],
return {"possible": True, "reason": "enough VRAM is already free",
"release": [], "shortfall_gb": 0.0}
# Any idle reclaimable tenant is a candidate, whatever its priority. An idle tenant
# is not using its VRAM, so ranking above the demander should not protect it -- an
# earlier version filtered on priority and thereby broke both directions in turn:
# ComfyUI could not preempt Ollama, and once that was corrected a starved Ollama
# could no longer reclaim from an idle ComfyUI.
#
# Priority decides who is asked *first* (lowest gives up memory soonest) and, being
# applied only to idle tenants, never interrupts work.
candidates = [
s for s in tenants_state
if s["name"] != demanding
and s.get("reclaimable")
and not s.get("busy")
and s.get("priority", 0) <= demander.get("priority", 0)
and s.get("vram_gb", 0) > 0
]
candidates.sort(key=lambda s: (s.get("priority", 0), -s.get("vram_gb", 0)))
@@ -385,7 +422,7 @@ def plan_release(demanding: str, tenants_state: List[Dict[str, Any]],
{"name": s["name"], "vram_gb": s.get("vram_gb", 0.0),
"why": ("busy" if s.get("busy") else
"declares no release mechanism" if not s.get("reclaimable") else
"higher priority")}
"enough was freed without it")}
for s in tenants_state
if s["name"] != demanding and s.get("vram_gb", 0) > 0 and s["name"] not in plan
]