Arbitrate over any number of tenants by priority
Completes the generalisation. Classification and release were already data; the decision loop was still two hardcoded rules -- yield Ollama when ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express a third participant. plan_release() works from the registry instead. A busy tenant that cannot reach its declared needs_vram_gb is starved, and the memory comes from idle reclaimable tenants below it in priority, lowest first, stopping once enough is freed. Tenants that cannot be released are named as blockers rather than passed over, so an impossible plan says which process is in the way. The plan is returned before being acted on, so the decision is testable and is logged before anything is released. Idle release is now per-tenant too, replacing the ComfyUI-specific purge timer. Two bugs found by running it against the live machine rather than only in tests: Starvation was measured against free VRAM alone, so a tenant working perfectly well on 13 GB was flagged as demanding simply because little was left over -- which is the normal state of a busy GPU, and would have caused pointless releases from everything else. A tenant is starved only if it cannot reach what it needs counting what it already holds. Fields added to the tenant schema were silently absent from the config already written to disk, so needs_vram_gb defaulted to 0 and starvation could never trigger for the two tenants that mattered. Shipped defaults are now merged into an existing config on load, with explicit user values still winning. Tests: 242 (was 231), including a three-application contention case -- the property the hardcoded pair of rules could not express. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
89
tenants.py
89
tenants.py
@@ -89,6 +89,13 @@ class GpuTenant:
|
||||
enabled: bool = True
|
||||
# Higher wins contention; a tenant yields to anything above it.
|
||||
priority: int = 50
|
||||
# How much free VRAM this application needs before it can work. Used to decide
|
||||
# whether a busy tenant is actually being starved, rather than merely busy.
|
||||
needs_vram_gb: float = 0.0
|
||||
# How long a reclaimable tenant may sit idle holding VRAM before it is asked for it
|
||||
# back. Iterating on a ComfyUI workflow should not pay a reload between every run,
|
||||
# so this is deliberately not immediate.
|
||||
idle_release_after_s: float = 30.0
|
||||
match: ProcessMatch = field(default_factory=ProcessMatch)
|
||||
busy: BusyProbe = field(default_factory=BusyProbe)
|
||||
release: ReleaseStrategy = field(default_factory=ReleaseStrategy)
|
||||
@@ -108,6 +115,8 @@ def _tenant_from_dict(d: Dict[str, Any]) -> GpuTenant:
|
||||
kind=d.get("kind", KIND_OTHER),
|
||||
enabled=d.get("enabled", True),
|
||||
priority=int(d.get("priority", 50)),
|
||||
needs_vram_gb=float(d.get("needs_vram_gb", 0.0)),
|
||||
idle_release_after_s=float(d.get("idle_release_after_s", 30.0)),
|
||||
match=ProcessMatch(**(d.get("match") or {})),
|
||||
busy=BusyProbe(**(d.get("busy") or {})),
|
||||
release=ReleaseStrategy(**(d.get("release") or {})),
|
||||
@@ -121,6 +130,8 @@ DEFAULT_TENANTS: List[Dict[str, Any]] = [
|
||||
"name": "ollama",
|
||||
"kind": KIND_LLM,
|
||||
"priority": 60,
|
||||
"needs_vram_gb": 4.0,
|
||||
"idle_release_after_s": 0.0,
|
||||
"match": {"names": ["ollama"], "cmdline": ["llama-server", "ollama"]},
|
||||
"busy": {"type": "http_count", "url": "http://localhost:11434/api/ps",
|
||||
"count_keys": ["models"]},
|
||||
@@ -133,6 +144,8 @@ DEFAULT_TENANTS: List[Dict[str, Any]] = [
|
||||
"name": "comfyui",
|
||||
"kind": KIND_DIFFUSION,
|
||||
"priority": 50,
|
||||
"needs_vram_gb": 6.0,
|
||||
"idle_release_after_s": 30.0,
|
||||
"match": {"cmdline": ["comfyui", "comfy"], "cmdline_endswith": ["main.py"]},
|
||||
"busy": {"type": "http_count", "url": "http://127.0.0.1:8188/queue",
|
||||
"count_keys": ["queue_running", "queue_pending"],
|
||||
@@ -187,8 +200,20 @@ def load_tenants(force: bool = False) -> List[GpuTenant]:
|
||||
except Exception as e:
|
||||
logger.warning(f"could not write {CONFIG_PATH}: {e}")
|
||||
|
||||
# Merge in any fields a shipped default has gained since the config was written.
|
||||
# Without this, adding a field silently disables the behaviour it controls for every
|
||||
# existing install -- needs_vram_gb defaulted to 0, which made starvation
|
||||
# undetectable for the two tenants that had been written out before it existed.
|
||||
defaults_by_name = {d["name"]: d for d in DEFAULT_TENANTS}
|
||||
tenants = []
|
||||
for d in raw:
|
||||
base = defaults_by_name.get(d.get("name"))
|
||||
if base:
|
||||
merged = {**base, **d}
|
||||
for key in ("match", "busy", "release"):
|
||||
if isinstance(base.get(key), dict):
|
||||
merged[key] = {**base[key], **(d.get(key) or {})}
|
||||
d = merged
|
||||
try:
|
||||
tenants.append(_tenant_from_dict(d))
|
||||
except Exception as e:
|
||||
@@ -311,3 +336,67 @@ def describe() -> List[Dict[str, Any]]:
|
||||
d["release_via"] = t.release.type
|
||||
out.append(d)
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------- arbitration
|
||||
|
||||
def plan_release(demanding: str, tenants_state: List[Dict[str, Any]],
|
||||
free_gb: float, needed_gb: float) -> Dict[str, Any]:
|
||||
"""Decide who should give up VRAM so a starved tenant can work.
|
||||
|
||||
Generic over any number of applications: candidates are every *reclaimable* tenant
|
||||
that is not itself busy and ranks below the demanding one, taken lowest priority
|
||||
first, until enough would be freed. The two-application version of this was a pair
|
||||
of hardcoded rules -- yield Ollama for ComfyUI, purge ComfyUI for Ollama -- which
|
||||
could not express a third participant at all.
|
||||
|
||||
Returns the plan rather than performing it, so the decision is testable and can be
|
||||
logged before anything is actually released.
|
||||
"""
|
||||
by_name = {s["name"]: s for s in tenants_state}
|
||||
demander = by_name.get(demanding)
|
||||
if not demander:
|
||||
return {"possible": False, "reason": f"unknown tenant '{demanding}'", "release": []}
|
||||
|
||||
# The demander keeps what it already holds; only the remainder must be found.
|
||||
shortfall = needed_gb - free_gb - demander.get("vram_gb", 0.0)
|
||||
if shortfall <= 0:
|
||||
return {"possible": True, "reason": "enough VRAM is already free",
|
||||
"release": [], "shortfall_gb": 0.0}
|
||||
|
||||
candidates = [
|
||||
s for s in tenants_state
|
||||
if s["name"] != demanding
|
||||
and s.get("reclaimable")
|
||||
and not s.get("busy")
|
||||
and s.get("priority", 0) <= demander.get("priority", 0)
|
||||
and s.get("vram_gb", 0) > 0
|
||||
]
|
||||
candidates.sort(key=lambda s: (s.get("priority", 0), -s.get("vram_gb", 0)))
|
||||
|
||||
plan, freed = [], 0.0
|
||||
for c in candidates:
|
||||
if freed >= shortfall:
|
||||
break
|
||||
plan.append(c["name"])
|
||||
freed += c.get("vram_gb", 0.0)
|
||||
|
||||
blockers = [
|
||||
{"name": s["name"], "vram_gb": s.get("vram_gb", 0.0),
|
||||
"why": ("busy" if s.get("busy") else
|
||||
"declares no release mechanism" if not s.get("reclaimable") else
|
||||
"higher priority")}
|
||||
for s in tenants_state
|
||||
if s["name"] != demanding and s.get("vram_gb", 0) > 0 and s["name"] not in plan
|
||||
]
|
||||
|
||||
return {
|
||||
"possible": freed >= shortfall,
|
||||
"shortfall_gb": round(shortfall, 2),
|
||||
"would_free_gb": round(freed, 2),
|
||||
"release": plan,
|
||||
"blockers": blockers,
|
||||
"reason": (f"releasing {', '.join(plan)} frees {freed:.2f} GB of the "
|
||||
f"{shortfall:.2f} GB shortfall" if plan else
|
||||
"no reclaimable idle tenant holds enough VRAM"),
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user