Files
gpu-program-swapper/tests/test_tenants.py
drjones ca97f18be6 Let any tenant declare its GPU profile and event source; fix priority semantics
Two remaining pieces of the two-application coupling are gone.

Overclock profiles were switched by naming 'comfy' and 'ollama' directly, so a third
application could never get tuned clocks. A tenant declares overclock_profile and the
arbitrator applies whichever the highest-priority *working* tenant asks for, falling
back to the idle profile when nothing is running.

The websocket listener parsed ComfyUI's message schema -- status, execution_start,
executing, execution_success -- which tied the fast path to one application. An event
source is now declarative and the messages are not parsed at all: any message means
"look now", and the tenant's own busy probe decides what is true. That gives the same
sub-second reaction to any application that emits anything on state change, with no
knowledge of what it emits.

Generalising this exposed a design error in the priority rule I had introduced.
plan_release excluded candidates ranking above the demander, which broke both
directions in turn. With the LLM at priority 60 and diffusion at 50, ComfyUI could
never reclaim from Ollama -- the premise the whole service is built on, and preserved
until now only by the ComfyUI-specific trigger that was about to be removed. Swapping
the ranks then broke the reverse: a starved Ollama could no longer reclaim from an
idle ComfyUI.

Priority now orders rather than vetoes. Any idle reclaimable tenant is a candidate,
because an idle tenant is not using its VRAM; priority decides who is asked first, and
busy tenants are never interrupted whatever their rank. Diffusion outranks the LLM,
whose weights reload from page cache in seconds. All three cases are pinned by tests,
including that busy work is never interrupted even by a far higher-priority demander.

Tests: 244 (was 242).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-07 15:23:48 -07:00

316 lines
14 KiB
Python

"""Tests for the GPU tenant registry.
The point of this service is fast handoff of one GPU between applications, and it should
work for any application -- not only the two it grew up around. Their names had ended up
compiled into process matching, VRAM attribution, busy detection and release calls alike.
These tests pin the properties that make the registry generic: adding an application is
configuration, and nothing in the arbitration logic knows a particular name.
"""
import asyncio
import json
import pytest
import tenants as T
@pytest.fixture
def cfg(tmp_path, monkeypatch):
path = tmp_path / "tenants.json"
monkeypatch.setattr(T, "CONFIG_PATH", str(path))
T._cache.update({"ts": 0.0, "tenants": None, "mtime": None})
return path
class TestProcessMatching:
def test_matches_by_process_name(self):
m = T.ProcessMatch(names=["ollama"])
assert m.matches("ollama", "/usr/bin/ollama serve")
assert not m.matches("python", "main.py")
def test_matches_by_cmdline_substring(self):
m = T.ProcessMatch(cmdline=["llama-server"])
assert m.matches("python", "/usr/local/lib/ollama/llama-server --model x")
def test_matches_by_cmdline_suffix(self):
# ComfyUI is a bare `python main.py`, with nothing else distinguishing it.
m = T.ProcessMatch(cmdline_endswith=["main.py"])
assert m.matches("python", "/opt/ComfyUI/venv/bin/python main.py")
assert not m.matches("python", "/opt/other/main.py --serve")
def test_matching_is_case_insensitive(self):
assert T.ProcessMatch(names=["Xorg"]).matches("XORG", "")
class TestDefaultsPreserveExistingBehaviour:
"""The shipped defaults must classify exactly as the hardcoded version did."""
@pytest.mark.parametrize("pname,cmdline,expected", [
("llama-server", "/usr/local/lib/ollama/llama-server --model x", "ollama"),
("ollama", "/usr/bin/ollama serve", "ollama"),
("python", "/home/u/ComfyUI/venv/bin/python main.py --listen", "comfyui"),
("gnome-shell", "/usr/bin/gnome-shell --mode=ubuntu", "desktop"),
("Xorg", "/usr/lib/xorg/Xorg :8", "desktop"),
("python", "/home/u/robopest-venv/bin/python /home/u/stt_relay.py", "unmanaged"),
("trainer", "/opt/ml/bin/trainer --epochs 3", "unmanaged"),
])
def test_classification(self, cfg, pname, cmdline, expected):
assert T.classify_process(pname, cmdline) == expected
def test_unknown_process_is_unmanaged_not_silently_owned(self, cfg):
# Misattributing a third party's VRAM to a tenant would make this service
# promise headroom it cannot deliver.
assert T.classify_process("weird", "/opt/x/weird --run") == "unmanaged"
class TestAddingAnApplicationIsConfiguration:
def test_a_new_tenant_is_recognised_without_code_changes(self, cfg):
cfg.write_text(json.dumps(T.DEFAULT_TENANTS + [{
"name": "trainer",
"kind": "other",
"priority": 80,
"match": {"cmdline": ["train.py"]},
"release": {"type": "http_post", "url": "http://localhost:9999/release"},
}]))
assert T.classify_process("python", "/opt/ml/train.py --epochs 3") == "trainer"
t = T.get_tenant("trainer")
assert t.priority == 80 and t.reclaimable
def test_first_run_writes_the_defaults(self, cfg):
assert not cfg.exists()
T.load_tenants(force=True)
assert cfg.exists()
assert {t["name"] for t in json.loads(cfg.read_text())} == {
"ollama", "comfyui", "desktop"}
def test_a_malformed_entry_is_skipped_not_fatal(self, cfg):
cfg.write_text(json.dumps([{"name": "ok", "match": {"names": ["a"]}},
{"no_name": True}]))
names = [t.name for t in T.load_tenants(force=True)]
assert names == ["ok"]
def test_corrupt_config_falls_back_to_defaults(self, cfg):
cfg.write_text("{ not json")
assert {t.name for t in T.load_tenants(force=True)} >= {"ollama", "comfyui"}
class TestReclaimability:
def test_a_tenant_with_no_release_strategy_is_not_reclaimable(self, cfg):
t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none"))
assert t.reclaimable is False
def test_release_refuses_rather_than_reporting_success(self, cfg):
t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none"))
res = asyncio.run(T.release_vram(t))
assert res["success"] is False and res["released"] is False
assert "no way to release" in res["reason"]
def test_per_model_release_with_nothing_loaded_is_a_no_op(self, cfg):
t = T.GpuTenant(name="ollama", release=T.ReleaseStrategy(
type="http_post", url="http://x/api", per_model=True))
res = asyncio.run(T.release_vram(t, models=[]))
assert res["success"] is True and res["released"] is False
class TestBusyProbe:
def _probe(self, monkeypatch, payload, status=200):
class _R:
status_code = status
def json(self_inner): return payload
class _C:
async def __aenter__(self): return self
async def __aexit__(self, *a): return False
async def get(self, url): return _R()
monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C())
def test_empty_queue_is_not_busy(self, monkeypatch):
self._probe(monkeypatch, {"queue_running": [], "queue_pending": []})
t = T.GpuTenant(name="c", busy=T.BusyProbe(
type="http_count", url="http://x/queue",
count_keys=["queue_running", "queue_pending"]))
assert asyncio.run(T.probe_busy(t))["busy"] is False
def test_queued_work_while_holding_no_vram_is_flagged_below_floor(self, monkeypatch):
# ComfyUI leaves dead jobs in queue_running; only its VRAM reveals that nothing
# is loaded.
self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []})
t = T.GpuTenant(name="c", busy=T.BusyProbe(
type="http_count", url="http://x/queue", count_keys=["queue_running"],
vram_floor_gb=1.5))
res = asyncio.run(T.probe_busy(t, vram_gb=0.56))
assert res["busy"] is True and res.get("below_floor") is True
def test_queued_work_with_a_checkpoint_loaded_is_plainly_busy(self, monkeypatch):
self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []})
t = T.GpuTenant(name="c", busy=T.BusyProbe(
type="http_count", url="http://x/queue", count_keys=["queue_running"],
vram_floor_gb=1.5))
res = asyncio.run(T.probe_busy(t, vram_gb=6.8))
assert res["busy"] is True and not res.get("below_floor")
def test_vram_probe_needs_no_http_endpoint(self):
# An application with no API can still be observed by what it holds.
t = T.GpuTenant(name="x", busy=T.BusyProbe(type="vram", vram_busy_gb=1.0))
assert asyncio.run(T.probe_busy(t, vram_gb=2.0))["busy"] is True
assert asyncio.run(T.probe_busy(t, vram_gb=0.5))["busy"] is False
def test_an_unreachable_probe_reports_not_busy_rather_than_raising(self, monkeypatch):
class _C:
async def __aenter__(self): return self
async def __aexit__(self, *a): return False
async def get(self, url): raise ConnectionError("refused")
monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C())
t = T.GpuTenant(name="c", busy=T.BusyProbe(type="http_count", url="http://x",
count_keys=["q"]))
res = asyncio.run(T.probe_busy(t))
assert res["busy"] is False and "failed" in res["reason"]
class TestPriority:
def test_describe_orders_by_priority(self, cfg):
rows = T.describe()
prios = [r["priority"] for r in rows]
assert prios == sorted(prios, reverse=True)
assert all("reclaimable" in r for r in rows)
class TestReleasePlanning:
"""Deciding who gives up VRAM, generically over any number of applications.
The two-application version was a pair of hardcoded rules -- yield Ollama when
ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express a
third participant at all.
"""
def _state(self, **overrides):
base = [
{"name": "desktop", "priority": 90, "vram_gb": 0.01, "busy": False,
"reclaimable": False},
{"name": "stt-relay", "priority": 70, "vram_gb": 0.8, "busy": False,
"reclaimable": False},
{"name": "ollama", "priority": 60, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "comfyui", "priority": 50, "vram_gb": 7.0, "busy": False,
"reclaimable": True},
]
for s in base:
s.update(overrides.get(s["name"], {}))
return base
def test_a_tenant_already_holding_what_it_needs_is_not_starved(self):
# A busy GPU has little free by definition. Comparing free VRAM alone flagged a
# tenant working fine on 13 GB as demanding, which would have caused pointless
# releases from everything else.
state = self._state(ollama={"vram_gb": 13.0})
plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=4.0)
assert plan["release"] == []
assert "already free" in plan["reason"]
def test_starved_tenant_reclaims_from_the_idle_one_below_it(self):
plan = T.plan_release("ollama", self._state(), free_gb=1.5, needed_gb=14.9)
assert plan["release"] == ["comfyui"]
def test_a_busy_tenant_is_never_a_victim(self):
state = self._state(comfyui={"busy": True})
plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=14.9)
assert plan["release"] == []
assert any(b["name"] == "comfyui" and b["why"] == "busy"
for b in plan["blockers"])
def test_unreclaimable_tenants_are_named_as_blockers_not_ignored(self):
# The user needs to know a third-party process is what stands in the way.
plan = T.plan_release("ollama", self._state(), free_gb=0.0, needed_gb=15.5)
blockers = {b["name"]: b["why"] for b in plan["blockers"]}
assert blockers["stt-relay"] == "declares no release mechanism"
assert plan["possible"] is False
def test_an_idle_tenant_yields_even_if_it_outranks_the_demander(self):
"""Priority orders who is asked first; it does not protect idle memory.
Filtering candidates by priority broke both directions in turn: with the LLM
ranked above diffusion, ComfyUI could never preempt Ollama -- the service's
central behaviour -- and once the ranks were swapped, a starved Ollama could no
longer reclaim from an idle ComfyUI. An idle tenant is not using its VRAM, so
outranking the demander is not a reason to keep it.
"""
state = self._state(comfyui={"priority": 99, "busy": False})
plan = T.plan_release("ollama", state, free_gb=1.0, needed_gb=14.9)
assert plan["release"] == ["comfyui"]
def test_priority_decides_who_is_asked_first(self):
state = [
{"name": "demander", "priority": 50, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "high", "priority": 90, "vram_gb": 4.0, "busy": False,
"reclaimable": True},
{"name": "low", "priority": 10, "vram_gb": 4.0, "busy": False,
"reclaimable": True},
]
plan = T.plan_release("demander", state, free_gb=0.0, needed_gb=5.0)
# The lowest-priority idle tenant gives up memory first.
assert plan["release"][0] == "low"
def test_busy_work_is_never_interrupted_whatever_the_priority(self):
state = [
{"name": "demander", "priority": 99, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "worker", "priority": 1, "vram_gb": 8.0, "busy": True,
"reclaimable": True},
]
plan = T.plan_release("demander", state, free_gb=0.0, needed_gb=8.0)
assert plan["release"] == []
assert plan["blockers"][0]["why"] == "busy"
def test_lowest_priority_is_released_first(self):
state = self._state() + [
{"name": "batch", "priority": 10, "vram_gb": 3.0, "busy": False,
"reclaimable": True}]
plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=5.0)
assert plan["release"][0] == "batch"
def test_releases_only_as_many_tenants_as_needed(self):
state = self._state() + [
{"name": "batch", "priority": 10, "vram_gb": 9.0, "busy": False,
"reclaimable": True}]
plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=8.0)
assert plan["release"] == ["batch"] # 9 GB covers it; comfyui is left alone
def test_three_applications_can_all_participate(self):
# The property the hardcoded pair of rules could not express.
state = [
{"name": "llm", "priority": 60, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "diffusion", "priority": 50, "vram_gb": 4.0, "busy": False,
"reclaimable": True},
{"name": "trainer", "priority": 40, "vram_gb": 5.0, "busy": False,
"reclaimable": True},
]
plan = T.plan_release("llm", state, free_gb=0.0, needed_gb=9.0)
assert set(plan["release"]) == {"trainer", "diffusion"}
assert plan["possible"] is True
def test_unknown_tenant_is_rejected_cleanly(self):
plan = T.plan_release("nope", self._state(), free_gb=0.0, needed_gb=1.0)
assert plan["possible"] is False and plan["release"] == []
class TestConfigUpgrade:
def test_fields_added_later_are_merged_into_an_existing_config(self, cfg):
# A config written before needs_vram_gb existed must not silently lose the
# behaviour that field controls.
cfg.write_text(json.dumps([{
"name": "ollama",
"match": {"names": ["ollama"]},
}]))
t = T.get_tenant("ollama")
assert t.needs_vram_gb > 0
assert t.release.type == "http_post"
def test_explicit_user_values_still_win_over_defaults(self, cfg):
cfg.write_text(json.dumps([{
"name": "ollama", "priority": 5, "needs_vram_gb": 99.0,
"match": {"names": ["ollama"]},
}]))
t = T.get_tenant("ollama")
assert t.priority == 5 and t.needs_vram_gb == 99.0