Measured failure: a diffusion job took 46 s instead of 3 s, squeezed into 1.6 GB. The LLM yielded correctly, an inference reloaded it two seconds later, and plan_release then refused to touch it because it was "busy" -- so ComfyUI crawled while Ollama held 13 GB for the whole run. "Never interrupt busy work" looks like the safe rule and is not. Preempting a lower-priority tenant is safe precisely because releasing is asynchronous: an Ollama unload queues behind its running request and applies when that finishes, so nothing is killed mid-flight. That is what makes fast handoff possible at all, and refusing to do it defeats the purpose of the service. A tenant may now be asked for memory if it is idle, whatever its rank, or if it is busy and ranks strictly below the demander. Equal or higher priority is never interrupted, so peers cannot fight. Idle tenants are still preferred over preempting busy ones. Verified against the real contention: with Ollama at 12.38 GB and 98% utilisation, a diffusion job released it within three seconds and completed in 18 s rather than 46 s. Tests updated to the corrected rule, and the fixture's priorities aligned with what actually ships -- it still had the LLM outranking diffusion from before that was swapped. Tests: 246. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
351 lines
16 KiB
Python
351 lines
16 KiB
Python
"""Tests for the GPU tenant registry.
|
|
|
|
The point of this service is fast handoff of one GPU between applications, and it should
|
|
work for any application -- not only the two it grew up around. Their names had ended up
|
|
compiled into process matching, VRAM attribution, busy detection and release calls alike.
|
|
These tests pin the properties that make the registry generic: adding an application is
|
|
configuration, and nothing in the arbitration logic knows a particular name.
|
|
"""
|
|
import asyncio
|
|
import json
|
|
|
|
import pytest
|
|
|
|
import tenants as T
|
|
|
|
|
|
@pytest.fixture
|
|
def cfg(tmp_path, monkeypatch):
|
|
path = tmp_path / "tenants.json"
|
|
monkeypatch.setattr(T, "CONFIG_PATH", str(path))
|
|
T._cache.update({"ts": 0.0, "tenants": None, "mtime": None})
|
|
return path
|
|
|
|
|
|
class TestProcessMatching:
|
|
def test_matches_by_process_name(self):
|
|
m = T.ProcessMatch(names=["ollama"])
|
|
assert m.matches("ollama", "/usr/bin/ollama serve")
|
|
assert not m.matches("python", "main.py")
|
|
|
|
def test_matches_by_cmdline_substring(self):
|
|
m = T.ProcessMatch(cmdline=["llama-server"])
|
|
assert m.matches("python", "/usr/local/lib/ollama/llama-server --model x")
|
|
|
|
def test_matches_by_cmdline_suffix(self):
|
|
# ComfyUI is a bare `python main.py`, with nothing else distinguishing it.
|
|
m = T.ProcessMatch(cmdline_endswith=["main.py"])
|
|
assert m.matches("python", "/opt/ComfyUI/venv/bin/python main.py")
|
|
assert not m.matches("python", "/opt/other/main.py --serve")
|
|
|
|
def test_matching_is_case_insensitive(self):
|
|
assert T.ProcessMatch(names=["Xorg"]).matches("XORG", "")
|
|
|
|
|
|
class TestDefaultsPreserveExistingBehaviour:
|
|
"""The shipped defaults must classify exactly as the hardcoded version did."""
|
|
|
|
@pytest.mark.parametrize("pname,cmdline,expected", [
|
|
("llama-server", "/usr/local/lib/ollama/llama-server --model x", "ollama"),
|
|
("ollama", "/usr/bin/ollama serve", "ollama"),
|
|
("python", "/home/u/ComfyUI/venv/bin/python main.py --listen", "comfyui"),
|
|
("gnome-shell", "/usr/bin/gnome-shell --mode=ubuntu", "desktop"),
|
|
("Xorg", "/usr/lib/xorg/Xorg :8", "desktop"),
|
|
("python", "/home/u/robopest-venv/bin/python /home/u/stt_relay.py", "unmanaged"),
|
|
("trainer", "/opt/ml/bin/trainer --epochs 3", "unmanaged"),
|
|
])
|
|
def test_classification(self, cfg, pname, cmdline, expected):
|
|
assert T.classify_process(pname, cmdline) == expected
|
|
|
|
def test_unknown_process_is_unmanaged_not_silently_owned(self, cfg):
|
|
# Misattributing a third party's VRAM to a tenant would make this service
|
|
# promise headroom it cannot deliver.
|
|
assert T.classify_process("weird", "/opt/x/weird --run") == "unmanaged"
|
|
|
|
|
|
class TestAddingAnApplicationIsConfiguration:
|
|
def test_a_new_tenant_is_recognised_without_code_changes(self, cfg):
|
|
cfg.write_text(json.dumps(T.DEFAULT_TENANTS + [{
|
|
"name": "trainer",
|
|
"kind": "other",
|
|
"priority": 80,
|
|
"match": {"cmdline": ["train.py"]},
|
|
"release": {"type": "http_post", "url": "http://localhost:9999/release"},
|
|
}]))
|
|
assert T.classify_process("python", "/opt/ml/train.py --epochs 3") == "trainer"
|
|
t = T.get_tenant("trainer")
|
|
assert t.priority == 80 and t.reclaimable
|
|
|
|
def test_first_run_writes_the_defaults(self, cfg):
|
|
assert not cfg.exists()
|
|
T.load_tenants(force=True)
|
|
assert cfg.exists()
|
|
assert {t["name"] for t in json.loads(cfg.read_text())} == {
|
|
"ollama", "comfyui", "desktop"}
|
|
|
|
def test_a_malformed_entry_is_skipped_not_fatal(self, cfg):
|
|
cfg.write_text(json.dumps([{"name": "ok", "match": {"names": ["a"]}},
|
|
{"no_name": True}]))
|
|
names = [t.name for t in T.load_tenants(force=True)]
|
|
assert names == ["ok"]
|
|
|
|
def test_corrupt_config_falls_back_to_defaults(self, cfg):
|
|
cfg.write_text("{ not json")
|
|
assert {t.name for t in T.load_tenants(force=True)} >= {"ollama", "comfyui"}
|
|
|
|
|
|
class TestReclaimability:
|
|
def test_a_tenant_with_no_release_strategy_is_not_reclaimable(self, cfg):
|
|
t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none"))
|
|
assert t.reclaimable is False
|
|
|
|
def test_release_refuses_rather_than_reporting_success(self, cfg):
|
|
t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none"))
|
|
res = asyncio.run(T.release_vram(t))
|
|
assert res["success"] is False and res["released"] is False
|
|
assert "no way to release" in res["reason"]
|
|
|
|
def test_per_model_release_with_nothing_loaded_is_a_no_op(self, cfg):
|
|
t = T.GpuTenant(name="ollama", release=T.ReleaseStrategy(
|
|
type="http_post", url="http://x/api", per_model=True))
|
|
res = asyncio.run(T.release_vram(t, models=[]))
|
|
assert res["success"] is True and res["released"] is False
|
|
|
|
|
|
class TestBusyProbe:
|
|
def _probe(self, monkeypatch, payload, status=200):
|
|
class _R:
|
|
status_code = status
|
|
def json(self_inner): return payload
|
|
class _C:
|
|
async def __aenter__(self): return self
|
|
async def __aexit__(self, *a): return False
|
|
async def get(self, url): return _R()
|
|
monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C())
|
|
|
|
def test_empty_queue_is_not_busy(self, monkeypatch):
|
|
self._probe(monkeypatch, {"queue_running": [], "queue_pending": []})
|
|
t = T.GpuTenant(name="c", busy=T.BusyProbe(
|
|
type="http_count", url="http://x/queue",
|
|
count_keys=["queue_running", "queue_pending"]))
|
|
assert asyncio.run(T.probe_busy(t))["busy"] is False
|
|
|
|
def test_queued_work_while_holding_no_vram_is_flagged_below_floor(self, monkeypatch):
|
|
# ComfyUI leaves dead jobs in queue_running; only its VRAM reveals that nothing
|
|
# is loaded.
|
|
self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []})
|
|
t = T.GpuTenant(name="c", busy=T.BusyProbe(
|
|
type="http_count", url="http://x/queue", count_keys=["queue_running"],
|
|
vram_floor_gb=1.5))
|
|
res = asyncio.run(T.probe_busy(t, vram_gb=0.56))
|
|
assert res["busy"] is True and res.get("below_floor") is True
|
|
|
|
def test_queued_work_with_a_checkpoint_loaded_is_plainly_busy(self, monkeypatch):
|
|
self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []})
|
|
t = T.GpuTenant(name="c", busy=T.BusyProbe(
|
|
type="http_count", url="http://x/queue", count_keys=["queue_running"],
|
|
vram_floor_gb=1.5))
|
|
res = asyncio.run(T.probe_busy(t, vram_gb=6.8))
|
|
assert res["busy"] is True and not res.get("below_floor")
|
|
|
|
def test_vram_probe_needs_no_http_endpoint(self):
|
|
# An application with no API can still be observed by what it holds.
|
|
t = T.GpuTenant(name="x", busy=T.BusyProbe(type="vram", vram_busy_gb=1.0))
|
|
assert asyncio.run(T.probe_busy(t, vram_gb=2.0))["busy"] is True
|
|
assert asyncio.run(T.probe_busy(t, vram_gb=0.5))["busy"] is False
|
|
|
|
def test_an_unreachable_probe_reports_not_busy_rather_than_raising(self, monkeypatch):
|
|
class _C:
|
|
async def __aenter__(self): return self
|
|
async def __aexit__(self, *a): return False
|
|
async def get(self, url): raise ConnectionError("refused")
|
|
monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C())
|
|
t = T.GpuTenant(name="c", busy=T.BusyProbe(type="http_count", url="http://x",
|
|
count_keys=["q"]))
|
|
res = asyncio.run(T.probe_busy(t))
|
|
assert res["busy"] is False and "failed" in res["reason"]
|
|
|
|
|
|
class TestPriority:
|
|
def test_describe_orders_by_priority(self, cfg):
|
|
rows = T.describe()
|
|
prios = [r["priority"] for r in rows]
|
|
assert prios == sorted(prios, reverse=True)
|
|
assert all("reclaimable" in r for r in rows)
|
|
|
|
|
|
class TestReleasePlanning:
|
|
"""Deciding who gives up VRAM, generically over any number of applications.
|
|
|
|
The two-application version was a pair of hardcoded rules -- yield Ollama when
|
|
ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express a
|
|
third participant at all.
|
|
"""
|
|
|
|
def _state(self, **overrides):
|
|
base = [
|
|
{"name": "desktop", "priority": 90, "vram_gb": 0.01, "busy": False,
|
|
"reclaimable": False},
|
|
{"name": "stt-relay", "priority": 70, "vram_gb": 0.8, "busy": False,
|
|
"reclaimable": False},
|
|
# Shipped priorities: diffusion outranks the LLM, whose weights reload
|
|
# from page cache in seconds.
|
|
{"name": "comfyui", "priority": 60, "vram_gb": 7.0, "busy": False,
|
|
"reclaimable": True},
|
|
{"name": "ollama", "priority": 50, "vram_gb": 0.0, "busy": True,
|
|
"reclaimable": True},
|
|
]
|
|
for s in base:
|
|
s.update(overrides.get(s["name"], {}))
|
|
return base
|
|
|
|
def test_a_tenant_already_holding_what_it_needs_is_not_starved(self):
|
|
# A busy GPU has little free by definition. Comparing free VRAM alone flagged a
|
|
# tenant working fine on 13 GB as demanding, which would have caused pointless
|
|
# releases from everything else.
|
|
state = self._state(ollama={"vram_gb": 13.0})
|
|
plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=4.0)
|
|
assert plan["release"] == []
|
|
assert "already free" in plan["reason"]
|
|
|
|
def test_starved_tenant_reclaims_from_the_idle_one_below_it(self):
|
|
plan = T.plan_release("ollama", self._state(), free_gb=1.5, needed_gb=14.9)
|
|
assert plan["release"] == ["comfyui"]
|
|
|
|
def test_a_busy_tenant_ranking_above_the_demander_is_not_a_victim(self):
|
|
# comfyui outranks ollama, so ollama may not interrupt it.
|
|
state = self._state(comfyui={"busy": True})
|
|
plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=14.9)
|
|
assert plan["release"] == []
|
|
assert any(b["name"] == "comfyui" and "busy" in b["why"]
|
|
for b in plan["blockers"])
|
|
|
|
def test_a_higher_priority_demander_preempts_busy_lower_priority_work(self):
|
|
"""The measured regression that made this rule necessary.
|
|
|
|
Refusing to touch anything busy looks safe and is not. With the LLM protected as
|
|
"busy", a diffusion job ran 46 s instead of 3 s, squeezed into 1.6 GB, because
|
|
the LLM reloaded immediately after yielding and was then untouchable. Preempting
|
|
a lower-priority tenant is safe because releasing is asynchronous: an Ollama
|
|
unload queues behind its running request rather than killing it.
|
|
"""
|
|
state = [
|
|
{"name": "comfyui", "priority": 60, "vram_gb": 1.65, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "ollama", "priority": 50, "vram_gb": 13.03, "busy": True,
|
|
"reclaimable": True},
|
|
]
|
|
plan = T.plan_release("comfyui", state, free_gb=0.28, needed_gb=6.0)
|
|
assert plan["release"] == ["ollama"]
|
|
|
|
def test_an_idle_tenant_is_preferred_over_preempting_a_busy_one(self):
|
|
state = [
|
|
{"name": "d", "priority": 60, "vram_gb": 0.0, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "busy_low", "priority": 10, "vram_gb": 8.0, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "idle_high", "priority": 90, "vram_gb": 8.0, "busy": False,
|
|
"reclaimable": True},
|
|
]
|
|
plan = T.plan_release("d", state, free_gb=0.0, needed_gb=8.0)
|
|
assert plan["release"] == ["idle_high"]
|
|
|
|
def test_unreclaimable_tenants_are_named_as_blockers_not_ignored(self):
|
|
# The user needs to know a third-party process is what stands in the way.
|
|
plan = T.plan_release("ollama", self._state(), free_gb=0.0, needed_gb=15.5)
|
|
blockers = {b["name"]: b["why"] for b in plan["blockers"]}
|
|
assert blockers["stt-relay"] == "declares no release mechanism"
|
|
assert plan["possible"] is False
|
|
|
|
def test_an_idle_tenant_yields_even_if_it_outranks_the_demander(self):
|
|
"""Priority orders who is asked first; it does not protect idle memory.
|
|
|
|
Filtering candidates by priority broke both directions in turn: with the LLM
|
|
ranked above diffusion, ComfyUI could never preempt Ollama -- the service's
|
|
central behaviour -- and once the ranks were swapped, a starved Ollama could no
|
|
longer reclaim from an idle ComfyUI. An idle tenant is not using its VRAM, so
|
|
outranking the demander is not a reason to keep it.
|
|
"""
|
|
state = self._state(comfyui={"priority": 99, "busy": False})
|
|
plan = T.plan_release("ollama", state, free_gb=1.0, needed_gb=14.9)
|
|
assert plan["release"] == ["comfyui"]
|
|
|
|
def test_priority_decides_who_is_asked_first(self):
|
|
state = [
|
|
{"name": "demander", "priority": 50, "vram_gb": 0.0, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "high", "priority": 90, "vram_gb": 4.0, "busy": False,
|
|
"reclaimable": True},
|
|
{"name": "low", "priority": 10, "vram_gb": 4.0, "busy": False,
|
|
"reclaimable": True},
|
|
]
|
|
plan = T.plan_release("demander", state, free_gb=0.0, needed_gb=5.0)
|
|
# The lowest-priority idle tenant gives up memory first.
|
|
assert plan["release"][0] == "low"
|
|
|
|
def test_peers_cannot_interrupt_each_other(self):
|
|
# Equal priority is never preempted, so two tenants at the same rank cannot
|
|
# fight over the card.
|
|
state = [
|
|
{"name": "a", "priority": 50, "vram_gb": 0.0, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "b", "priority": 50, "vram_gb": 8.0, "busy": True,
|
|
"reclaimable": True},
|
|
]
|
|
plan = T.plan_release("a", state, free_gb=0.0, needed_gb=8.0)
|
|
assert plan["release"] == []
|
|
assert "busy" in plan["blockers"][0]["why"]
|
|
|
|
def test_lowest_priority_is_released_first(self):
|
|
state = self._state() + [
|
|
{"name": "batch", "priority": 10, "vram_gb": 3.0, "busy": False,
|
|
"reclaimable": True}]
|
|
plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=5.0)
|
|
assert plan["release"][0] == "batch"
|
|
|
|
def test_releases_only_as_many_tenants_as_needed(self):
|
|
state = self._state() + [
|
|
{"name": "batch", "priority": 10, "vram_gb": 9.0, "busy": False,
|
|
"reclaimable": True}]
|
|
plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=8.0)
|
|
assert plan["release"] == ["batch"] # 9 GB covers it; comfyui is left alone
|
|
|
|
def test_three_applications_can_all_participate(self):
|
|
# The property the hardcoded pair of rules could not express.
|
|
state = [
|
|
{"name": "llm", "priority": 60, "vram_gb": 0.0, "busy": True,
|
|
"reclaimable": True},
|
|
{"name": "diffusion", "priority": 50, "vram_gb": 4.0, "busy": False,
|
|
"reclaimable": True},
|
|
{"name": "trainer", "priority": 40, "vram_gb": 5.0, "busy": False,
|
|
"reclaimable": True},
|
|
]
|
|
plan = T.plan_release("llm", state, free_gb=0.0, needed_gb=9.0)
|
|
assert set(plan["release"]) == {"trainer", "diffusion"}
|
|
assert plan["possible"] is True
|
|
|
|
def test_unknown_tenant_is_rejected_cleanly(self):
|
|
plan = T.plan_release("nope", self._state(), free_gb=0.0, needed_gb=1.0)
|
|
assert plan["possible"] is False and plan["release"] == []
|
|
|
|
|
|
class TestConfigUpgrade:
|
|
def test_fields_added_later_are_merged_into_an_existing_config(self, cfg):
|
|
# A config written before needs_vram_gb existed must not silently lose the
|
|
# behaviour that field controls.
|
|
cfg.write_text(json.dumps([{
|
|
"name": "ollama",
|
|
"match": {"names": ["ollama"]},
|
|
}]))
|
|
t = T.get_tenant("ollama")
|
|
assert t.needs_vram_gb > 0
|
|
assert t.release.type == "http_post"
|
|
|
|
def test_explicit_user_values_still_win_over_defaults(self, cfg):
|
|
cfg.write_text(json.dumps([{
|
|
"name": "ollama", "priority": 5, "needs_vram_gb": 99.0,
|
|
"match": {"names": ["ollama"]},
|
|
}]))
|
|
t = T.get_tenant("ollama")
|
|
assert t.priority == 5 and t.needs_vram_gb == 99.0
|