"""Tests for the GPU tenant registry. The point of this service is fast handoff of one GPU between applications, and it should work for any application -- not only the two it grew up around. Their names had ended up compiled into process matching, VRAM attribution, busy detection and release calls alike. These tests pin the properties that make the registry generic: adding an application is configuration, and nothing in the arbitration logic knows a particular name. """ import asyncio import json import pytest import tenants as T @pytest.fixture def cfg(tmp_path, monkeypatch): path = tmp_path / "tenants.json" monkeypatch.setattr(T, "CONFIG_PATH", str(path)) T._cache.update({"ts": 0.0, "tenants": None, "mtime": None}) return path class TestProcessMatching: def test_matches_by_process_name(self): m = T.ProcessMatch(names=["ollama"]) assert m.matches("ollama", "/usr/bin/ollama serve") assert not m.matches("python", "main.py") def test_matches_by_cmdline_substring(self): m = T.ProcessMatch(cmdline=["llama-server"]) assert m.matches("python", "/usr/local/lib/ollama/llama-server --model x") def test_matches_by_cmdline_suffix(self): # ComfyUI is a bare `python main.py`, with nothing else distinguishing it. m = T.ProcessMatch(cmdline_endswith=["main.py"]) assert m.matches("python", "/opt/ComfyUI/venv/bin/python main.py") assert not m.matches("python", "/opt/other/main.py --serve") def test_matching_is_case_insensitive(self): assert T.ProcessMatch(names=["Xorg"]).matches("XORG", "") class TestDefaultsPreserveExistingBehaviour: """The shipped defaults must classify exactly as the hardcoded version did.""" @pytest.mark.parametrize("pname,cmdline,expected", [ ("llama-server", "/usr/local/lib/ollama/llama-server --model x", "ollama"), ("ollama", "/usr/bin/ollama serve", "ollama"), ("python", "/home/u/ComfyUI/venv/bin/python main.py --listen", "comfyui"), ("gnome-shell", "/usr/bin/gnome-shell --mode=ubuntu", "desktop"), ("Xorg", "/usr/lib/xorg/Xorg :8", "desktop"), ("python", "/home/u/robopest-venv/bin/python /home/u/stt_relay.py", "unmanaged"), ("trainer", "/opt/ml/bin/trainer --epochs 3", "unmanaged"), ]) def test_classification(self, cfg, pname, cmdline, expected): assert T.classify_process(pname, cmdline) == expected def test_unknown_process_is_unmanaged_not_silently_owned(self, cfg): # Misattributing a third party's VRAM to a tenant would make this service # promise headroom it cannot deliver. assert T.classify_process("weird", "/opt/x/weird --run") == "unmanaged" class TestAddingAnApplicationIsConfiguration: def test_a_new_tenant_is_recognised_without_code_changes(self, cfg): cfg.write_text(json.dumps(T.DEFAULT_TENANTS + [{ "name": "trainer", "kind": "other", "priority": 80, "match": {"cmdline": ["train.py"]}, "release": {"type": "http_post", "url": "http://localhost:9999/release"}, }])) assert T.classify_process("python", "/opt/ml/train.py --epochs 3") == "trainer" t = T.get_tenant("trainer") assert t.priority == 80 and t.reclaimable def test_first_run_writes_the_defaults(self, cfg): assert not cfg.exists() T.load_tenants(force=True) assert cfg.exists() assert {t["name"] for t in json.loads(cfg.read_text())} == { "ollama", "comfyui", "desktop"} def test_a_malformed_entry_is_skipped_not_fatal(self, cfg): cfg.write_text(json.dumps([{"name": "ok", "match": {"names": ["a"]}}, {"no_name": True}])) names = [t.name for t in T.load_tenants(force=True)] assert names == ["ok"] def test_corrupt_config_falls_back_to_defaults(self, cfg): cfg.write_text("{ not json") assert {t.name for t in T.load_tenants(force=True)} >= {"ollama", "comfyui"} class TestReclaimability: def test_a_tenant_with_no_release_strategy_is_not_reclaimable(self, cfg): t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none")) assert t.reclaimable is False def test_release_refuses_rather_than_reporting_success(self, cfg): t = T.GpuTenant(name="x", release=T.ReleaseStrategy(type="none")) res = asyncio.run(T.release_vram(t)) assert res["success"] is False and res["released"] is False assert "no way to release" in res["reason"] def test_per_model_release_with_nothing_loaded_is_a_no_op(self, cfg): t = T.GpuTenant(name="ollama", release=T.ReleaseStrategy( type="http_post", url="http://x/api", per_model=True)) res = asyncio.run(T.release_vram(t, models=[])) assert res["success"] is True and res["released"] is False class TestBusyProbe: def _probe(self, monkeypatch, payload, status=200): class _R: status_code = status def json(self_inner): return payload class _C: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def get(self, url): return _R() monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C()) def test_empty_queue_is_not_busy(self, monkeypatch): self._probe(monkeypatch, {"queue_running": [], "queue_pending": []}) t = T.GpuTenant(name="c", busy=T.BusyProbe( type="http_count", url="http://x/queue", count_keys=["queue_running", "queue_pending"])) assert asyncio.run(T.probe_busy(t))["busy"] is False def test_queued_work_while_holding_no_vram_is_flagged_below_floor(self, monkeypatch): # ComfyUI leaves dead jobs in queue_running; only its VRAM reveals that nothing # is loaded. self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []}) t = T.GpuTenant(name="c", busy=T.BusyProbe( type="http_count", url="http://x/queue", count_keys=["queue_running"], vram_floor_gb=1.5)) res = asyncio.run(T.probe_busy(t, vram_gb=0.56)) assert res["busy"] is True and res.get("below_floor") is True def test_queued_work_with_a_checkpoint_loaded_is_plainly_busy(self, monkeypatch): self._probe(monkeypatch, {"queue_running": [[1, "abc"]], "queue_pending": []}) t = T.GpuTenant(name="c", busy=T.BusyProbe( type="http_count", url="http://x/queue", count_keys=["queue_running"], vram_floor_gb=1.5)) res = asyncio.run(T.probe_busy(t, vram_gb=6.8)) assert res["busy"] is True and not res.get("below_floor") def test_vram_probe_needs_no_http_endpoint(self): # An application with no API can still be observed by what it holds. t = T.GpuTenant(name="x", busy=T.BusyProbe(type="vram", vram_busy_gb=1.0)) assert asyncio.run(T.probe_busy(t, vram_gb=2.0))["busy"] is True assert asyncio.run(T.probe_busy(t, vram_gb=0.5))["busy"] is False def test_an_unreachable_probe_reports_not_busy_rather_than_raising(self, monkeypatch): class _C: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def get(self, url): raise ConnectionError("refused") monkeypatch.setattr(T.httpx, "AsyncClient", lambda **k: _C()) t = T.GpuTenant(name="c", busy=T.BusyProbe(type="http_count", url="http://x", count_keys=["q"])) res = asyncio.run(T.probe_busy(t)) assert res["busy"] is False and "failed" in res["reason"] class TestPriority: def test_describe_orders_by_priority(self, cfg): rows = T.describe() prios = [r["priority"] for r in rows] assert prios == sorted(prios, reverse=True) assert all("reclaimable" in r for r in rows) class TestReleasePlanning: """Deciding who gives up VRAM, generically over any number of applications. The two-application version was a pair of hardcoded rules -- yield Ollama when ComfyUI is busy, purge ComfyUI when Ollama is starved -- which could not express a third participant at all. """ def _state(self, **overrides): base = [ {"name": "desktop", "priority": 90, "vram_gb": 0.01, "busy": False, "reclaimable": False}, {"name": "stt-relay", "priority": 70, "vram_gb": 0.8, "busy": False, "reclaimable": False}, # Shipped priorities: diffusion outranks the LLM, whose weights reload # from page cache in seconds. {"name": "comfyui", "priority": 60, "vram_gb": 7.0, "busy": False, "reclaimable": True}, {"name": "ollama", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True}, ] for s in base: s.update(overrides.get(s["name"], {})) return base def test_a_tenant_already_holding_what_it_needs_is_not_starved(self): # A busy GPU has little free by definition. Comparing free VRAM alone flagged a # tenant working fine on 13 GB as demanding, which would have caused pointless # releases from everything else. state = self._state(ollama={"vram_gb": 13.0}) plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=4.0) assert plan["release"] == [] assert "already free" in plan["reason"] def test_starved_tenant_reclaims_from_the_idle_one_below_it(self): plan = T.plan_release("ollama", self._state(), free_gb=1.5, needed_gb=14.9) assert plan["release"] == ["comfyui"] def test_a_busy_tenant_ranking_above_the_demander_is_not_a_victim(self): # comfyui outranks ollama, so ollama may not interrupt it. state = self._state(comfyui={"busy": True}) plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=14.9) assert plan["release"] == [] assert any(b["name"] == "comfyui" and "busy" in b["why"] for b in plan["blockers"]) def test_a_higher_priority_demander_preempts_busy_lower_priority_work(self): """The measured regression that made this rule necessary. Refusing to touch anything busy looks safe and is not. With the LLM protected as "busy", a diffusion job ran 46 s instead of 3 s, squeezed into 1.6 GB, because the LLM reloaded immediately after yielding and was then untouchable. Preempting a lower-priority tenant is safe because releasing is asynchronous: an Ollama unload queues behind its running request rather than killing it. """ state = [ {"name": "comfyui", "priority": 60, "vram_gb": 1.65, "busy": True, "reclaimable": True}, {"name": "ollama", "priority": 50, "vram_gb": 13.03, "busy": True, "reclaimable": True}, ] plan = T.plan_release("comfyui", state, free_gb=0.28, needed_gb=6.0) assert plan["release"] == ["ollama"] def test_an_idle_tenant_is_preferred_over_preempting_a_busy_one(self): state = [ {"name": "d", "priority": 60, "vram_gb": 0.0, "busy": True, "reclaimable": True}, {"name": "busy_low", "priority": 10, "vram_gb": 8.0, "busy": True, "reclaimable": True}, {"name": "idle_high", "priority": 90, "vram_gb": 8.0, "busy": False, "reclaimable": True}, ] plan = T.plan_release("d", state, free_gb=0.0, needed_gb=8.0) assert plan["release"] == ["idle_high"] def test_unreclaimable_tenants_are_named_as_blockers_not_ignored(self): # The user needs to know a third-party process is what stands in the way. plan = T.plan_release("ollama", self._state(), free_gb=0.0, needed_gb=15.5) blockers = {b["name"]: b["why"] for b in plan["blockers"]} assert blockers["stt-relay"] == "declares no release mechanism" assert plan["possible"] is False def test_an_idle_tenant_yields_even_if_it_outranks_the_demander(self): """Priority orders who is asked first; it does not protect idle memory. Filtering candidates by priority broke both directions in turn: with the LLM ranked above diffusion, ComfyUI could never preempt Ollama -- the service's central behaviour -- and once the ranks were swapped, a starved Ollama could no longer reclaim from an idle ComfyUI. An idle tenant is not using its VRAM, so outranking the demander is not a reason to keep it. """ state = self._state(comfyui={"priority": 99, "busy": False}) plan = T.plan_release("ollama", state, free_gb=1.0, needed_gb=14.9) assert plan["release"] == ["comfyui"] def test_priority_decides_who_is_asked_first(self): state = [ {"name": "demander", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True}, {"name": "high", "priority": 90, "vram_gb": 4.0, "busy": False, "reclaimable": True}, {"name": "low", "priority": 10, "vram_gb": 4.0, "busy": False, "reclaimable": True}, ] plan = T.plan_release("demander", state, free_gb=0.0, needed_gb=5.0) # The lowest-priority idle tenant gives up memory first. assert plan["release"][0] == "low" def test_peers_cannot_interrupt_each_other(self): # Equal priority is never preempted, so two tenants at the same rank cannot # fight over the card. state = [ {"name": "a", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True}, {"name": "b", "priority": 50, "vram_gb": 8.0, "busy": True, "reclaimable": True}, ] plan = T.plan_release("a", state, free_gb=0.0, needed_gb=8.0) assert plan["release"] == [] assert "busy" in plan["blockers"][0]["why"] def test_lowest_priority_is_released_first(self): state = self._state() + [ {"name": "batch", "priority": 10, "vram_gb": 3.0, "busy": False, "reclaimable": True}] plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=5.0) assert plan["release"][0] == "batch" def test_releases_only_as_many_tenants_as_needed(self): state = self._state() + [ {"name": "batch", "priority": 10, "vram_gb": 9.0, "busy": False, "reclaimable": True}] plan = T.plan_release("ollama", state, free_gb=0.0, needed_gb=8.0) assert plan["release"] == ["batch"] # 9 GB covers it; comfyui is left alone def test_three_applications_can_all_participate(self): # The property the hardcoded pair of rules could not express. state = [ {"name": "llm", "priority": 60, "vram_gb": 0.0, "busy": True, "reclaimable": True}, {"name": "diffusion", "priority": 50, "vram_gb": 4.0, "busy": False, "reclaimable": True}, {"name": "trainer", "priority": 40, "vram_gb": 5.0, "busy": False, "reclaimable": True}, ] plan = T.plan_release("llm", state, free_gb=0.0, needed_gb=9.0) assert set(plan["release"]) == {"trainer", "diffusion"} assert plan["possible"] is True def test_unknown_tenant_is_rejected_cleanly(self): plan = T.plan_release("nope", self._state(), free_gb=0.0, needed_gb=1.0) assert plan["possible"] is False and plan["release"] == [] class TestConfigUpgrade: def test_fields_added_later_are_merged_into_an_existing_config(self, cfg): # A config written before needs_vram_gb existed must not silently lose the # behaviour that field controls. cfg.write_text(json.dumps([{ "name": "ollama", "match": {"names": ["ollama"]}, }])) t = T.get_tenant("ollama") assert t.needs_vram_gb > 0 assert t.release.type == "http_post" def test_explicit_user_values_still_win_over_defaults(self, cfg): cfg.write_text(json.dumps([{ "name": "ollama", "priority": 5, "needs_vram_gb": 99.0, "match": {"names": ["ollama"]}, }])) t = T.get_tenant("ollama") assert t.priority == 5 and t.needs_vram_gb == 99.0 class TestVramFloor: """VRAM that survives a release must not be promised to anyone else. ComfyUI keeps its CUDA context for as long as the process lives, so a purge does not return everything it holds. Ignoring that made plan_release report it would free 0.37 GB against a 0.33 GB shortfall; the job was cleared to run and the memory never arrived, so it waited two minutes and then failed. """ def test_only_memory_above_the_floor_counts_as_freeable(self): state = [ {"name": "llm", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True, "vram_floor_gb": 0.0}, {"name": "diffusion", "priority": 60, "vram_gb": 0.44, "busy": False, "reclaimable": True, "vram_floor_gb": 0.45}, ] plan = T.plan_release("llm", state, free_gb=14.6, needed_gb=14.93) assert plan["possible"] is False assert plan["release"] == [] def test_a_loaded_checkpoint_is_still_freeable_above_its_floor(self): state = [ {"name": "llm", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True, "vram_floor_gb": 0.0}, {"name": "diffusion", "priority": 60, "vram_gb": 7.0, "busy": False, "reclaimable": True, "vram_floor_gb": 0.45}, ] plan = T.plan_release("llm", state, free_gb=7.9, needed_gb=14.0) assert plan["release"] == ["diffusion"] # 7.0 held minus a 0.45 floor. assert abs(plan["would_free_gb"] - 6.55) < 0.01 def test_a_tenant_at_its_floor_is_not_even_listed_for_release(self): state = [ {"name": "llm", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True, "vram_floor_gb": 0.0}, {"name": "at_floor", "priority": 10, "vram_gb": 0.3, "busy": False, "reclaimable": True, "vram_floor_gb": 0.45}, {"name": "has_room", "priority": 20, "vram_gb": 5.0, "busy": False, "reclaimable": True, "vram_floor_gb": 0.0}, ] plan = T.plan_release("llm", state, free_gb=0.0, needed_gb=4.0) assert plan["release"] == ["has_room"] def test_default_floor_is_zero_so_existing_configs_are_unchanged(self): state = [ {"name": "a", "priority": 50, "vram_gb": 0.0, "busy": True, "reclaimable": True}, {"name": "b", "priority": 40, "vram_gb": 5.0, "busy": False, "reclaimable": True}, ] plan = T.plan_release("a", state, free_gb=0.0, needed_gb=5.0) assert plan["release"] == ["b"] and plan["possible"] is True