"""Tests for live engine-configuration reporting. These settings live outside this codebase but dictate how arbitration must behave, and working out why a yield behaved a certain way once meant reading journald by hand. The dashboard previously asserted them as hardcoded text, which happened to be accurate -- worse than being wrong, because it would have stayed accurate-looking after the settings changed. """ import asyncio import pytest import engines class _Proc: def __init__(self, stdout=""): self.stdout = stdout self.returncode = 0 class TestOllamaEnvironmentParsing: def test_parses_the_real_unit_environment(self, monkeypatch): # Verbatim from `systemctl show ollama -p Environment --value` on this machine. raw = ("OLLAMA_HOST=0.0.0.0:11434 OLLAMA_FLASH_ATTENTION=1 " "OLLAMA_KV_CACHE_TYPE=q4_0 OLLAMA_KEEP_ALIVE=30m " "OLLAMA_MAX_LOADED_MODELS=1 OLLAMA_NUM_PARALLEL=1 OLLAMA_NUM_BATCH=2048") monkeypatch.setattr(engines.subprocess, "run", lambda *a, **k: _Proc(raw)) env = engines._ollama_unit_environment() assert env["OLLAMA_NUM_PARALLEL"] == "1" assert env["OLLAMA_MAX_LOADED_MODELS"] == "1" assert env["OLLAMA_KV_CACHE_TYPE"] == "q4_0" def test_ignores_non_ollama_variables(self, monkeypatch): monkeypatch.setattr(engines.subprocess, "run", lambda *a, **k: _Proc("PATH=/usr/bin OLLAMA_HOST=x:1 HOME=/root")) env = engines._ollama_unit_environment() assert set(env) == {"OLLAMA_HOST"} def test_returns_empty_rather_than_raising_when_systemctl_fails(self, monkeypatch): def boom(*a, **k): raise FileNotFoundError("systemctl") monkeypatch.setattr(engines.subprocess, "run", boom) assert engines._ollama_unit_environment() == {} class TestEngineConfigReport: def _run(self, monkeypatch, env, comfy_ok=True): monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: env) class _Resp: status_code = 200 if comfy_ok else 500 def json(self): return {"system": {"comfyui_version": "0.33.1", "pytorch_version": "2.11.0+cu128", "python_version": "3.14.4 (main)", "argv": ["main.py", "--listen", "0.0.0.0"]}, "devices": [{"name": "cuda:0 NVIDIA GeForce RTX 4080 SUPER " ": cudaMallocAsync"}]} class _Client: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def get(self, url): return _Resp() monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client()) return asyncio.run(engines.get_engine_config()) def test_surfaces_the_settings_that_drive_arbitration(self, monkeypatch): d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1", "OLLAMA_MAX_LOADED_MODELS": "1", "OLLAMA_KEEP_ALIVE": "30m"}) assert d["ollama"]["num_parallel"] == "1" assert d["ollama"]["max_loaded_models"] == "1" assert d["ollama"]["keep_alive"] == "30m" def test_num_parallel_explains_the_deferred_yield_behaviour(self, monkeypatch): d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1"}) note = next(s["means"] for s in d["ollama"]["settings"] if s["key"] == "OLLAMA_NUM_PARALLEL") # The explanation is the point: it is why a busy model is deferred, not failed. assert "queue" in note.lower() def test_summary_reflects_actual_flags_not_a_fixed_string(self, monkeypatch): on = self._run(monkeypatch, {"OLLAMA_FLASH_ATTENTION": "1", "OLLAMA_KV_CACHE_TYPE": "q4_0"}) assert "FlashAttention" in on["ollama"]["summary"] assert "q4_0" in on["ollama"]["summary"] off = self._run(monkeypatch, {}) assert "FlashAttention" not in off["ollama"]["summary"] assert off["ollama"]["config_source"] == "unavailable" def test_port_comes_from_ollama_host(self, monkeypatch): d = self._run(monkeypatch, {"OLLAMA_HOST": "0.0.0.0:11500"}) assert d["ollama"]["port"] == "11500" def test_comfy_allocator_and_vram_mode_are_read_not_asserted(self, monkeypatch): d = self._run(monkeypatch, {}) assert d["comfyui"]["allocator"] == "cudaMallocAsync" assert d["comfyui"]["vram_mode"] == "default (auto)" assert d["comfyui"]["version"] == "0.33.1" def test_comfy_vram_flag_is_detected_when_present(self, monkeypatch): monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {}) class _Resp: status_code = 200 def json(self): return {"system": {"argv": ["main.py", "--lowvram"]}, "devices": [{"name": "cuda:0 X : cudaMalloc"}]} class _Client: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def get(self, url): return _Resp() monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client()) d = asyncio.run(engines.get_engine_config()) assert d["comfyui"]["vram_mode"] == "--lowvram" assert d["comfyui"]["allocator"] == "cudaMalloc" def test_offline_comfy_is_reported_not_raised(self, monkeypatch): monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {}) class _Client: async def __aenter__(self): return self async def __aexit__(self, *a): return False async def get(self, url): raise ConnectionError("refused") monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client()) d = asyncio.run(engines.get_engine_config()) assert d["comfyui"]["online"] is False assert "error" in d["comfyui"]