Read engine configuration live instead of asserting it in the dashboard
The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.
engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.
The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.
Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.
Tests: 192 (was 182).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
133
tests/test_engines.py
Normal file
133
tests/test_engines.py
Normal file
@@ -0,0 +1,133 @@
|
||||
"""Tests for live engine-configuration reporting.
|
||||
|
||||
These settings live outside this codebase but dictate how arbitration must behave, and
|
||||
working out why a yield behaved a certain way once meant reading journald by hand. The
|
||||
dashboard previously asserted them as hardcoded text, which happened to be accurate --
|
||||
worse than being wrong, because it would have stayed accurate-looking after the settings
|
||||
changed.
|
||||
"""
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
import engines
|
||||
|
||||
|
||||
class _Proc:
|
||||
def __init__(self, stdout=""):
|
||||
self.stdout = stdout
|
||||
self.returncode = 0
|
||||
|
||||
|
||||
class TestOllamaEnvironmentParsing:
|
||||
def test_parses_the_real_unit_environment(self, monkeypatch):
|
||||
# Verbatim from `systemctl show ollama -p Environment --value` on this machine.
|
||||
raw = ("OLLAMA_HOST=0.0.0.0:11434 OLLAMA_FLASH_ATTENTION=1 "
|
||||
"OLLAMA_KV_CACHE_TYPE=q4_0 OLLAMA_KEEP_ALIVE=30m "
|
||||
"OLLAMA_MAX_LOADED_MODELS=1 OLLAMA_NUM_PARALLEL=1 OLLAMA_NUM_BATCH=2048")
|
||||
monkeypatch.setattr(engines.subprocess, "run", lambda *a, **k: _Proc(raw))
|
||||
env = engines._ollama_unit_environment()
|
||||
assert env["OLLAMA_NUM_PARALLEL"] == "1"
|
||||
assert env["OLLAMA_MAX_LOADED_MODELS"] == "1"
|
||||
assert env["OLLAMA_KV_CACHE_TYPE"] == "q4_0"
|
||||
|
||||
def test_ignores_non_ollama_variables(self, monkeypatch):
|
||||
monkeypatch.setattr(engines.subprocess, "run",
|
||||
lambda *a, **k: _Proc("PATH=/usr/bin OLLAMA_HOST=x:1 HOME=/root"))
|
||||
env = engines._ollama_unit_environment()
|
||||
assert set(env) == {"OLLAMA_HOST"}
|
||||
|
||||
def test_returns_empty_rather_than_raising_when_systemctl_fails(self, monkeypatch):
|
||||
def boom(*a, **k):
|
||||
raise FileNotFoundError("systemctl")
|
||||
monkeypatch.setattr(engines.subprocess, "run", boom)
|
||||
assert engines._ollama_unit_environment() == {}
|
||||
|
||||
|
||||
class TestEngineConfigReport:
|
||||
def _run(self, monkeypatch, env, comfy_ok=True):
|
||||
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: env)
|
||||
|
||||
class _Resp:
|
||||
status_code = 200 if comfy_ok else 500
|
||||
def json(self):
|
||||
return {"system": {"comfyui_version": "0.33.1",
|
||||
"pytorch_version": "2.11.0+cu128",
|
||||
"python_version": "3.14.4 (main)",
|
||||
"argv": ["main.py", "--listen", "0.0.0.0"]},
|
||||
"devices": [{"name": "cuda:0 NVIDIA GeForce RTX 4080 SUPER "
|
||||
": cudaMallocAsync"}]}
|
||||
|
||||
class _Client:
|
||||
async def __aenter__(self): return self
|
||||
async def __aexit__(self, *a): return False
|
||||
async def get(self, url): return _Resp()
|
||||
|
||||
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
||||
return asyncio.run(engines.get_engine_config())
|
||||
|
||||
def test_surfaces_the_settings_that_drive_arbitration(self, monkeypatch):
|
||||
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1",
|
||||
"OLLAMA_MAX_LOADED_MODELS": "1",
|
||||
"OLLAMA_KEEP_ALIVE": "30m"})
|
||||
assert d["ollama"]["num_parallel"] == "1"
|
||||
assert d["ollama"]["max_loaded_models"] == "1"
|
||||
assert d["ollama"]["keep_alive"] == "30m"
|
||||
|
||||
def test_num_parallel_explains_the_deferred_yield_behaviour(self, monkeypatch):
|
||||
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1"})
|
||||
note = next(s["means"] for s in d["ollama"]["settings"]
|
||||
if s["key"] == "OLLAMA_NUM_PARALLEL")
|
||||
# The explanation is the point: it is why a busy model is deferred, not failed.
|
||||
assert "queue" in note.lower()
|
||||
|
||||
def test_summary_reflects_actual_flags_not_a_fixed_string(self, monkeypatch):
|
||||
on = self._run(monkeypatch, {"OLLAMA_FLASH_ATTENTION": "1",
|
||||
"OLLAMA_KV_CACHE_TYPE": "q4_0"})
|
||||
assert "FlashAttention" in on["ollama"]["summary"]
|
||||
assert "q4_0" in on["ollama"]["summary"]
|
||||
off = self._run(monkeypatch, {})
|
||||
assert "FlashAttention" not in off["ollama"]["summary"]
|
||||
assert off["ollama"]["config_source"] == "unavailable"
|
||||
|
||||
def test_port_comes_from_ollama_host(self, monkeypatch):
|
||||
d = self._run(monkeypatch, {"OLLAMA_HOST": "0.0.0.0:11500"})
|
||||
assert d["ollama"]["port"] == "11500"
|
||||
|
||||
def test_comfy_allocator_and_vram_mode_are_read_not_asserted(self, monkeypatch):
|
||||
d = self._run(monkeypatch, {})
|
||||
assert d["comfyui"]["allocator"] == "cudaMallocAsync"
|
||||
assert d["comfyui"]["vram_mode"] == "default (auto)"
|
||||
assert d["comfyui"]["version"] == "0.33.1"
|
||||
|
||||
def test_comfy_vram_flag_is_detected_when_present(self, monkeypatch):
|
||||
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
|
||||
|
||||
class _Resp:
|
||||
status_code = 200
|
||||
def json(self):
|
||||
return {"system": {"argv": ["main.py", "--lowvram"]},
|
||||
"devices": [{"name": "cuda:0 X : cudaMalloc"}]}
|
||||
|
||||
class _Client:
|
||||
async def __aenter__(self): return self
|
||||
async def __aexit__(self, *a): return False
|
||||
async def get(self, url): return _Resp()
|
||||
|
||||
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
||||
d = asyncio.run(engines.get_engine_config())
|
||||
assert d["comfyui"]["vram_mode"] == "--lowvram"
|
||||
assert d["comfyui"]["allocator"] == "cudaMalloc"
|
||||
|
||||
def test_offline_comfy_is_reported_not_raised(self, monkeypatch):
|
||||
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
|
||||
|
||||
class _Client:
|
||||
async def __aenter__(self): return self
|
||||
async def __aexit__(self, *a): return False
|
||||
async def get(self, url): raise ConnectionError("refused")
|
||||
|
||||
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
||||
d = asyncio.run(engines.get_engine_config())
|
||||
assert d["comfyui"]["online"] is False
|
||||
assert "error" in d["comfyui"]
|
||||
Reference in New Issue
Block a user