Read engine configuration live instead of asserting it in the dashboard

The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.

engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.

The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.

Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.

Tests: 192 (was 182).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-06 10:56:04 -07:00
parent b53b026ced
commit 043d61722b
7 changed files with 315 additions and 3 deletions

133
tests/test_engines.py Normal file
View File

@@ -0,0 +1,133 @@
"""Tests for live engine-configuration reporting.
These settings live outside this codebase but dictate how arbitration must behave, and
working out why a yield behaved a certain way once meant reading journald by hand. The
dashboard previously asserted them as hardcoded text, which happened to be accurate --
worse than being wrong, because it would have stayed accurate-looking after the settings
changed.
"""
import asyncio
import pytest
import engines
class _Proc:
def __init__(self, stdout=""):
self.stdout = stdout
self.returncode = 0
class TestOllamaEnvironmentParsing:
def test_parses_the_real_unit_environment(self, monkeypatch):
# Verbatim from `systemctl show ollama -p Environment --value` on this machine.
raw = ("OLLAMA_HOST=0.0.0.0:11434 OLLAMA_FLASH_ATTENTION=1 "
"OLLAMA_KV_CACHE_TYPE=q4_0 OLLAMA_KEEP_ALIVE=30m "
"OLLAMA_MAX_LOADED_MODELS=1 OLLAMA_NUM_PARALLEL=1 OLLAMA_NUM_BATCH=2048")
monkeypatch.setattr(engines.subprocess, "run", lambda *a, **k: _Proc(raw))
env = engines._ollama_unit_environment()
assert env["OLLAMA_NUM_PARALLEL"] == "1"
assert env["OLLAMA_MAX_LOADED_MODELS"] == "1"
assert env["OLLAMA_KV_CACHE_TYPE"] == "q4_0"
def test_ignores_non_ollama_variables(self, monkeypatch):
monkeypatch.setattr(engines.subprocess, "run",
lambda *a, **k: _Proc("PATH=/usr/bin OLLAMA_HOST=x:1 HOME=/root"))
env = engines._ollama_unit_environment()
assert set(env) == {"OLLAMA_HOST"}
def test_returns_empty_rather_than_raising_when_systemctl_fails(self, monkeypatch):
def boom(*a, **k):
raise FileNotFoundError("systemctl")
monkeypatch.setattr(engines.subprocess, "run", boom)
assert engines._ollama_unit_environment() == {}
class TestEngineConfigReport:
def _run(self, monkeypatch, env, comfy_ok=True):
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: env)
class _Resp:
status_code = 200 if comfy_ok else 500
def json(self):
return {"system": {"comfyui_version": "0.33.1",
"pytorch_version": "2.11.0+cu128",
"python_version": "3.14.4 (main)",
"argv": ["main.py", "--listen", "0.0.0.0"]},
"devices": [{"name": "cuda:0 NVIDIA GeForce RTX 4080 SUPER "
": cudaMallocAsync"}]}
class _Client:
async def __aenter__(self): return self
async def __aexit__(self, *a): return False
async def get(self, url): return _Resp()
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
return asyncio.run(engines.get_engine_config())
def test_surfaces_the_settings_that_drive_arbitration(self, monkeypatch):
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1",
"OLLAMA_MAX_LOADED_MODELS": "1",
"OLLAMA_KEEP_ALIVE": "30m"})
assert d["ollama"]["num_parallel"] == "1"
assert d["ollama"]["max_loaded_models"] == "1"
assert d["ollama"]["keep_alive"] == "30m"
def test_num_parallel_explains_the_deferred_yield_behaviour(self, monkeypatch):
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1"})
note = next(s["means"] for s in d["ollama"]["settings"]
if s["key"] == "OLLAMA_NUM_PARALLEL")
# The explanation is the point: it is why a busy model is deferred, not failed.
assert "queue" in note.lower()
def test_summary_reflects_actual_flags_not_a_fixed_string(self, monkeypatch):
on = self._run(monkeypatch, {"OLLAMA_FLASH_ATTENTION": "1",
"OLLAMA_KV_CACHE_TYPE": "q4_0"})
assert "FlashAttention" in on["ollama"]["summary"]
assert "q4_0" in on["ollama"]["summary"]
off = self._run(monkeypatch, {})
assert "FlashAttention" not in off["ollama"]["summary"]
assert off["ollama"]["config_source"] == "unavailable"
def test_port_comes_from_ollama_host(self, monkeypatch):
d = self._run(monkeypatch, {"OLLAMA_HOST": "0.0.0.0:11500"})
assert d["ollama"]["port"] == "11500"
def test_comfy_allocator_and_vram_mode_are_read_not_asserted(self, monkeypatch):
d = self._run(monkeypatch, {})
assert d["comfyui"]["allocator"] == "cudaMallocAsync"
assert d["comfyui"]["vram_mode"] == "default (auto)"
assert d["comfyui"]["version"] == "0.33.1"
def test_comfy_vram_flag_is_detected_when_present(self, monkeypatch):
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
class _Resp:
status_code = 200
def json(self):
return {"system": {"argv": ["main.py", "--lowvram"]},
"devices": [{"name": "cuda:0 X : cudaMalloc"}]}
class _Client:
async def __aenter__(self): return self
async def __aexit__(self, *a): return False
async def get(self, url): return _Resp()
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
d = asyncio.run(engines.get_engine_config())
assert d["comfyui"]["vram_mode"] == "--lowvram"
assert d["comfyui"]["allocator"] == "cudaMalloc"
def test_offline_comfy_is_reported_not_raised(self, monkeypatch):
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
class _Client:
async def __aenter__(self): return self
async def __aexit__(self, *a): return False
async def get(self, url): raise ConnectionError("refused")
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
d = asyncio.run(engines.get_engine_config())
assert d["comfyui"]["online"] is False
assert "error" in d["comfyui"]