The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.
engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.
The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.
Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.
Tests: 192 (was 182).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
134 lines
5.9 KiB
Python
134 lines
5.9 KiB
Python
"""Tests for live engine-configuration reporting.
|
|
|
|
These settings live outside this codebase but dictate how arbitration must behave, and
|
|
working out why a yield behaved a certain way once meant reading journald by hand. The
|
|
dashboard previously asserted them as hardcoded text, which happened to be accurate --
|
|
worse than being wrong, because it would have stayed accurate-looking after the settings
|
|
changed.
|
|
"""
|
|
import asyncio
|
|
|
|
import pytest
|
|
|
|
import engines
|
|
|
|
|
|
class _Proc:
|
|
def __init__(self, stdout=""):
|
|
self.stdout = stdout
|
|
self.returncode = 0
|
|
|
|
|
|
class TestOllamaEnvironmentParsing:
|
|
def test_parses_the_real_unit_environment(self, monkeypatch):
|
|
# Verbatim from `systemctl show ollama -p Environment --value` on this machine.
|
|
raw = ("OLLAMA_HOST=0.0.0.0:11434 OLLAMA_FLASH_ATTENTION=1 "
|
|
"OLLAMA_KV_CACHE_TYPE=q4_0 OLLAMA_KEEP_ALIVE=30m "
|
|
"OLLAMA_MAX_LOADED_MODELS=1 OLLAMA_NUM_PARALLEL=1 OLLAMA_NUM_BATCH=2048")
|
|
monkeypatch.setattr(engines.subprocess, "run", lambda *a, **k: _Proc(raw))
|
|
env = engines._ollama_unit_environment()
|
|
assert env["OLLAMA_NUM_PARALLEL"] == "1"
|
|
assert env["OLLAMA_MAX_LOADED_MODELS"] == "1"
|
|
assert env["OLLAMA_KV_CACHE_TYPE"] == "q4_0"
|
|
|
|
def test_ignores_non_ollama_variables(self, monkeypatch):
|
|
monkeypatch.setattr(engines.subprocess, "run",
|
|
lambda *a, **k: _Proc("PATH=/usr/bin OLLAMA_HOST=x:1 HOME=/root"))
|
|
env = engines._ollama_unit_environment()
|
|
assert set(env) == {"OLLAMA_HOST"}
|
|
|
|
def test_returns_empty_rather_than_raising_when_systemctl_fails(self, monkeypatch):
|
|
def boom(*a, **k):
|
|
raise FileNotFoundError("systemctl")
|
|
monkeypatch.setattr(engines.subprocess, "run", boom)
|
|
assert engines._ollama_unit_environment() == {}
|
|
|
|
|
|
class TestEngineConfigReport:
|
|
def _run(self, monkeypatch, env, comfy_ok=True):
|
|
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: env)
|
|
|
|
class _Resp:
|
|
status_code = 200 if comfy_ok else 500
|
|
def json(self):
|
|
return {"system": {"comfyui_version": "0.33.1",
|
|
"pytorch_version": "2.11.0+cu128",
|
|
"python_version": "3.14.4 (main)",
|
|
"argv": ["main.py", "--listen", "0.0.0.0"]},
|
|
"devices": [{"name": "cuda:0 NVIDIA GeForce RTX 4080 SUPER "
|
|
": cudaMallocAsync"}]}
|
|
|
|
class _Client:
|
|
async def __aenter__(self): return self
|
|
async def __aexit__(self, *a): return False
|
|
async def get(self, url): return _Resp()
|
|
|
|
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
|
return asyncio.run(engines.get_engine_config())
|
|
|
|
def test_surfaces_the_settings_that_drive_arbitration(self, monkeypatch):
|
|
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1",
|
|
"OLLAMA_MAX_LOADED_MODELS": "1",
|
|
"OLLAMA_KEEP_ALIVE": "30m"})
|
|
assert d["ollama"]["num_parallel"] == "1"
|
|
assert d["ollama"]["max_loaded_models"] == "1"
|
|
assert d["ollama"]["keep_alive"] == "30m"
|
|
|
|
def test_num_parallel_explains_the_deferred_yield_behaviour(self, monkeypatch):
|
|
d = self._run(monkeypatch, {"OLLAMA_NUM_PARALLEL": "1"})
|
|
note = next(s["means"] for s in d["ollama"]["settings"]
|
|
if s["key"] == "OLLAMA_NUM_PARALLEL")
|
|
# The explanation is the point: it is why a busy model is deferred, not failed.
|
|
assert "queue" in note.lower()
|
|
|
|
def test_summary_reflects_actual_flags_not_a_fixed_string(self, monkeypatch):
|
|
on = self._run(monkeypatch, {"OLLAMA_FLASH_ATTENTION": "1",
|
|
"OLLAMA_KV_CACHE_TYPE": "q4_0"})
|
|
assert "FlashAttention" in on["ollama"]["summary"]
|
|
assert "q4_0" in on["ollama"]["summary"]
|
|
off = self._run(monkeypatch, {})
|
|
assert "FlashAttention" not in off["ollama"]["summary"]
|
|
assert off["ollama"]["config_source"] == "unavailable"
|
|
|
|
def test_port_comes_from_ollama_host(self, monkeypatch):
|
|
d = self._run(monkeypatch, {"OLLAMA_HOST": "0.0.0.0:11500"})
|
|
assert d["ollama"]["port"] == "11500"
|
|
|
|
def test_comfy_allocator_and_vram_mode_are_read_not_asserted(self, monkeypatch):
|
|
d = self._run(monkeypatch, {})
|
|
assert d["comfyui"]["allocator"] == "cudaMallocAsync"
|
|
assert d["comfyui"]["vram_mode"] == "default (auto)"
|
|
assert d["comfyui"]["version"] == "0.33.1"
|
|
|
|
def test_comfy_vram_flag_is_detected_when_present(self, monkeypatch):
|
|
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
|
|
|
|
class _Resp:
|
|
status_code = 200
|
|
def json(self):
|
|
return {"system": {"argv": ["main.py", "--lowvram"]},
|
|
"devices": [{"name": "cuda:0 X : cudaMalloc"}]}
|
|
|
|
class _Client:
|
|
async def __aenter__(self): return self
|
|
async def __aexit__(self, *a): return False
|
|
async def get(self, url): return _Resp()
|
|
|
|
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
|
d = asyncio.run(engines.get_engine_config())
|
|
assert d["comfyui"]["vram_mode"] == "--lowvram"
|
|
assert d["comfyui"]["allocator"] == "cudaMalloc"
|
|
|
|
def test_offline_comfy_is_reported_not_raised(self, monkeypatch):
|
|
monkeypatch.setattr(engines, "_ollama_unit_environment", lambda: {})
|
|
|
|
class _Client:
|
|
async def __aenter__(self): return self
|
|
async def __aexit__(self, *a): return False
|
|
async def get(self, url): raise ConnectionError("refused")
|
|
|
|
monkeypatch.setattr(engines.httpx, "AsyncClient", lambda **k: _Client())
|
|
d = asyncio.run(engines.get_engine_config())
|
|
assert d["comfyui"]["online"] is False
|
|
assert "error" in d["comfyui"]
|