Read engine configuration live instead of asserting it in the dashboard
The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.
engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.
The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.
Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.
Tests: 192 (was 182).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
117
engines.py
Normal file
117
engines.py
Normal file
@@ -0,0 +1,117 @@
|
||||
"""Live configuration of the two engines HyperSwap arbitrates between.
|
||||
|
||||
Arbitration behaviour is largely dictated by settings that live outside this codebase.
|
||||
Working out why a yield behaved the way it did meant reading journald and the ollama
|
||||
unit by hand: OLLAMA_NUM_PARALLEL decides whether an unload queues behind a running
|
||||
generation, OLLAMA_MAX_LOADED_MODELS decides whether more than one model can be
|
||||
resident, and a pinned n_gpu_layers decides whether a model that will not fit spills to
|
||||
the CPU or fails outright. Those are worth reading and explaining rather than hardcoding
|
||||
into a dashboard subtitle that silently goes stale.
|
||||
"""
|
||||
import json
|
||||
import logging
|
||||
import subprocess
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import httpx
|
||||
|
||||
import vram_arbitrator
|
||||
|
||||
logger = logging.getLogger("engines")
|
||||
|
||||
# Settings that change how the arbitrator must behave, with what they imply.
|
||||
OLLAMA_SETTING_NOTES = {
|
||||
"OLLAMA_NUM_PARALLEL": (
|
||||
"Requests per model. At 1, a keep_alive:0 unload queues behind any running "
|
||||
"generation and applies when it finishes — which is why a busy model is "
|
||||
"reported as deferred rather than failed."),
|
||||
"OLLAMA_MAX_LOADED_MODELS": (
|
||||
"How many models may be resident at once. At 1, Ollama evicts the previous "
|
||||
"model on every swap."),
|
||||
"OLLAMA_KEEP_ALIVE": (
|
||||
"Default residency after a request. Long values keep VRAM occupied and make "
|
||||
"ComfyUI wait for an explicit yield."),
|
||||
"OLLAMA_FLASH_ATTENTION": "FlashAttention kernels for attention.",
|
||||
"OLLAMA_KV_CACHE_TYPE": "KV cache quantisation; smaller types cut VRAM per context.",
|
||||
"OLLAMA_NUM_BATCH": "Prompt-evaluation batch size.",
|
||||
}
|
||||
|
||||
|
||||
def _ollama_unit_environment() -> Dict[str, str]:
|
||||
"""Read the ollama service's environment. Empty if it is not a systemd unit."""
|
||||
env: Dict[str, str] = {}
|
||||
try:
|
||||
proc = subprocess.run(["systemctl", "show", "ollama", "-p", "Environment",
|
||||
"--value"], capture_output=True, text=True, timeout=8)
|
||||
for token in proc.stdout.split():
|
||||
if "=" in token and token.startswith("OLLAMA"):
|
||||
k, _, v = token.partition("=")
|
||||
env[k] = v
|
||||
except Exception as e:
|
||||
logger.debug(f"could not read ollama unit environment: {e}")
|
||||
return env
|
||||
|
||||
|
||||
async def get_engine_config() -> Dict[str, Any]:
|
||||
"""Real, live configuration of both engines, with arbitration implications."""
|
||||
ollama_env = _ollama_unit_environment()
|
||||
ollama_settings = [
|
||||
{"key": k, "value": v, "means": OLLAMA_SETTING_NOTES.get(k, "")}
|
||||
for k, v in sorted(ollama_env.items())
|
||||
]
|
||||
|
||||
# A short, honest summary line to replace the dashboard's hardcoded subtitle.
|
||||
feature_bits: List[str] = []
|
||||
if ollama_env.get("OLLAMA_FLASH_ATTENTION") == "1":
|
||||
feature_bits.append("FlashAttention")
|
||||
kv = ollama_env.get("OLLAMA_KV_CACHE_TYPE")
|
||||
if kv:
|
||||
feature_bits.append(f"{kv} KV cache")
|
||||
host = ollama_env.get("OLLAMA_HOST", "")
|
||||
port = host.rsplit(":", 1)[-1] if ":" in host else "11434"
|
||||
|
||||
ollama = {
|
||||
"port": port,
|
||||
"settings": ollama_settings,
|
||||
"summary": " + ".join(feature_bits) if feature_bits else "default configuration",
|
||||
"max_loaded_models": ollama_env.get("OLLAMA_MAX_LOADED_MODELS"),
|
||||
"num_parallel": ollama_env.get("OLLAMA_NUM_PARALLEL"),
|
||||
"keep_alive": ollama_env.get("OLLAMA_KEEP_ALIVE"),
|
||||
"config_source": "systemd unit environment" if ollama_env else "unavailable",
|
||||
}
|
||||
|
||||
comfy: Dict[str, Any] = {"online": False}
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=4.0) as c:
|
||||
r = await c.get(f"{vram_arbitrator.COMFY_API_BASE}/system_stats")
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
system = data.get("system", {})
|
||||
argv = system.get("argv") or []
|
||||
devices = data.get("devices") or []
|
||||
dev = devices[0] if devices else {}
|
||||
# The allocator is named in the device string; it is the closest thing
|
||||
# ComfyUI reports to the "async offload" the old subtitle asserted.
|
||||
dev_name = dev.get("name", "")
|
||||
allocator = ("cudaMallocAsync" if "cudaMallocAsync" in dev_name
|
||||
else "cudaMalloc" if "cudaMalloc" in dev_name else "unknown")
|
||||
vram_flags = [a for a in argv
|
||||
if a in ("--lowvram", "--novram", "--highvram", "--normalvram",
|
||||
"--gpu-only", "--cpu")]
|
||||
comfy = {
|
||||
"online": True,
|
||||
"version": system.get("comfyui_version"),
|
||||
"pytorch": system.get("pytorch_version"),
|
||||
# split()[0] on an absent version raises IndexError, which would have
|
||||
# been swallowed by the except below and reported ComfyUI as offline.
|
||||
"python": ((system.get("python_version") or "").split() or [None])[0],
|
||||
"argv": argv,
|
||||
"vram_mode": vram_flags[0] if vram_flags else "default (auto)",
|
||||
"allocator": allocator,
|
||||
"device": dev_name,
|
||||
"summary": f"{allocator}, {vram_flags[0] if vram_flags else 'auto VRAM'}",
|
||||
}
|
||||
except Exception as e:
|
||||
comfy = {"online": False, "error": str(e)[:120]}
|
||||
|
||||
return {"ollama": ollama, "comfyui": comfy}
|
||||
Reference in New Issue
Block a user