Files
gpu-program-swapper/engines.py
drjones 043d61722b Read engine configuration live instead of asserting it in the dashboard
The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.

engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.

The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.

Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.

Tests: 192 (was 182).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 10:56:04 -07:00

118 lines
5.3 KiB
Python

"""Live configuration of the two engines HyperSwap arbitrates between.
Arbitration behaviour is largely dictated by settings that live outside this codebase.
Working out why a yield behaved the way it did meant reading journald and the ollama
unit by hand: OLLAMA_NUM_PARALLEL decides whether an unload queues behind a running
generation, OLLAMA_MAX_LOADED_MODELS decides whether more than one model can be
resident, and a pinned n_gpu_layers decides whether a model that will not fit spills to
the CPU or fails outright. Those are worth reading and explaining rather than hardcoding
into a dashboard subtitle that silently goes stale.
"""
import json
import logging
import subprocess
from typing import Any, Dict, List, Optional
import httpx
import vram_arbitrator
logger = logging.getLogger("engines")
# Settings that change how the arbitrator must behave, with what they imply.
OLLAMA_SETTING_NOTES = {
"OLLAMA_NUM_PARALLEL": (
"Requests per model. At 1, a keep_alive:0 unload queues behind any running "
"generation and applies when it finishes — which is why a busy model is "
"reported as deferred rather than failed."),
"OLLAMA_MAX_LOADED_MODELS": (
"How many models may be resident at once. At 1, Ollama evicts the previous "
"model on every swap."),
"OLLAMA_KEEP_ALIVE": (
"Default residency after a request. Long values keep VRAM occupied and make "
"ComfyUI wait for an explicit yield."),
"OLLAMA_FLASH_ATTENTION": "FlashAttention kernels for attention.",
"OLLAMA_KV_CACHE_TYPE": "KV cache quantisation; smaller types cut VRAM per context.",
"OLLAMA_NUM_BATCH": "Prompt-evaluation batch size.",
}
def _ollama_unit_environment() -> Dict[str, str]:
"""Read the ollama service's environment. Empty if it is not a systemd unit."""
env: Dict[str, str] = {}
try:
proc = subprocess.run(["systemctl", "show", "ollama", "-p", "Environment",
"--value"], capture_output=True, text=True, timeout=8)
for token in proc.stdout.split():
if "=" in token and token.startswith("OLLAMA"):
k, _, v = token.partition("=")
env[k] = v
except Exception as e:
logger.debug(f"could not read ollama unit environment: {e}")
return env
async def get_engine_config() -> Dict[str, Any]:
"""Real, live configuration of both engines, with arbitration implications."""
ollama_env = _ollama_unit_environment()
ollama_settings = [
{"key": k, "value": v, "means": OLLAMA_SETTING_NOTES.get(k, "")}
for k, v in sorted(ollama_env.items())
]
# A short, honest summary line to replace the dashboard's hardcoded subtitle.
feature_bits: List[str] = []
if ollama_env.get("OLLAMA_FLASH_ATTENTION") == "1":
feature_bits.append("FlashAttention")
kv = ollama_env.get("OLLAMA_KV_CACHE_TYPE")
if kv:
feature_bits.append(f"{kv} KV cache")
host = ollama_env.get("OLLAMA_HOST", "")
port = host.rsplit(":", 1)[-1] if ":" in host else "11434"
ollama = {
"port": port,
"settings": ollama_settings,
"summary": " + ".join(feature_bits) if feature_bits else "default configuration",
"max_loaded_models": ollama_env.get("OLLAMA_MAX_LOADED_MODELS"),
"num_parallel": ollama_env.get("OLLAMA_NUM_PARALLEL"),
"keep_alive": ollama_env.get("OLLAMA_KEEP_ALIVE"),
"config_source": "systemd unit environment" if ollama_env else "unavailable",
}
comfy: Dict[str, Any] = {"online": False}
try:
async with httpx.AsyncClient(timeout=4.0) as c:
r = await c.get(f"{vram_arbitrator.COMFY_API_BASE}/system_stats")
if r.status_code == 200:
data = r.json()
system = data.get("system", {})
argv = system.get("argv") or []
devices = data.get("devices") or []
dev = devices[0] if devices else {}
# The allocator is named in the device string; it is the closest thing
# ComfyUI reports to the "async offload" the old subtitle asserted.
dev_name = dev.get("name", "")
allocator = ("cudaMallocAsync" if "cudaMallocAsync" in dev_name
else "cudaMalloc" if "cudaMalloc" in dev_name else "unknown")
vram_flags = [a for a in argv
if a in ("--lowvram", "--novram", "--highvram", "--normalvram",
"--gpu-only", "--cpu")]
comfy = {
"online": True,
"version": system.get("comfyui_version"),
"pytorch": system.get("pytorch_version"),
# split()[0] on an absent version raises IndexError, which would have
# been swallowed by the except below and reported ComfyUI as offline.
"python": ((system.get("python_version") or "").split() or [None])[0],
"argv": argv,
"vram_mode": vram_flags[0] if vram_flags else "default (auto)",
"allocator": allocator,
"device": dev_name,
"summary": f"{allocator}, {vram_flags[0] if vram_flags else 'auto VRAM'}",
}
except Exception as e:
comfy = {"online": False, "error": str(e)[:120]}
return {"ollama": ollama, "comfyui": comfy}