"""Live configuration of the two engines HyperSwap arbitrates between. Arbitration behaviour is largely dictated by settings that live outside this codebase. Working out why a yield behaved the way it did meant reading journald and the ollama unit by hand: OLLAMA_NUM_PARALLEL decides whether an unload queues behind a running generation, OLLAMA_MAX_LOADED_MODELS decides whether more than one model can be resident, and a pinned n_gpu_layers decides whether a model that will not fit spills to the CPU or fails outright. Those are worth reading and explaining rather than hardcoding into a dashboard subtitle that silently goes stale. """ import json import logging import subprocess from typing import Any, Dict, List, Optional import httpx import vram_arbitrator logger = logging.getLogger("engines") # Settings that change how the arbitrator must behave, with what they imply. OLLAMA_SETTING_NOTES = { "OLLAMA_NUM_PARALLEL": ( "Requests per model. At 1, a keep_alive:0 unload queues behind any running " "generation and applies when it finishes — which is why a busy model is " "reported as deferred rather than failed."), "OLLAMA_MAX_LOADED_MODELS": ( "How many models may be resident at once. At 1, Ollama evicts the previous " "model on every swap."), "OLLAMA_KEEP_ALIVE": ( "Default residency after a request. Long values keep VRAM occupied and make " "ComfyUI wait for an explicit yield."), "OLLAMA_FLASH_ATTENTION": "FlashAttention kernels for attention.", "OLLAMA_KV_CACHE_TYPE": "KV cache quantisation; smaller types cut VRAM per context.", "OLLAMA_NUM_BATCH": "Prompt-evaluation batch size.", } def _ollama_unit_environment() -> Dict[str, str]: """Read the ollama service's environment. Empty if it is not a systemd unit.""" env: Dict[str, str] = {} try: proc = subprocess.run(["systemctl", "show", "ollama", "-p", "Environment", "--value"], capture_output=True, text=True, timeout=8) for token in proc.stdout.split(): if "=" in token and token.startswith("OLLAMA"): k, _, v = token.partition("=") env[k] = v except Exception as e: logger.debug(f"could not read ollama unit environment: {e}") return env async def get_engine_config() -> Dict[str, Any]: """Real, live configuration of both engines, with arbitration implications.""" ollama_env = _ollama_unit_environment() ollama_settings = [ {"key": k, "value": v, "means": OLLAMA_SETTING_NOTES.get(k, "")} for k, v in sorted(ollama_env.items()) ] # A short, honest summary line to replace the dashboard's hardcoded subtitle. feature_bits: List[str] = [] if ollama_env.get("OLLAMA_FLASH_ATTENTION") == "1": feature_bits.append("FlashAttention") kv = ollama_env.get("OLLAMA_KV_CACHE_TYPE") if kv: feature_bits.append(f"{kv} KV cache") host = ollama_env.get("OLLAMA_HOST", "") port = host.rsplit(":", 1)[-1] if ":" in host else "11434" ollama = { "port": port, "settings": ollama_settings, "summary": " + ".join(feature_bits) if feature_bits else "default configuration", "max_loaded_models": ollama_env.get("OLLAMA_MAX_LOADED_MODELS"), "num_parallel": ollama_env.get("OLLAMA_NUM_PARALLEL"), "keep_alive": ollama_env.get("OLLAMA_KEEP_ALIVE"), "config_source": "systemd unit environment" if ollama_env else "unavailable", } comfy: Dict[str, Any] = {"online": False} try: async with httpx.AsyncClient(timeout=4.0) as c: r = await c.get(f"{vram_arbitrator.COMFY_API_BASE}/system_stats") if r.status_code == 200: data = r.json() system = data.get("system", {}) argv = system.get("argv") or [] devices = data.get("devices") or [] dev = devices[0] if devices else {} # The allocator is named in the device string; it is the closest thing # ComfyUI reports to the "async offload" the old subtitle asserted. dev_name = dev.get("name", "") allocator = ("cudaMallocAsync" if "cudaMallocAsync" in dev_name else "cudaMalloc" if "cudaMalloc" in dev_name else "unknown") vram_flags = [a for a in argv if a in ("--lowvram", "--novram", "--highvram", "--normalvram", "--gpu-only", "--cpu")] comfy = { "online": True, "version": system.get("comfyui_version"), "pytorch": system.get("pytorch_version"), # split()[0] on an absent version raises IndexError, which would have # been swallowed by the except below and reported ComfyUI as offline. "python": ((system.get("python_version") or "").split() or [None])[0], "argv": argv, "vram_mode": vram_flags[0] if vram_flags else "default (auto)", "allocator": allocator, "device": dev_name, "summary": f"{allocator}, {vram_flags[0] if vram_flags else 'auto VRAM'}", } except Exception as e: comfy = {"online": False, "error": str(e)[:120]} return {"ollama": ollama, "comfyui": comfy}