The engine subtitles were hardcoded: "FlashAttention + Q4 KV Cache" and "DynamicVRAM
+ Pinned Async Offload". The first turned out to be accurate -- OLLAMA_FLASH_ATTENTION
and OLLAMA_KV_CACHE_TYPE really are set -- which is worse than being wrong, because it
would have gone on looking accurate after the settings changed.
engines.py reads both engines' real configuration: the ollama service environment via
systemd, and ComfyUI's own /system_stats for version, allocator, VRAM mode and argv.
Exposed at GET /api/engines, as an MCP tool, and in the dashboard subtitles with the
full settings list as a tooltip.
The settings worth surfacing are the ones that dictate how this service must behave
and that previously had to be discovered by reading journald: OLLAMA_NUM_PARALLEL=1
is why an unload queues behind a running generation and is reported as deferred
rather than failed, and OLLAMA_MAX_LOADED_MODELS=1 is why every swap evicts the
previous model. Each is reported with that explanation attached.
Writing the tests found a bug in the new code: (system.get("python_version") or
"").split()[0] raises IndexError when ComfyUI omits the field, and the surrounding
except would have swallowed it and reported ComfyUI as entirely offline.
Tests: 192 (was 182).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
310 lines
13 KiB
Python
310 lines
13 KiB
Python
"""Model Context Protocol (MCP) Server for GPU Program Swapper and Memory Orchestrator."""
|
|
import os
|
|
import sys
|
|
import json
|
|
import asyncio
|
|
import logging
|
|
from typing import Dict, List, Any, Optional
|
|
|
|
from mcp.server import MCPServer
|
|
import autotune
|
|
import engines
|
|
import health
|
|
import overclock_manager
|
|
import ram_optimizer
|
|
import telemetry_store
|
|
import thermal_governor
|
|
import vram_arbitrator
|
|
|
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(name)s: %(message)s")
|
|
logger = logging.getLogger("gpu_swapper_mcp")
|
|
|
|
mcp = MCPServer(
|
|
name="gpu-program-swapper",
|
|
version="2.0.0",
|
|
description="Orchestrates high-speed GPU VRAM hot-swaps between Ollama LLMs and ComfyUI with 64GB RAM cache telemetry."
|
|
)
|
|
|
|
# ==========================================
|
|
# MCP TOOLS
|
|
# ==========================================
|
|
|
|
@mcp.tool()
|
|
def get_gpu_status() -> str:
|
|
"""Get live NVIDIA GPU hardware telemetry, VRAM allocation (Ollama vs ComfyUI vs System), temperature, power, and active compute PIDs."""
|
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
|
return json.dumps(stats, indent=2)
|
|
|
|
@mcp.tool()
|
|
def get_host_memory_status() -> str:
|
|
"""Get host system RAM breakdown (Total, Apps Used, OS Page Cache containing models in RAM, Free memory) and cache hit percentage."""
|
|
mem = ram_optimizer.get_detailed_meminfo()
|
|
return json.dumps(mem, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def switch_ollama_model(model_name: str, keep_alive: str = "30m") -> str:
|
|
"""Hot-swap active Ollama LLM in VRAM. Measures swap latency (ms), eval speed (tokens/sec), and whether it was an instant RAM cache hit."""
|
|
res = await vram_arbitrator.switch_ollama_model(model_name, keep_alive=keep_alive)
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def soft_yield_ollama_vram(model_name: Optional[str] = None) -> str:
|
|
"""Soft-yield Ollama VRAM down to 0 MB in ~15 milliseconds while keeping all model weights resident in 64GB host RAM cache."""
|
|
res = await vram_arbitrator.instant_free_ollama_vram(model_name)
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def purge_comfyui_vram() -> str:
|
|
"""Purge loaded diffusion models and VRAM cache from the ComfyUI pipeline."""
|
|
res = await vram_arbitrator.instant_free_comfyui_vram()
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def prewarm_all_models_to_ram() -> str:
|
|
"""Pre-read and fault all installed Ollama GGUF models and ComfyUI Safetensors checkpoints into the Linux OS Page Cache for PCIe-speed hot swapping."""
|
|
res = await ram_optimizer.warm_all_models()
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def prewarm_single_model(model_name: Optional[str] = None, filepath: Optional[str] = None) -> str:
|
|
"""Pre-warm a specific Ollama model or individual model file path into Linux RAM cache."""
|
|
if model_name:
|
|
res = await ram_optimizer.warm_ollama_model(model_name, keep_alive="1m")
|
|
elif filepath:
|
|
res = ram_optimizer.warm_file_to_ram(filepath)
|
|
else:
|
|
res = {"error": "Either model_name or filepath must be provided"}
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def list_available_models() -> str:
|
|
"""List all installed Ollama models and discovered ComfyUI model checkpoints/safetensors on disk with sizes and quantization levels."""
|
|
ollama_state = await vram_arbitrator.get_ollama_live_state()
|
|
comfy_models = ram_optimizer.find_comfy_model_files()
|
|
result = {
|
|
"ollama_models": ollama_state.get("installed_models", []),
|
|
"comfy_models": comfy_models,
|
|
}
|
|
return json.dumps(result, indent=2)
|
|
|
|
@mcp.tool()
|
|
def get_switch_history(limit: int = 20) -> str:
|
|
"""Get the recent history log of model switch events, swap durations (in ms), and RAM cache hit status."""
|
|
history = vram_arbitrator.get_switch_history()
|
|
return json.dumps(history[:limit], indent=2)
|
|
|
|
@mcp.tool()
|
|
async def run_model_switch_benchmark(iterations: int = 2) -> str:
|
|
"""Run an automated benchmark swapping between available models to measure round-trip latency and RAM cache effectiveness."""
|
|
ollama_state = await vram_arbitrator.get_ollama_live_state()
|
|
installed = [m.get("name") for m in ollama_state.get("installed_models", []) if m.get("name")]
|
|
if len(installed) < 2:
|
|
return json.dumps({"error": "Need at least 2 installed models to benchmark", "installed": installed})
|
|
|
|
m1, m2 = installed[0], installed[1]
|
|
results = []
|
|
|
|
for i in range(iterations):
|
|
# Swap to m1
|
|
r1 = await vram_arbitrator.switch_ollama_model(m1, keep_alive="5m")
|
|
results.append({"iteration": i+1, "direction": f"-> {m1}", "latency_ms": r1.get("total_duration_ms", 0), "load_ms": r1.get("load_duration_ms", 0), "is_ram_hit": r1.get("is_ram_hit", False)})
|
|
await asyncio.sleep(0.5)
|
|
# Swap to m2
|
|
r2 = await vram_arbitrator.switch_ollama_model(m2, keep_alive="5m")
|
|
results.append({"iteration": i+1, "direction": f"-> {m2}", "latency_ms": r2.get("total_duration_ms", 0), "load_ms": r2.get("load_duration_ms", 0), "is_ram_hit": r2.get("is_ram_hit", False)})
|
|
await asyncio.sleep(0.5)
|
|
|
|
avg_latency = round(sum(r["latency_ms"] for r in results) / len(results), 2) if results else 0
|
|
return json.dumps({
|
|
"status": "completed",
|
|
"tested_models": [m1, m2],
|
|
"iterations": iterations,
|
|
"average_swap_latency_ms": avg_latency,
|
|
"rounds": results,
|
|
}, indent=2)
|
|
|
|
@mcp.tool()
|
|
def get_gpu_fan_status() -> str:
|
|
"""Get current GPU fan control mode (auto vs manual) and target speed."""
|
|
status = overclock_manager.get_fan_status()
|
|
return json.dumps(status, indent=2)
|
|
|
|
@mcp.tool()
|
|
def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str:
|
|
"""Set GPU fan speed mode ('auto' or 'manual') with target percent (30-100%)."""
|
|
if mode.lower() == "manual" and percent is not None:
|
|
res = overclock_manager.set_fan_speed(percent)
|
|
else:
|
|
res = overclock_manager.set_fan_auto()
|
|
return json.dumps(res, indent=2)
|
|
|
|
@mcp.tool()
|
|
async def get_engine_config() -> str:
|
|
"""Live configuration of Ollama and ComfyUI (parallelism, max loaded models,
|
|
keep-alive, KV cache type, ComfyUI VRAM mode and allocator), with what each setting
|
|
implies for VRAM arbitration."""
|
|
return json.dumps(await engines.get_engine_config(), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
async def check_system_health() -> str:
|
|
"""Check every dependency HyperSwap needs (NVML, sudo nvidia-smi, fan control via the
|
|
headless X server, Ollama, ComfyUI, the telemetry store, model directories) and report
|
|
what is broken, what it breaks, and how to fix it."""
|
|
return json.dumps(await health.run_health_checks(), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_page_cache_residency(include_files: bool = True) -> str:
|
|
"""Measure how much of each model on disk is genuinely resident in the Linux page cache.
|
|
|
|
Uses cachestat(2) where the kernel permits it and a read-rate probe where it does not
|
|
(Ollama blobs are owned by another user). Reports which method was used per file, and
|
|
marks anything it cannot measure rather than guessing."""
|
|
report = ram_optimizer.get_cache_report(include_files=include_files)
|
|
report["capability"] = ram_optimizer.residency_capability()
|
|
return json.dumps(report, indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_warm_plan(budget_gb: Optional[float] = None) -> str:
|
|
"""Preview which models pre-warming would load into RAM, in what order, and what it
|
|
would skip — ranked by recency/frequency and capped by a byte budget. Does not warm."""
|
|
return json.dumps(ram_optimizer.build_warm_plan(budget_gb), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
async def request_vram_for_ollama(needed_gb: float = 0.0) -> str:
|
|
"""Free VRAM for an LLM right now: purges ComfyUI's cached checkpoints immediately if
|
|
there is not enough headroom, instead of waiting for the normal idle timer."""
|
|
res = await vram_arbitrator.arbitrator.request_vram_for_ollama(needed_gb)
|
|
return json.dumps(res, indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_profile_performance(days: float = 7.0) -> str:
|
|
"""Compare measured decode throughput and thermals per overclock profile, from
|
|
persisted history. Answers whether a given profile is actually delivering more tok/s."""
|
|
return json.dumps({
|
|
"window_days": days,
|
|
"profiles": telemetry_store.profile_comparison(days),
|
|
"swaps": telemetry_store.swap_stats(days),
|
|
}, indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_thermal_governor_status() -> str:
|
|
"""Current thermal derate level, why it was applied, and the escalation history."""
|
|
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def set_thermal_governor(enabled: Optional[bool] = None, reset: bool = False) -> str:
|
|
"""Enable or disable the thermal governor, or clear an active derate and reapply the
|
|
full profile."""
|
|
if enabled is not None:
|
|
thermal_governor.governor.set_enabled(enabled)
|
|
if reset:
|
|
thermal_governor.governor.reset()
|
|
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_overclock_status() -> str:
|
|
"""Active overclock profile, all profiles with the evidence behind their settings, and
|
|
which hardware levers this driver actually honours (clock offsets are ignored on some)."""
|
|
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def apply_overclock_profile(profile: str) -> str:
|
|
"""Apply an overclock profile by name: ollama | comfy | balanced."""
|
|
return json.dumps(overclock_manager.apply_profile(profile), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def restore_stock_gpu_state() -> str:
|
|
"""Drop all clock locks and offsets, restore the default power limit, and return the
|
|
fans to automatic control."""
|
|
return json.dumps(overclock_manager.restore_safe("MCP request"), indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
async def run_overclock_sweep(knob: str = "power_limit_w", profile: str = "ollama",
|
|
workload: str = "auto", start: Optional[int] = None,
|
|
stop: Optional[int] = None, repeats: int = 1,
|
|
apply_best: bool = False) -> str:
|
|
"""Sweep one GPU knob against a real workload and report the fastest stable value.
|
|
|
|
knob: power_limit_w | lock_mem_mhz | lock_core_max | mem_offset_mhz | core_offset_mhz
|
|
workload: 'ollama' (decode tok/s), 'comfy' (SDXL it/s), or 'auto' to match the profile.
|
|
|
|
Verifies the knob actually moves the hardware before sweeping, refuses to run while
|
|
ComfyUI is busy, and always restores the original profile. Takes minutes."""
|
|
res = await autotune.sweep(knob=knob, profile=profile, workload=workload,
|
|
start=start, stop=stop, repeats=repeats,
|
|
apply_best=apply_best)
|
|
return json.dumps(res, indent=2, default=str)
|
|
|
|
|
|
@mcp.tool()
|
|
def get_autotune_status() -> str:
|
|
"""Sweep progress, the last sweep's full result table, and every recorded autotune step."""
|
|
return json.dumps(autotune.get_status(), indent=2, default=str)
|
|
|
|
|
|
# ==========================================
|
|
# MCP RESOURCES
|
|
# ==========================================
|
|
|
|
@mcp.resource("gpu://metrics/live")
|
|
def get_live_metrics_resource() -> str:
|
|
"""Live snapshot of GPU hardware sensors, VRAM, and RAM cache."""
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
ram = ram_optimizer.get_detailed_meminfo()
|
|
return json.dumps({"gpu": gpu, "ram": ram}, indent=2)
|
|
|
|
@mcp.resource("gpu://models/catalog")
|
|
async def get_models_catalog_resource() -> str:
|
|
"""Catalog of all local Ollama and ComfyUI models."""
|
|
return await list_available_models()
|
|
|
|
@mcp.resource("gpu://history/switches")
|
|
def get_switch_history_resource() -> str:
|
|
"""Recent model switch events and latencies."""
|
|
return json.dumps(vram_arbitrator.get_switch_history(), indent=2)
|
|
|
|
@mcp.resource("gpu://cache/residency")
|
|
def get_cache_residency_resource() -> str:
|
|
"""Measured page-cache residency across every model on disk."""
|
|
return json.dumps(ram_optimizer.get_cache_report(include_files=True), indent=2, default=str)
|
|
|
|
@mcp.resource("gpu://analytics/profiles")
|
|
def get_profile_analytics_resource() -> str:
|
|
"""Measured throughput and thermals per overclock profile, from persisted history."""
|
|
return json.dumps(telemetry_store.profile_comparison(7.0), indent=2, default=str)
|
|
|
|
@mcp.resource("gpu://overclock/profiles")
|
|
def get_overclock_profiles_resource() -> str:
|
|
"""Overclock profiles, including the measurement recorded behind each setting."""
|
|
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import argparse
|
|
parser = argparse.ArgumentParser(description="HyperSwap GPU Program Swapper MCP Server")
|
|
parser.add_argument("--stdio", action="store_true", help="Run in stdio mode (default)")
|
|
parser.add_argument("--sse", action="store_true", help="Run with SSE transport")
|
|
parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
|
|
args = parser.parse_args()
|
|
|
|
# Only when run as a standalone server. server.py imports this module for the
|
|
# benchmark tool, and starting the store at import scope would spin up a writer as a
|
|
# side effect of that import.
|
|
telemetry_store.start()
|
|
|
|
if args.sse:
|
|
mcp.run(transport="sse", host="0.0.0.0", port=args.port)
|
|
else:
|
|
mcp.run(transport="stdio")
|