Recalibrate cache-hit thresholds against measured loads; bring MCP to parity

Calibration. The same 12.87GB model loaded through Ollama on this box:

  3.1% resident (FADV_DONTNEED) -> 34.3s -> 0.38 GB/s
  100% resident (force-warmed)  ->  4.9s -> 2.63 GB/s

The thresholds had been guessed from PCIe bus bandwidth: cache hit at >=5 GB/s. A fully
warm load only reaches 2.63 GB/s, because load_duration covers host-to-device transfer
and model init as well as the file read -- the page cache itself reads at 6.4 GB/s. The
5 GB/s bar was therefore unreachable, and every warm load was being reported as a
partial hit. Now 2.0 / 0.8 GB/s, either side of the measured 6.9x separation.

Warm-skip was also unsafe. A 12.87GB blob was skipped as already resident on the
strength of twelve 2MB probe windows, then loaded at 2.44 GB/s. Skipping now requires
warm_confident: an exact cachestat reading, or a probe finding every one of 32 denser
samples resident. warm_file_to_ram/warm_ollama_blob take force=True, exposed on the
warm-model endpoint, whose Pydantic model was missing the field entirely.

MCP parity: the server had drifted well behind the REST API. Adds tools for measured
residency, warm planning, VRAM requests, per-profile analytics, thermal governor
control, overclock status/apply/restore, and autotune sweeps plus status -- 23 tools
and 6 resources, up from 12 and 3. The telemetry store now starts in __main__ rather
than at import scope, since server.py imports this module for the benchmark tool.

README: replaced the remaining theoretical claims (31.5 GB/s bus rate, sub-1.5s loads,
15ms yields) with the measured numbers.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-08-28 14:40:05 -07:00
parent c689ec8711
commit 01d2f4cfdd
5 changed files with 204 additions and 32 deletions

View File

@@ -7,16 +7,19 @@ import logging
from typing import Dict, List, Any, Optional
from mcp.server import MCPServer
import ram_optimizer
import vram_arbitrator
import autotune
import overclock_manager
import ram_optimizer
import telemetry_store
import thermal_governor
import vram_arbitrator
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(name)s: %(message)s")
logger = logging.getLogger("gpu_swapper_mcp")
mcp = MCPServer(
name="gpu-program-swapper",
version="1.0.0",
version="2.0.0",
description="Orchestrates high-speed GPU VRAM hot-swaps between Ollama LLMs and ComfyUI with 64GB RAM cache telemetry."
)
@@ -133,6 +136,105 @@ def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str:
res = overclock_manager.set_fan_auto()
return json.dumps(res, indent=2)
@mcp.tool()
def get_page_cache_residency(include_files: bool = True) -> str:
"""Measure how much of each model on disk is genuinely resident in the Linux page cache.
Uses cachestat(2) where the kernel permits it and a read-rate probe where it does not
(Ollama blobs are owned by another user). Reports which method was used per file, and
marks anything it cannot measure rather than guessing."""
report = ram_optimizer.get_cache_report(include_files=include_files)
report["capability"] = ram_optimizer.residency_capability()
return json.dumps(report, indent=2, default=str)
@mcp.tool()
def get_warm_plan(budget_gb: Optional[float] = None) -> str:
"""Preview which models pre-warming would load into RAM, in what order, and what it
would skip — ranked by recency/frequency and capped by a byte budget. Does not warm."""
return json.dumps(ram_optimizer.build_warm_plan(budget_gb), indent=2, default=str)
@mcp.tool()
async def request_vram_for_ollama(needed_gb: float = 0.0) -> str:
"""Free VRAM for an LLM right now: purges ComfyUI's cached checkpoints immediately if
there is not enough headroom, instead of waiting for the normal idle timer."""
res = await vram_arbitrator.arbitrator.request_vram_for_ollama(needed_gb)
return json.dumps(res, indent=2, default=str)
@mcp.tool()
def get_profile_performance(days: float = 7.0) -> str:
"""Compare measured decode throughput and thermals per overclock profile, from
persisted history. Answers whether a given profile is actually delivering more tok/s."""
return json.dumps({
"window_days": days,
"profiles": telemetry_store.profile_comparison(days),
"swaps": telemetry_store.swap_stats(days),
}, indent=2, default=str)
@mcp.tool()
def get_thermal_governor_status() -> str:
"""Current thermal derate level, why it was applied, and the escalation history."""
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
@mcp.tool()
def set_thermal_governor(enabled: Optional[bool] = None, reset: bool = False) -> str:
"""Enable or disable the thermal governor, or clear an active derate and reapply the
full profile."""
if enabled is not None:
thermal_governor.governor.set_enabled(enabled)
if reset:
thermal_governor.governor.reset()
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
@mcp.tool()
def get_overclock_status() -> str:
"""Active overclock profile, all profiles with the evidence behind their settings, and
which hardware levers this driver actually honours (clock offsets are ignored on some)."""
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
@mcp.tool()
def apply_overclock_profile(profile: str) -> str:
"""Apply an overclock profile by name: ollama | comfy | balanced."""
return json.dumps(overclock_manager.apply_profile(profile), indent=2, default=str)
@mcp.tool()
def restore_stock_gpu_state() -> str:
"""Drop all clock locks and offsets, restore the default power limit, and return the
fans to automatic control."""
return json.dumps(overclock_manager.restore_safe("MCP request"), indent=2, default=str)
@mcp.tool()
async def run_overclock_sweep(knob: str = "power_limit_w", profile: str = "ollama",
workload: str = "auto", start: Optional[int] = None,
stop: Optional[int] = None, repeats: int = 1,
apply_best: bool = False) -> str:
"""Sweep one GPU knob against a real workload and report the fastest stable value.
knob: power_limit_w | lock_mem_mhz | lock_core_max | mem_offset_mhz | core_offset_mhz
workload: 'ollama' (decode tok/s), 'comfy' (SDXL it/s), or 'auto' to match the profile.
Verifies the knob actually moves the hardware before sweeping, refuses to run while
ComfyUI is busy, and always restores the original profile. Takes minutes."""
res = await autotune.sweep(knob=knob, profile=profile, workload=workload,
start=start, stop=stop, repeats=repeats,
apply_best=apply_best)
return json.dumps(res, indent=2, default=str)
@mcp.tool()
def get_autotune_status() -> str:
"""Sweep progress, the last sweep's full result table, and every recorded autotune step."""
return json.dumps(autotune.get_status(), indent=2, default=str)
# ==========================================
# MCP RESOURCES
# ==========================================
@@ -154,6 +256,21 @@ def get_switch_history_resource() -> str:
"""Recent model switch events and latencies."""
return json.dumps(vram_arbitrator.get_switch_history(), indent=2)
@mcp.resource("gpu://cache/residency")
def get_cache_residency_resource() -> str:
"""Measured page-cache residency across every model on disk."""
return json.dumps(ram_optimizer.get_cache_report(include_files=True), indent=2, default=str)
@mcp.resource("gpu://analytics/profiles")
def get_profile_analytics_resource() -> str:
"""Measured throughput and thermals per overclock profile, from persisted history."""
return json.dumps(telemetry_store.profile_comparison(7.0), indent=2, default=str)
@mcp.resource("gpu://overclock/profiles")
def get_overclock_profiles_resource() -> str:
"""Overclock profiles, including the measurement recorded behind each setting."""
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
if __name__ == "__main__":
import argparse
@@ -163,6 +280,11 @@ if __name__ == "__main__":
parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
args = parser.parse_args()
# Only when run as a standalone server. server.py imports this module for the
# benchmark tool, and starting the store at import scope would spin up a writer as a
# side effect of that import.
telemetry_store.start()
if args.sse:
mcp.run(transport="sse", host="0.0.0.0", port=args.port)
else: