Recalibrate cache-hit thresholds against measured loads; bring MCP to parity
Calibration. The same 12.87GB model loaded through Ollama on this box: 3.1% resident (FADV_DONTNEED) -> 34.3s -> 0.38 GB/s 100% resident (force-warmed) -> 4.9s -> 2.63 GB/s The thresholds had been guessed from PCIe bus bandwidth: cache hit at >=5 GB/s. A fully warm load only reaches 2.63 GB/s, because load_duration covers host-to-device transfer and model init as well as the file read -- the page cache itself reads at 6.4 GB/s. The 5 GB/s bar was therefore unreachable, and every warm load was being reported as a partial hit. Now 2.0 / 0.8 GB/s, either side of the measured 6.9x separation. Warm-skip was also unsafe. A 12.87GB blob was skipped as already resident on the strength of twelve 2MB probe windows, then loaded at 2.44 GB/s. Skipping now requires warm_confident: an exact cachestat reading, or a probe finding every one of 32 denser samples resident. warm_file_to_ram/warm_ollama_blob take force=True, exposed on the warm-model endpoint, whose Pydantic model was missing the field entirely. MCP parity: the server had drifted well behind the REST API. Adds tools for measured residency, warm planning, VRAM requests, per-profile analytics, thermal governor control, overclock status/apply/restore, and autotune sweeps plus status -- 23 tools and 6 resources, up from 12 and 3. The telemetry store now starts in __main__ rather than at import scope, since server.py imports this module for the benchmark tool. README: replaced the remaining theoretical claims (31.5 GB/s bus rate, sub-1.5s loads, 15ms yields) with the measured numbers. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
128
mcp_server.py
128
mcp_server.py
@@ -7,16 +7,19 @@ import logging
|
||||
from typing import Dict, List, Any, Optional
|
||||
|
||||
from mcp.server import MCPServer
|
||||
import ram_optimizer
|
||||
import vram_arbitrator
|
||||
import autotune
|
||||
import overclock_manager
|
||||
import ram_optimizer
|
||||
import telemetry_store
|
||||
import thermal_governor
|
||||
import vram_arbitrator
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(name)s: %(message)s")
|
||||
logger = logging.getLogger("gpu_swapper_mcp")
|
||||
|
||||
mcp = MCPServer(
|
||||
name="gpu-program-swapper",
|
||||
version="1.0.0",
|
||||
version="2.0.0",
|
||||
description="Orchestrates high-speed GPU VRAM hot-swaps between Ollama LLMs and ComfyUI with 64GB RAM cache telemetry."
|
||||
)
|
||||
|
||||
@@ -133,6 +136,105 @@ def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str:
|
||||
res = overclock_manager.set_fan_auto()
|
||||
return json.dumps(res, indent=2)
|
||||
|
||||
@mcp.tool()
|
||||
def get_page_cache_residency(include_files: bool = True) -> str:
|
||||
"""Measure how much of each model on disk is genuinely resident in the Linux page cache.
|
||||
|
||||
Uses cachestat(2) where the kernel permits it and a read-rate probe where it does not
|
||||
(Ollama blobs are owned by another user). Reports which method was used per file, and
|
||||
marks anything it cannot measure rather than guessing."""
|
||||
report = ram_optimizer.get_cache_report(include_files=include_files)
|
||||
report["capability"] = ram_optimizer.residency_capability()
|
||||
return json.dumps(report, indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def get_warm_plan(budget_gb: Optional[float] = None) -> str:
|
||||
"""Preview which models pre-warming would load into RAM, in what order, and what it
|
||||
would skip — ranked by recency/frequency and capped by a byte budget. Does not warm."""
|
||||
return json.dumps(ram_optimizer.build_warm_plan(budget_gb), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
async def request_vram_for_ollama(needed_gb: float = 0.0) -> str:
|
||||
"""Free VRAM for an LLM right now: purges ComfyUI's cached checkpoints immediately if
|
||||
there is not enough headroom, instead of waiting for the normal idle timer."""
|
||||
res = await vram_arbitrator.arbitrator.request_vram_for_ollama(needed_gb)
|
||||
return json.dumps(res, indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def get_profile_performance(days: float = 7.0) -> str:
|
||||
"""Compare measured decode throughput and thermals per overclock profile, from
|
||||
persisted history. Answers whether a given profile is actually delivering more tok/s."""
|
||||
return json.dumps({
|
||||
"window_days": days,
|
||||
"profiles": telemetry_store.profile_comparison(days),
|
||||
"swaps": telemetry_store.swap_stats(days),
|
||||
}, indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def get_thermal_governor_status() -> str:
|
||||
"""Current thermal derate level, why it was applied, and the escalation history."""
|
||||
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def set_thermal_governor(enabled: Optional[bool] = None, reset: bool = False) -> str:
|
||||
"""Enable or disable the thermal governor, or clear an active derate and reapply the
|
||||
full profile."""
|
||||
if enabled is not None:
|
||||
thermal_governor.governor.set_enabled(enabled)
|
||||
if reset:
|
||||
thermal_governor.governor.reset()
|
||||
return json.dumps(thermal_governor.governor.get_status(), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def get_overclock_status() -> str:
|
||||
"""Active overclock profile, all profiles with the evidence behind their settings, and
|
||||
which hardware levers this driver actually honours (clock offsets are ignored on some)."""
|
||||
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def apply_overclock_profile(profile: str) -> str:
|
||||
"""Apply an overclock profile by name: ollama | comfy | balanced."""
|
||||
return json.dumps(overclock_manager.apply_profile(profile), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def restore_stock_gpu_state() -> str:
|
||||
"""Drop all clock locks and offsets, restore the default power limit, and return the
|
||||
fans to automatic control."""
|
||||
return json.dumps(overclock_manager.restore_safe("MCP request"), indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
async def run_overclock_sweep(knob: str = "power_limit_w", profile: str = "ollama",
|
||||
workload: str = "auto", start: Optional[int] = None,
|
||||
stop: Optional[int] = None, repeats: int = 1,
|
||||
apply_best: bool = False) -> str:
|
||||
"""Sweep one GPU knob against a real workload and report the fastest stable value.
|
||||
|
||||
knob: power_limit_w | lock_mem_mhz | lock_core_max | mem_offset_mhz | core_offset_mhz
|
||||
workload: 'ollama' (decode tok/s), 'comfy' (SDXL it/s), or 'auto' to match the profile.
|
||||
|
||||
Verifies the knob actually moves the hardware before sweeping, refuses to run while
|
||||
ComfyUI is busy, and always restores the original profile. Takes minutes."""
|
||||
res = await autotune.sweep(knob=knob, profile=profile, workload=workload,
|
||||
start=start, stop=stop, repeats=repeats,
|
||||
apply_best=apply_best)
|
||||
return json.dumps(res, indent=2, default=str)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def get_autotune_status() -> str:
|
||||
"""Sweep progress, the last sweep's full result table, and every recorded autotune step."""
|
||||
return json.dumps(autotune.get_status(), indent=2, default=str)
|
||||
|
||||
|
||||
# ==========================================
|
||||
# MCP RESOURCES
|
||||
# ==========================================
|
||||
@@ -154,6 +256,21 @@ def get_switch_history_resource() -> str:
|
||||
"""Recent model switch events and latencies."""
|
||||
return json.dumps(vram_arbitrator.get_switch_history(), indent=2)
|
||||
|
||||
@mcp.resource("gpu://cache/residency")
|
||||
def get_cache_residency_resource() -> str:
|
||||
"""Measured page-cache residency across every model on disk."""
|
||||
return json.dumps(ram_optimizer.get_cache_report(include_files=True), indent=2, default=str)
|
||||
|
||||
@mcp.resource("gpu://analytics/profiles")
|
||||
def get_profile_analytics_resource() -> str:
|
||||
"""Measured throughput and thermals per overclock profile, from persisted history."""
|
||||
return json.dumps(telemetry_store.profile_comparison(7.0), indent=2, default=str)
|
||||
|
||||
@mcp.resource("gpu://overclock/profiles")
|
||||
def get_overclock_profiles_resource() -> str:
|
||||
"""Overclock profiles, including the measurement recorded behind each setting."""
|
||||
return json.dumps(overclock_manager.get_status(), indent=2, default=str)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
@@ -163,6 +280,11 @@ if __name__ == "__main__":
|
||||
parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Only when run as a standalone server. server.py imports this module for the
|
||||
# benchmark tool, and starting the store at import scope would spin up a writer as a
|
||||
# side effect of that import.
|
||||
telemetry_store.start()
|
||||
|
||||
if args.sse:
|
||||
mcp.run(transport="sse", host="0.0.0.0", port=args.port)
|
||||
else:
|
||||
|
||||
Reference in New Issue
Block a user