Recalibrate cache-hit thresholds against measured loads; bring MCP to parity

Calibration. The same 12.87GB model loaded through Ollama on this box:

  3.1% resident (FADV_DONTNEED) -> 34.3s -> 0.38 GB/s
  100% resident (force-warmed)  ->  4.9s -> 2.63 GB/s

The thresholds had been guessed from PCIe bus bandwidth: cache hit at >=5 GB/s. A fully
warm load only reaches 2.63 GB/s, because load_duration covers host-to-device transfer
and model init as well as the file read -- the page cache itself reads at 6.4 GB/s. The
5 GB/s bar was therefore unreachable, and every warm load was being reported as a
partial hit. Now 2.0 / 0.8 GB/s, either side of the measured 6.9x separation.

Warm-skip was also unsafe. A 12.87GB blob was skipped as already resident on the
strength of twelve 2MB probe windows, then loaded at 2.44 GB/s. Skipping now requires
warm_confident: an exact cachestat reading, or a probe finding every one of 32 denser
samples resident. warm_file_to_ram/warm_ollama_blob take force=True, exposed on the
warm-model endpoint, whose Pydantic model was missing the field entirely.

MCP parity: the server had drifted well behind the REST API. Adds tools for measured
residency, warm planning, VRAM requests, per-profile analytics, thermal governor
control, overclock status/apply/restore, and autotune sweeps plus status -- 23 tools
and 6 resources, up from 12 and 3. The telemetry store now starts in __main__ rather
than at import scope, since server.py imports this module for the benchmark tool.

README: replaced the remaining theoretical claims (31.5 GB/s bus rate, sub-1.5s loads,
15ms yields) with the measured numbers.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-08-28 14:40:05 -07:00
parent c689ec8711
commit 01d2f4cfdd
5 changed files with 204 additions and 32 deletions

View File

@@ -143,7 +143,7 @@ PROBE_WINDOW_BYTES = 2 * 1024 * 1024
PROBE_CACHED_GBPS = 1.5
def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
def _throughput_probe(fd: int, size: int, windows_override: Optional[int] = None) -> Dict[str, Any]:
"""Infer residency by timing reads of small windows spread across the file.
Used only where cachestat is not permitted (Ollama's blobs are owned by uid `ollama`).
@@ -158,7 +158,7 @@ def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
pollution the probe itself created, and leaving them behind would slowly warm the
cache with data nobody asked for.
"""
windows = min(PROBE_WINDOWS, max(int(size // PROBE_WINDOW_BYTES), 1))
windows = min(windows_override or PROBE_WINDOWS, max(int(size // PROBE_WINDOW_BYTES), 1))
if windows <= 0:
return {"resident_pct": 0.0, "windows": 0}
@@ -194,7 +194,8 @@ def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
}
def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
def page_residency(filepath: str, allow_probe: bool = True,
probe_windows: Optional[int] = None) -> Dict[str, Any]:
"""Measure what fraction of a file is resident in the Linux page cache."""
try:
size = os.path.getsize(filepath)
@@ -216,7 +217,7 @@ def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
method, measurable = "cachestat", True
extra = {"dirty_pages": cs.nr_dirty, "evicted_pages": cs.nr_evicted}
elif allow_probe:
probe = _throughput_probe(fd, size)
probe = _throughput_probe(fd, size, probe_windows)
pct = probe["resident_pct"]
method, measurable = "probe", True
extra = {"probe_windows": probe["windows"], "probe_median_gbps": probe.get("median_gbps")}
@@ -236,6 +237,11 @@ def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
"method": method,
"measurable": measurable,
"warm": pct >= WARM_SKIP_THRESHOLD_PCT,
# Only an exact measurement is trustworthy enough to skip work on. A probe of a
# dozen 2 MB windows can clear 90% on a file that is mostly cold -- observed
# here as a 12.87 GB "already resident" blob that then loaded at 2.44 GB/s.
"warm_confident": (method == "cachestat" and pct >= WARM_SKIP_THRESHOLD_PCT)
or (method == "probe" and pct >= 100.0),
**extra,
}
except Exception as e:
@@ -468,16 +474,19 @@ def get_cache_report(include_files: bool = True, force_refresh: bool = False) ->
# ---------------------------------------------------------------- warming
def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024,
skip_if_warm: bool = True) -> Dict[str, Any]:
"""Pre-fault a file into the Linux page cache, skipping it if already resident."""
skip_if_warm: bool = True, force: bool = False) -> Dict[str, Any]:
"""Pre-fault a file into the Linux page cache, skipping it only if confidently resident."""
if not os.path.exists(filepath):
return {"success": False, "error": f"File not found: {filepath}", "duration_ms": 0}
before = page_residency(filepath)
if skip_if_warm and before.get("warm"):
# Probe densely here: this decision skips real work, so it is worth 32 samples
# rather than 12.
before = page_residency(filepath, probe_windows=32)
if skip_if_warm and not force and before.get("warm_confident"):
return {
"success": True, "filepath": filepath, "skipped": True,
"reason": "already resident", "resident_pct": before.get("resident_pct"),
"method": before.get("method"),
"size_mb": round(before.get("size_bytes", 0) / (1024**2), 2),
"duration_ms": 0.0, "bytes_read": 0,
}
@@ -499,7 +508,7 @@ def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024,
bytes_read += n
duration = time.perf_counter() - t0
after = page_residency(filepath)
after = page_residency(filepath, probe_windows=32)
return {
"success": True,
"filepath": filepath,
@@ -551,11 +560,11 @@ async def warm_ollama_model(model_name: str, keep_alive: str = "5m") -> Dict[str
"duration_ms": round((time.perf_counter() - t0) * 1000, 2)}
def warm_ollama_blob(model_name: str) -> Dict[str, Any]:
def warm_ollama_blob(model_name: str, force: bool = False) -> Dict[str, Any]:
"""Warm a specific Ollama model's GGUF into page cache without touching VRAM."""
for f in find_ollama_model_files():
if f["model"] == model_name:
res = warm_file_to_ram(f["full_path"])
res = warm_file_to_ram(f["full_path"], force=force)
res["model"] = model_name
return res
return {"success": False, "error": f"no blob found for model '{model_name}'"}
@@ -603,13 +612,13 @@ def build_warm_plan(budget_gb: Optional[float] = None) -> Dict[str, Any]:
if c["full_path"] in seen_paths:
continue
seen_paths.add(c["full_path"])
res = page_residency(c["full_path"])
res = page_residency(c["full_path"], probe_windows=32)
entry = {
"name": c["name"], "kind": c["kind"], "full_path": c["full_path"],
"size_gb": c.get("size_gb", 0), "score": round(c["score"], 4),
"resident_pct": res.get("resident_pct", 0.0),
}
if res.get("warm"):
if res.get("warm_confident"):
entry["action"] = "already-warm"
skipped.append(entry)
continue