Recalibrate cache-hit thresholds against measured loads; bring MCP to parity
Calibration. The same 12.87GB model loaded through Ollama on this box: 3.1% resident (FADV_DONTNEED) -> 34.3s -> 0.38 GB/s 100% resident (force-warmed) -> 4.9s -> 2.63 GB/s The thresholds had been guessed from PCIe bus bandwidth: cache hit at >=5 GB/s. A fully warm load only reaches 2.63 GB/s, because load_duration covers host-to-device transfer and model init as well as the file read -- the page cache itself reads at 6.4 GB/s. The 5 GB/s bar was therefore unreachable, and every warm load was being reported as a partial hit. Now 2.0 / 0.8 GB/s, either side of the measured 6.9x separation. Warm-skip was also unsafe. A 12.87GB blob was skipped as already resident on the strength of twelve 2MB probe windows, then loaded at 2.44 GB/s. Skipping now requires warm_confident: an exact cachestat reading, or a probe finding every one of 32 denser samples resident. warm_file_to_ram/warm_ollama_blob take force=True, exposed on the warm-model endpoint, whose Pydantic model was missing the field entirely. MCP parity: the server had drifted well behind the REST API. Adds tools for measured residency, warm planning, VRAM requests, per-profile analytics, thermal governor control, overclock status/apply/restore, and autotune sweeps plus status -- 23 tools and 6 resources, up from 12 and 3. The telemetry store now starts in __main__ rather than at import scope, since server.py imports this module for the benchmark tool. README: replaced the remaining theoretical claims (31.5 GB/s bus rate, sub-1.5s loads, 15ms yields) with the measured numbers. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -143,7 +143,7 @@ PROBE_WINDOW_BYTES = 2 * 1024 * 1024
|
||||
PROBE_CACHED_GBPS = 1.5
|
||||
|
||||
|
||||
def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
|
||||
def _throughput_probe(fd: int, size: int, windows_override: Optional[int] = None) -> Dict[str, Any]:
|
||||
"""Infer residency by timing reads of small windows spread across the file.
|
||||
|
||||
Used only where cachestat is not permitted (Ollama's blobs are owned by uid `ollama`).
|
||||
@@ -158,7 +158,7 @@ def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
|
||||
pollution the probe itself created, and leaving them behind would slowly warm the
|
||||
cache with data nobody asked for.
|
||||
"""
|
||||
windows = min(PROBE_WINDOWS, max(int(size // PROBE_WINDOW_BYTES), 1))
|
||||
windows = min(windows_override or PROBE_WINDOWS, max(int(size // PROBE_WINDOW_BYTES), 1))
|
||||
if windows <= 0:
|
||||
return {"resident_pct": 0.0, "windows": 0}
|
||||
|
||||
@@ -194,7 +194,8 @@ def _throughput_probe(fd: int, size: int) -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
|
||||
def page_residency(filepath: str, allow_probe: bool = True,
|
||||
probe_windows: Optional[int] = None) -> Dict[str, Any]:
|
||||
"""Measure what fraction of a file is resident in the Linux page cache."""
|
||||
try:
|
||||
size = os.path.getsize(filepath)
|
||||
@@ -216,7 +217,7 @@ def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
|
||||
method, measurable = "cachestat", True
|
||||
extra = {"dirty_pages": cs.nr_dirty, "evicted_pages": cs.nr_evicted}
|
||||
elif allow_probe:
|
||||
probe = _throughput_probe(fd, size)
|
||||
probe = _throughput_probe(fd, size, probe_windows)
|
||||
pct = probe["resident_pct"]
|
||||
method, measurable = "probe", True
|
||||
extra = {"probe_windows": probe["windows"], "probe_median_gbps": probe.get("median_gbps")}
|
||||
@@ -236,6 +237,11 @@ def page_residency(filepath: str, allow_probe: bool = True) -> Dict[str, Any]:
|
||||
"method": method,
|
||||
"measurable": measurable,
|
||||
"warm": pct >= WARM_SKIP_THRESHOLD_PCT,
|
||||
# Only an exact measurement is trustworthy enough to skip work on. A probe of a
|
||||
# dozen 2 MB windows can clear 90% on a file that is mostly cold -- observed
|
||||
# here as a 12.87 GB "already resident" blob that then loaded at 2.44 GB/s.
|
||||
"warm_confident": (method == "cachestat" and pct >= WARM_SKIP_THRESHOLD_PCT)
|
||||
or (method == "probe" and pct >= 100.0),
|
||||
**extra,
|
||||
}
|
||||
except Exception as e:
|
||||
@@ -468,16 +474,19 @@ def get_cache_report(include_files: bool = True, force_refresh: bool = False) ->
|
||||
# ---------------------------------------------------------------- warming
|
||||
|
||||
def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024,
|
||||
skip_if_warm: bool = True) -> Dict[str, Any]:
|
||||
"""Pre-fault a file into the Linux page cache, skipping it if already resident."""
|
||||
skip_if_warm: bool = True, force: bool = False) -> Dict[str, Any]:
|
||||
"""Pre-fault a file into the Linux page cache, skipping it only if confidently resident."""
|
||||
if not os.path.exists(filepath):
|
||||
return {"success": False, "error": f"File not found: {filepath}", "duration_ms": 0}
|
||||
|
||||
before = page_residency(filepath)
|
||||
if skip_if_warm and before.get("warm"):
|
||||
# Probe densely here: this decision skips real work, so it is worth 32 samples
|
||||
# rather than 12.
|
||||
before = page_residency(filepath, probe_windows=32)
|
||||
if skip_if_warm and not force and before.get("warm_confident"):
|
||||
return {
|
||||
"success": True, "filepath": filepath, "skipped": True,
|
||||
"reason": "already resident", "resident_pct": before.get("resident_pct"),
|
||||
"method": before.get("method"),
|
||||
"size_mb": round(before.get("size_bytes", 0) / (1024**2), 2),
|
||||
"duration_ms": 0.0, "bytes_read": 0,
|
||||
}
|
||||
@@ -499,7 +508,7 @@ def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024,
|
||||
bytes_read += n
|
||||
|
||||
duration = time.perf_counter() - t0
|
||||
after = page_residency(filepath)
|
||||
after = page_residency(filepath, probe_windows=32)
|
||||
return {
|
||||
"success": True,
|
||||
"filepath": filepath,
|
||||
@@ -551,11 +560,11 @@ async def warm_ollama_model(model_name: str, keep_alive: str = "5m") -> Dict[str
|
||||
"duration_ms": round((time.perf_counter() - t0) * 1000, 2)}
|
||||
|
||||
|
||||
def warm_ollama_blob(model_name: str) -> Dict[str, Any]:
|
||||
def warm_ollama_blob(model_name: str, force: bool = False) -> Dict[str, Any]:
|
||||
"""Warm a specific Ollama model's GGUF into page cache without touching VRAM."""
|
||||
for f in find_ollama_model_files():
|
||||
if f["model"] == model_name:
|
||||
res = warm_file_to_ram(f["full_path"])
|
||||
res = warm_file_to_ram(f["full_path"], force=force)
|
||||
res["model"] = model_name
|
||||
return res
|
||||
return {"success": False, "error": f"no blob found for model '{model_name}'"}
|
||||
@@ -603,13 +612,13 @@ def build_warm_plan(budget_gb: Optional[float] = None) -> Dict[str, Any]:
|
||||
if c["full_path"] in seen_paths:
|
||||
continue
|
||||
seen_paths.add(c["full_path"])
|
||||
res = page_residency(c["full_path"])
|
||||
res = page_residency(c["full_path"], probe_windows=32)
|
||||
entry = {
|
||||
"name": c["name"], "kind": c["kind"], "full_path": c["full_path"],
|
||||
"size_gb": c.get("size_gb", 0), "score": round(c["score"], 4),
|
||||
"resident_pct": res.get("resident_pct", 0.0),
|
||||
}
|
||||
if res.get("warm"):
|
||||
if res.get("warm_confident"):
|
||||
entry["action"] = "already-warm"
|
||||
skipped.append(entry)
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user