Nine changes, in rough order of how much they affect real behaviour: 1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload; measured here, the HTTP call returns in 63ms while the driver takes a further 77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up allocating into VRAM that is still occupied. instant_free_ollama_vram() polls NVML until the allocation is actually gone and reports request/confirm split. 2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full checkpoint reload on each workflow iteration. It is held for 30s of genuinely empty queue, with an immediate purge when Ollama actually asks for the memory. 3. Cache-hit classification uses achieved bandwidth (size / load duration) rather than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit. 4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB resident on a box with 46GB of page cache: the kernel only permits page-cache introspection on files you own, and the Ollama blobs are owned by uid ollama, for which mincore answers "all resident" instead of failing. Uses cachestat(2) where permitted and a randomised read-rate probe elsewhere, labelling which was used. Fixed-offset probing was self-fulfilling, so windows are random and cold ones are returned with FADV_DONTNEED. 5. Warming is budgeted and ranked by recency/frequency instead of reading every file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first. 6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a 50-entry in-memory deque, so /api/analytics/profiles can finally answer whether an overclock profile actually delivers more tok/s. 7. Thermal governor walks the overclock back on sustained heat or hardware throttling, with hysteresis, fed from the existing sampler. 8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid errors and degenerate output, and restores the profile in a finally block. 9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost. Nothing previously undid a locked clock or a manually pinned fan. Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than every client re-running the whole snapshot; wall-clock timestamps in place of the event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
264 lines
11 KiB
Python
264 lines
11 KiB
Python
"""Closed-loop overclock autotuner.
|
|
|
|
The profiles in this repo were hand-tuned and had already drifted apart from the defaults
|
|
in overclock_manager.py, with no record of which numbers were actually faster. This module
|
|
answers that empirically: it walks a clock offset upward, measures real decode throughput
|
|
at each step, watches for instability, and reports the highest setting that was both
|
|
stable and fastest.
|
|
|
|
Safety properties:
|
|
* The original profile is always restored, including on exception or cancellation.
|
|
* A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's
|
|
render by yanking clocks mid-graph.
|
|
* Every step is bounded by a temperature ceiling and checked for kernel Xid messages,
|
|
and the sweep stops climbing the moment a step looks unstable.
|
|
"""
|
|
import asyncio
|
|
import logging
|
|
import subprocess
|
|
import time
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
import overclock_manager
|
|
import telemetry_store
|
|
import vram_arbitrator
|
|
|
|
logger = logging.getLogger("autotune")
|
|
|
|
BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging "
|
|
"works in a modern operating system kernel.")
|
|
BENCH_TOKENS = 160
|
|
SETTLE_S = 2.5
|
|
TEMP_CEILING_C = 84.0
|
|
|
|
KNOBS = {
|
|
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100},
|
|
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25},
|
|
}
|
|
|
|
|
|
def _xid_since(since_ts: float) -> List[str]:
|
|
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
|
|
try:
|
|
since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts))
|
|
proc = subprocess.run(
|
|
["journalctl", "-k", "--since", since, "--no-pager", "-q"],
|
|
capture_output=True, text=True, timeout=10,
|
|
)
|
|
return [ln.strip() for ln in proc.stdout.splitlines()
|
|
if "Xid" in ln or "NVRM:" in ln]
|
|
except Exception as e:
|
|
logger.debug(f"Xid check unavailable: {e}")
|
|
return []
|
|
|
|
|
|
async def _decode_benchmark(model: str) -> Dict[str, Any]:
|
|
"""One fixed decode run. Throughput here is the thing being optimised."""
|
|
client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0)
|
|
t0 = time.perf_counter()
|
|
resp = await client.post("/api/generate", json={
|
|
"model": model,
|
|
"prompt": BENCH_PROMPT,
|
|
"stream": False,
|
|
"keep_alive": "10m",
|
|
"options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42},
|
|
})
|
|
wall_ms = round((time.perf_counter() - t0) * 1000, 2)
|
|
if resp.status_code != 200:
|
|
return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}",
|
|
"wall_ms": wall_ms}
|
|
data = resp.json()
|
|
eval_ms = data.get("eval_duration", 0) / 1e6
|
|
eval_count = data.get("eval_count", 0)
|
|
text = data.get("response", "") or ""
|
|
return {
|
|
"ok": True,
|
|
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
|
|
"eval_count": eval_count,
|
|
"eval_ms": round(eval_ms, 2),
|
|
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
|
|
"wall_ms": wall_ms,
|
|
"response_chars": len(text),
|
|
# A model producing almost nothing, or pure repetition, is a corruption signal.
|
|
"degenerate": eval_count < BENCH_TOKENS * 0.5 or len(set(text.split())) < 8,
|
|
}
|
|
|
|
|
|
class SweepState:
|
|
def __init__(self) -> None:
|
|
self.running = False
|
|
self.cancel = False
|
|
self.current: Optional[Dict[str, Any]] = None
|
|
self.last_result: Optional[Dict[str, Any]] = None
|
|
|
|
|
|
state = SweepState()
|
|
|
|
|
|
async def sweep(knob: str = "mem_offset_mhz",
|
|
profile: str = "ollama",
|
|
model: Optional[str] = None,
|
|
start: Optional[int] = None,
|
|
stop: Optional[int] = None,
|
|
step: Optional[int] = None,
|
|
repeats: int = 1,
|
|
apply_best: bool = False) -> Dict[str, Any]:
|
|
"""Sweep one clock offset and return the fastest stable value."""
|
|
if knob not in KNOBS:
|
|
return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"}
|
|
if state.running:
|
|
return {"success": False, "error": "a sweep is already running"}
|
|
|
|
comfy = await vram_arbitrator.get_comfyui_live_state()
|
|
if comfy.get("executing") or comfy.get("queue_remaining"):
|
|
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
|
|
|
|
if not model:
|
|
ollama = await vram_arbitrator.get_ollama_live_state()
|
|
model = ollama.get("active_model_name")
|
|
if not model:
|
|
installed = ollama.get("installed_models") or []
|
|
if not installed:
|
|
return {"success": False, "error": "no Ollama model available to benchmark"}
|
|
model = installed[0].get("name")
|
|
|
|
defaults = KNOBS[knob]
|
|
start = defaults["default_start"] if start is None else start
|
|
stop = defaults["default_stop"] if stop is None else stop
|
|
step = defaults["default_step"] if step is None else step
|
|
if step <= 0 or stop < start:
|
|
return {"success": False, "error": "invalid sweep range"}
|
|
|
|
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
|
|
state.running = True
|
|
state.cancel = False
|
|
results: List[Dict[str, Any]] = []
|
|
t_start = time.time()
|
|
|
|
try:
|
|
# Load the model once up front so the first step does not pay the load cost.
|
|
await _decode_benchmark(model)
|
|
|
|
value = start
|
|
while value <= stop and not state.cancel:
|
|
overclock_manager.apply_profile(profile, overrides={knob: value})
|
|
await asyncio.sleep(SETTLE_S)
|
|
step_started = time.time()
|
|
|
|
samples = []
|
|
for _ in range(max(repeats, 1)):
|
|
samples.append(await _decode_benchmark(model))
|
|
if state.cancel:
|
|
break
|
|
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
xids = _xid_since(step_started)
|
|
ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")]
|
|
temp = gpu.get("temperature_c", 0) or 0
|
|
|
|
instability = []
|
|
if xids:
|
|
instability.append(f"kernel Xid: {xids[0][:120]}")
|
|
if len(ok_samples) < len(samples):
|
|
instability.append("benchmark failed or produced degenerate output")
|
|
if temp >= TEMP_CEILING_C:
|
|
instability.append(f"temperature ceiling hit ({temp}°C)")
|
|
|
|
tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2)
|
|
row = {
|
|
"knob": knob,
|
|
"value": value,
|
|
"profile": profile,
|
|
"model": model,
|
|
"tokens_per_sec": tok_s,
|
|
"temp_c": temp,
|
|
"power_w": gpu.get("power_w"),
|
|
"clock_sm_mhz": gpu.get("clock_graphics_mhz"),
|
|
"clock_mem_mhz": gpu.get("clock_mem_mhz"),
|
|
"throttle_reasons": gpu.get("throttle_reasons"),
|
|
"stable": not instability,
|
|
"instability": "; ".join(instability) or None,
|
|
"samples": samples,
|
|
}
|
|
results.append(row)
|
|
telemetry_store.record_autotune({
|
|
"profile": profile, "knob": knob,
|
|
"core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"),
|
|
"mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"),
|
|
"tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"),
|
|
"stable": row["stable"], "instability": row["instability"],
|
|
"note": f"sweep {knob} {start}..{stop} step {step}",
|
|
})
|
|
state.current = {"knob": knob, "value": value, "stop": stop,
|
|
"tokens_per_sec": tok_s, "stable": row["stable"]}
|
|
logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, "
|
|
f"stable={row['stable']} {row['instability'] or ''}")
|
|
|
|
if not row["stable"]:
|
|
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
|
|
break
|
|
value += step
|
|
|
|
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
|
|
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
|
|
baseline = next((r for r in results if r["value"] == start), None)
|
|
gain_pct = None
|
|
if best and baseline and baseline["tokens_per_sec"] > 0:
|
|
gain_pct = round((best["tokens_per_sec"] / baseline["tokens_per_sec"] - 1) * 100, 2)
|
|
|
|
applied = None
|
|
if apply_best and best:
|
|
overclock_manager.set_profile(profile, {knob: best["value"]})
|
|
applied = {knob: best["value"], "profile": profile}
|
|
logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'")
|
|
|
|
result = {
|
|
"success": True,
|
|
"knob": knob,
|
|
"profile": profile,
|
|
"model": model,
|
|
"range": {"start": start, "stop": stop, "step": step},
|
|
"steps_run": len(results),
|
|
"duration_s": round(time.time() - t_start, 1),
|
|
"cancelled": state.cancel,
|
|
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
|
|
"clock_sm_mhz")} if best else None,
|
|
"baseline_tokens_per_sec": baseline["tokens_per_sec"] if baseline else None,
|
|
"gain_pct": gain_pct,
|
|
"applied_to_profile": applied,
|
|
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
|
|
for r in results if not r["stable"]), None),
|
|
"table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w",
|
|
"clock_mem_mhz", "clock_sm_mhz", "stable",
|
|
"instability")} for r in results],
|
|
}
|
|
state.last_result = result
|
|
return result
|
|
finally:
|
|
# Always hand the card back exactly as we found it.
|
|
state.running = False
|
|
state.current = None
|
|
try:
|
|
overclock_manager.apply_profile(profile)
|
|
logger.info(f"autotune restored profile '{profile}'")
|
|
except Exception as e:
|
|
logger.error(f"autotune failed to restore profile, forcing stock: {e}")
|
|
overclock_manager.restore_safe("autotune restore failed")
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
return {
|
|
"running": state.running,
|
|
"current": state.current,
|
|
"last_result": state.last_result,
|
|
"knobs": KNOBS,
|
|
"history": telemetry_store.autotune_history(100),
|
|
}
|
|
|
|
|
|
def cancel() -> Dict[str, Any]:
|
|
if not state.running:
|
|
return {"cancelled": False, "reason": "no sweep running"}
|
|
state.cancel = True
|
|
return {"cancelled": True}
|