Files
gpu-program-swapper/autotune.py
drjones 63297dc49d Verify autotune knobs against hardware before sweeping
Sweeping a knob the driver ignores measures nothing but benchmark noise, and the
tuner would then confidently report 'best = the highest value tried'. On this box
(driver 595.84) nvidia-settings accepts GPUGraphicsClockOffset/GPUMemoryTransferRate
Offset and silently discards them: assigning 0 reports success and reads back 250.
The ollama profile's core_offset_mhz=35 and mem_offset_mhz=200 have therefore been
doing nothing.

- _knob_effective() applies a probe value and confirms the hardware actually moved
  before any sweep starts, choosing the candidate furthest from the current reading
  (probing with the maximum fails when the card already sits at its top clock).
- Adds discrete clock-lock knobs (lock_mem_mhz, lock_core_max) driven by the card's
  own supported-clock list, since -lmc/-lgc do work where offsets do not.
- Reasoning models return their output in 'thinking' with an empty 'response', which
  the degeneracy check was flagging as corruption. Token count is now the primary
  signal.
- Separates gain-vs-current-setting from gain-vs-slowest-value-tried. Reporting the
  latter as 'gain vs baseline' implied a +102% speedup that nobody would observe.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-28 09:10:06 -07:00

396 lines
18 KiB
Python

"""Closed-loop overclock autotuner.
The profiles in this repo were hand-tuned and had already drifted apart from the defaults
in overclock_manager.py, with no record of which numbers were actually faster. This module
answers that empirically: it walks a clock offset upward, measures real decode throughput
at each step, watches for instability, and reports the highest setting that was both
stable and fastest.
Safety properties:
* The original profile is always restored, including on exception or cancellation.
* A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's
render by yanking clocks mid-graph.
* Every step is bounded by a temperature ceiling and checked for kernel Xid messages,
and the sweep stops climbing the moment a step looks unstable.
"""
import asyncio
import logging
import subprocess
import time
from typing import Any, Dict, List, Optional
import overclock_manager
import telemetry_store
import vram_arbitrator
logger = logging.getLogger("autotune")
BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging "
"works in a modern operating system kernel.")
BENCH_TOKENS = 160
SETTLE_S = 2.5
TEMP_CEILING_C = 84.0
KNOBS = {
# Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
# attribute is accepted and then silently ignored -- assigning 0 reports success and
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
# checks before any sweep runs.
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
"kind": "offset", "verify": "mem_offset"},
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
"kind": "offset", "verify": "core_offset"},
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
"values": None}, # filled from the card's supported clock list
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
"values": None},
}
def _supported_clocks(which: str = "mem") -> List[int]:
"""Discrete clock values the card will actually accept for -lmc / -lgc."""
try:
proc = subprocess.run(
["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
col = 0 if which == "mem" else 1
vals = set()
for line in proc.stdout.splitlines():
parts = [p.strip() for p in line.split(",")]
if len(parts) > col and parts[col].isdigit():
vals.add(int(parts[col]))
return sorted(vals)
except Exception as e:
logger.debug(f"supported clock query failed: {e}")
return []
def _read_hw(field: str) -> Optional[float]:
"""Read back the hardware state a knob is supposed to move."""
gpu = vram_arbitrator.get_gpu_hardware_stats()
if field == "clock_mem":
return gpu.get("clock_mem_mhz")
if field == "clock_sm":
return gpu.get("clock_graphics_mhz")
if field in ("mem_offset", "core_offset"):
r = overclock_manager._nvidia_settings(
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
for line in (r.get("out") or "").splitlines():
if "Attribute" in line and "):" in line:
try:
return float(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
return None
def _knob_effective(knob: str, profile: str, candidates: List[int],
baseline_value: int) -> Dict[str, Any]:
"""Verify a knob actually moves the hardware before we sweep it.
Without this the tuner happily reports "best = the highest value tried" from run-to-run
benchmark noise on a knob the driver is ignoring.
The probe value is chosen as the candidate *furthest* from where the hardware currently
sits. Probing with the maximum is not good enough: if the card already happens to be at
its top clock, setting it there again moves nothing and a perfectly good knob looks
broken.
"""
field = KNOBS[knob].get("verify")
before = _read_hw(field)
if before is not None and candidates:
probe_value = max(candidates, key=lambda v: abs(v - before))
else:
probe_value = candidates[-1] if candidates else 0
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
time.sleep(2.0)
after = _read_hw(field)
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
return {
"effective": bool(moved),
"field": field,
"before": before,
"after": after,
"probe_value": probe_value,
"detail": (f"{field} moved {before} -> {after}" if moved else
f"{field} stayed at {after} after setting {knob}={probe_value}; "
f"this driver accepts the setting and ignores it"),
}
def _xid_since(since_ts: float) -> List[str]:
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
try:
since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts))
proc = subprocess.run(
["journalctl", "-k", "--since", since, "--no-pager", "-q"],
capture_output=True, text=True, timeout=10,
)
return [ln.strip() for ln in proc.stdout.splitlines()
if "Xid" in ln or "NVRM:" in ln]
except Exception as e:
logger.debug(f"Xid check unavailable: {e}")
return []
async def _decode_benchmark(model: str) -> Dict[str, Any]:
"""One fixed decode run. Throughput here is the thing being optimised."""
client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0)
t0 = time.perf_counter()
resp = await client.post("/api/generate", json={
"model": model,
"prompt": BENCH_PROMPT,
"stream": False,
"keep_alive": "10m",
"options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42},
})
wall_ms = round((time.perf_counter() - t0) * 1000, 2)
if resp.status_code != 200:
return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}",
"wall_ms": wall_ms}
data = resp.json()
eval_ms = data.get("eval_duration", 0) / 1e6
eval_count = data.get("eval_count", 0)
# Reasoning models put their output in `thinking` and leave `response` empty, so a
# check that only looked at `response` flagged every one of them as degenerate.
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
return {
"ok": True,
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
"eval_count": eval_count,
"eval_ms": round(eval_ms, 2),
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
"wall_ms": wall_ms,
"response_chars": len(text),
# Corruption signals: the model stopped far short of the token budget, or it
# produced text that is pure repetition. Token count is the primary signal --
# empty text alone is not enough, since output can arrive in other fields.
"degenerate": (eval_count < BENCH_TOKENS * 0.5
or (len(text) > 0 and len(set(text.split())) < 8)),
}
class SweepState:
def __init__(self) -> None:
self.running = False
self.cancel = False
self.current: Optional[Dict[str, Any]] = None
self.last_result: Optional[Dict[str, Any]] = None
state = SweepState()
async def sweep(knob: str = "mem_offset_mhz",
profile: str = "ollama",
model: Optional[str] = None,
start: Optional[int] = None,
stop: Optional[int] = None,
step: Optional[int] = None,
repeats: int = 1,
apply_best: bool = False) -> Dict[str, Any]:
"""Sweep one clock offset and return the fastest stable value."""
if knob not in KNOBS:
return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"}
if state.running:
return {"success": False, "error": "a sweep is already running"}
comfy = await vram_arbitrator.get_comfyui_live_state()
if comfy.get("executing") or comfy.get("queue_remaining"):
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
if not model:
ollama = await vram_arbitrator.get_ollama_live_state()
model = ollama.get("active_model_name")
if not model:
installed = ollama.get("installed_models") or []
if not installed:
return {"success": False, "error": "no Ollama model available to benchmark"}
model = installed[0].get("name")
defaults = KNOBS[knob]
if defaults.get("kind") == "discrete":
supported = defaults.get("values") or _supported_clocks(
"mem" if knob == "lock_mem_mhz" else "gr")
if not supported:
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
values = [v for v in supported
if (start is None or v >= start) and (stop is None or v <= stop)]
if not values:
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
start, stop, step = values[0], values[-1], None
else:
start = defaults["default_start"] if start is None else start
stop = defaults["default_stop"] if stop is None else stop
step = defaults["default_step"] if step is None else step
if step <= 0 or stop < start:
return {"success": False, "error": "invalid sweep range"}
values = list(range(start, stop + 1, step))
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
# Refuse to sweep a knob the driver is going to ignore.
effectiveness = _knob_effective(knob, profile, values, baseline_value)
if not effectiveness["effective"]:
return {
"success": False,
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
"effectiveness": effectiveness,
}
state.running = True
state.cancel = False
results: List[Dict[str, Any]] = []
t_start = time.time()
try:
# Load the model once up front so the first step does not pay the load cost.
await _decode_benchmark(model)
for value in values:
if state.cancel:
break
overclock_manager.apply_profile(profile, overrides={knob: value})
await asyncio.sleep(SETTLE_S)
step_started = time.time()
samples = []
for _ in range(max(repeats, 1)):
samples.append(await _decode_benchmark(model))
if state.cancel:
break
gpu = vram_arbitrator.get_gpu_hardware_stats()
xids = _xid_since(step_started)
ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")]
temp = gpu.get("temperature_c", 0) or 0
instability = []
if xids:
instability.append(f"kernel Xid: {xids[0][:120]}")
if len(ok_samples) < len(samples):
instability.append("benchmark failed or produced degenerate output")
if temp >= TEMP_CEILING_C:
instability.append(f"temperature ceiling hit ({temp}°C)")
tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2)
row = {
"knob": knob,
"value": value,
"profile": profile,
"model": model,
"tokens_per_sec": tok_s,
"temp_c": temp,
"power_w": gpu.get("power_w"),
"clock_sm_mhz": gpu.get("clock_graphics_mhz"),
"clock_mem_mhz": gpu.get("clock_mem_mhz"),
"throttle_reasons": gpu.get("throttle_reasons"),
"stable": not instability,
"instability": "; ".join(instability) or None,
"samples": samples,
}
results.append(row)
telemetry_store.record_autotune({
"profile": profile, "knob": knob,
"core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"),
"mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"),
"tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"),
"stable": row["stable"], "instability": row["instability"],
"note": f"sweep {knob} {start}..{stop} step {step}",
})
state.current = {"knob": knob, "value": value, "stop": stop,
"tokens_per_sec": tok_s, "stable": row["stable"]}
logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, "
f"stable={row['stable']} {row['instability'] or ''}")
if not row["stable"]:
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
break
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
# Two different comparisons, and conflating them is how a sweep oversells itself.
# The first step of the range is not "baseline" unless it happens to be what the
# profile is already set to -- reporting "+102%" against the slowest value tried
# implies a speedup the user would not actually observe.
first_step = next((r for r in results if r["value"] == values[0]), None)
current = next((r for r in results if r["value"] == baseline_value), None)
gain_vs_first_step_pct = None
if best and first_step and first_step["tokens_per_sec"] > 0:
gain_vs_first_step_pct = round(
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
gain_vs_current_pct = None
if best and current and current["tokens_per_sec"] > 0:
gain_vs_current_pct = round(
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
applied = None
if apply_best and best:
overclock_manager.set_profile(profile, {knob: best["value"]})
applied = {knob: best["value"], "profile": profile}
logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'")
result = {
"success": True,
"knob": knob,
"profile": profile,
"model": model,
"range": {"start": start, "stop": stop, "step": step, "values": values},
"effectiveness": effectiveness,
"steps_run": len(results),
"duration_s": round(time.time() - t_start, 1),
"cancelled": state.cancel,
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
"clock_sm_mhz")} if best else None,
"current_profile_value": baseline_value,
"first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
"gain_vs_first_step_pct": gain_vs_first_step_pct,
"gain_vs_current_pct": gain_vs_current_pct,
"gain_note": ("compared against the profile's current setting"
if current is not None else
f"the profile's current value ({baseline_value}) was not in the "
f"swept range, so only the spread across tried values is shown"),
"applied_to_profile": applied,
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
for r in results if not r["stable"]), None),
"table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w",
"clock_mem_mhz", "clock_sm_mhz", "stable",
"instability")} for r in results],
}
state.last_result = result
return result
finally:
# Always hand the card back exactly as we found it.
state.running = False
state.current = None
try:
overclock_manager.apply_profile(profile)
logger.info(f"autotune restored profile '{profile}'")
except Exception as e:
logger.error(f"autotune failed to restore profile, forcing stock: {e}")
overclock_manager.restore_safe("autotune restore failed")
def get_status() -> Dict[str, Any]:
return {
"running": state.running,
"current": state.current,
"last_result": state.last_result,
"knobs": KNOBS,
"history": telemetry_store.autotune_history(100),
}
def cancel() -> Dict[str, Any]:
if not state.running:
return {"cancelled": False, "reason": "no sweep running"}
state.cancel = True
return {"cancelled": True}