Sweeping a knob the driver ignores measures nothing but benchmark noise, and the tuner would then confidently report 'best = the highest value tried'. On this box (driver 595.84) nvidia-settings accepts GPUGraphicsClockOffset/GPUMemoryTransferRate Offset and silently discards them: assigning 0 reports success and reads back 250. The ollama profile's core_offset_mhz=35 and mem_offset_mhz=200 have therefore been doing nothing. - _knob_effective() applies a probe value and confirms the hardware actually moved before any sweep starts, choosing the candidate furthest from the current reading (probing with the maximum fails when the card already sits at its top clock). - Adds discrete clock-lock knobs (lock_mem_mhz, lock_core_max) driven by the card's own supported-clock list, since -lmc/-lgc do work where offsets do not. - Reasoning models return their output in 'thinking' with an empty 'response', which the degeneracy check was flagging as corruption. Token count is now the primary signal. - Separates gain-vs-current-setting from gain-vs-slowest-value-tried. Reporting the latter as 'gain vs baseline' implied a +102% speedup that nobody would observe. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
396 lines
18 KiB
Python
396 lines
18 KiB
Python
"""Closed-loop overclock autotuner.
|
|
|
|
The profiles in this repo were hand-tuned and had already drifted apart from the defaults
|
|
in overclock_manager.py, with no record of which numbers were actually faster. This module
|
|
answers that empirically: it walks a clock offset upward, measures real decode throughput
|
|
at each step, watches for instability, and reports the highest setting that was both
|
|
stable and fastest.
|
|
|
|
Safety properties:
|
|
* The original profile is always restored, including on exception or cancellation.
|
|
* A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's
|
|
render by yanking clocks mid-graph.
|
|
* Every step is bounded by a temperature ceiling and checked for kernel Xid messages,
|
|
and the sweep stops climbing the moment a step looks unstable.
|
|
"""
|
|
import asyncio
|
|
import logging
|
|
import subprocess
|
|
import time
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
import overclock_manager
|
|
import telemetry_store
|
|
import vram_arbitrator
|
|
|
|
logger = logging.getLogger("autotune")
|
|
|
|
BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging "
|
|
"works in a modern operating system kernel.")
|
|
BENCH_TOKENS = 160
|
|
SETTLE_S = 2.5
|
|
TEMP_CEILING_C = 84.0
|
|
|
|
KNOBS = {
|
|
# Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
|
|
# attribute is accepted and then silently ignored -- assigning 0 reports success and
|
|
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
|
|
# checks before any sweep runs.
|
|
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
|
|
"kind": "offset", "verify": "mem_offset"},
|
|
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
|
|
"kind": "offset", "verify": "core_offset"},
|
|
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
|
|
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
|
|
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
|
|
"values": None}, # filled from the card's supported clock list
|
|
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
|
|
"values": None},
|
|
}
|
|
|
|
|
|
def _supported_clocks(which: str = "mem") -> List[int]:
|
|
"""Discrete clock values the card will actually accept for -lmc / -lgc."""
|
|
try:
|
|
proc = subprocess.run(
|
|
["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}",
|
|
"--format=csv,noheader,nounits"],
|
|
capture_output=True, text=True, timeout=10)
|
|
col = 0 if which == "mem" else 1
|
|
vals = set()
|
|
for line in proc.stdout.splitlines():
|
|
parts = [p.strip() for p in line.split(",")]
|
|
if len(parts) > col and parts[col].isdigit():
|
|
vals.add(int(parts[col]))
|
|
return sorted(vals)
|
|
except Exception as e:
|
|
logger.debug(f"supported clock query failed: {e}")
|
|
return []
|
|
|
|
|
|
def _read_hw(field: str) -> Optional[float]:
|
|
"""Read back the hardware state a knob is supposed to move."""
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
if field == "clock_mem":
|
|
return gpu.get("clock_mem_mhz")
|
|
if field == "clock_sm":
|
|
return gpu.get("clock_graphics_mhz")
|
|
if field in ("mem_offset", "core_offset"):
|
|
r = overclock_manager._nvidia_settings(
|
|
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
|
|
for line in (r.get("out") or "").splitlines():
|
|
if "Attribute" in line and "):" in line:
|
|
try:
|
|
return float(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _knob_effective(knob: str, profile: str, candidates: List[int],
|
|
baseline_value: int) -> Dict[str, Any]:
|
|
"""Verify a knob actually moves the hardware before we sweep it.
|
|
|
|
Without this the tuner happily reports "best = the highest value tried" from run-to-run
|
|
benchmark noise on a knob the driver is ignoring.
|
|
|
|
The probe value is chosen as the candidate *furthest* from where the hardware currently
|
|
sits. Probing with the maximum is not good enough: if the card already happens to be at
|
|
its top clock, setting it there again moves nothing and a perfectly good knob looks
|
|
broken.
|
|
"""
|
|
field = KNOBS[knob].get("verify")
|
|
before = _read_hw(field)
|
|
if before is not None and candidates:
|
|
probe_value = max(candidates, key=lambda v: abs(v - before))
|
|
else:
|
|
probe_value = candidates[-1] if candidates else 0
|
|
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
|
|
time.sleep(2.0)
|
|
after = _read_hw(field)
|
|
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
|
|
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
|
|
return {
|
|
"effective": bool(moved),
|
|
"field": field,
|
|
"before": before,
|
|
"after": after,
|
|
"probe_value": probe_value,
|
|
"detail": (f"{field} moved {before} -> {after}" if moved else
|
|
f"{field} stayed at {after} after setting {knob}={probe_value}; "
|
|
f"this driver accepts the setting and ignores it"),
|
|
}
|
|
|
|
|
|
def _xid_since(since_ts: float) -> List[str]:
|
|
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
|
|
try:
|
|
since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts))
|
|
proc = subprocess.run(
|
|
["journalctl", "-k", "--since", since, "--no-pager", "-q"],
|
|
capture_output=True, text=True, timeout=10,
|
|
)
|
|
return [ln.strip() for ln in proc.stdout.splitlines()
|
|
if "Xid" in ln or "NVRM:" in ln]
|
|
except Exception as e:
|
|
logger.debug(f"Xid check unavailable: {e}")
|
|
return []
|
|
|
|
|
|
async def _decode_benchmark(model: str) -> Dict[str, Any]:
|
|
"""One fixed decode run. Throughput here is the thing being optimised."""
|
|
client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0)
|
|
t0 = time.perf_counter()
|
|
resp = await client.post("/api/generate", json={
|
|
"model": model,
|
|
"prompt": BENCH_PROMPT,
|
|
"stream": False,
|
|
"keep_alive": "10m",
|
|
"options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42},
|
|
})
|
|
wall_ms = round((time.perf_counter() - t0) * 1000, 2)
|
|
if resp.status_code != 200:
|
|
return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}",
|
|
"wall_ms": wall_ms}
|
|
data = resp.json()
|
|
eval_ms = data.get("eval_duration", 0) / 1e6
|
|
eval_count = data.get("eval_count", 0)
|
|
# Reasoning models put their output in `thinking` and leave `response` empty, so a
|
|
# check that only looked at `response` flagged every one of them as degenerate.
|
|
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
|
|
return {
|
|
"ok": True,
|
|
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
|
|
"eval_count": eval_count,
|
|
"eval_ms": round(eval_ms, 2),
|
|
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
|
|
"wall_ms": wall_ms,
|
|
"response_chars": len(text),
|
|
# Corruption signals: the model stopped far short of the token budget, or it
|
|
# produced text that is pure repetition. Token count is the primary signal --
|
|
# empty text alone is not enough, since output can arrive in other fields.
|
|
"degenerate": (eval_count < BENCH_TOKENS * 0.5
|
|
or (len(text) > 0 and len(set(text.split())) < 8)),
|
|
}
|
|
|
|
|
|
class SweepState:
|
|
def __init__(self) -> None:
|
|
self.running = False
|
|
self.cancel = False
|
|
self.current: Optional[Dict[str, Any]] = None
|
|
self.last_result: Optional[Dict[str, Any]] = None
|
|
|
|
|
|
state = SweepState()
|
|
|
|
|
|
async def sweep(knob: str = "mem_offset_mhz",
|
|
profile: str = "ollama",
|
|
model: Optional[str] = None,
|
|
start: Optional[int] = None,
|
|
stop: Optional[int] = None,
|
|
step: Optional[int] = None,
|
|
repeats: int = 1,
|
|
apply_best: bool = False) -> Dict[str, Any]:
|
|
"""Sweep one clock offset and return the fastest stable value."""
|
|
if knob not in KNOBS:
|
|
return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"}
|
|
if state.running:
|
|
return {"success": False, "error": "a sweep is already running"}
|
|
|
|
comfy = await vram_arbitrator.get_comfyui_live_state()
|
|
if comfy.get("executing") or comfy.get("queue_remaining"):
|
|
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
|
|
|
|
if not model:
|
|
ollama = await vram_arbitrator.get_ollama_live_state()
|
|
model = ollama.get("active_model_name")
|
|
if not model:
|
|
installed = ollama.get("installed_models") or []
|
|
if not installed:
|
|
return {"success": False, "error": "no Ollama model available to benchmark"}
|
|
model = installed[0].get("name")
|
|
|
|
defaults = KNOBS[knob]
|
|
if defaults.get("kind") == "discrete":
|
|
supported = defaults.get("values") or _supported_clocks(
|
|
"mem" if knob == "lock_mem_mhz" else "gr")
|
|
if not supported:
|
|
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
|
|
values = [v for v in supported
|
|
if (start is None or v >= start) and (stop is None or v <= stop)]
|
|
if not values:
|
|
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
|
|
start, stop, step = values[0], values[-1], None
|
|
else:
|
|
start = defaults["default_start"] if start is None else start
|
|
stop = defaults["default_stop"] if stop is None else stop
|
|
step = defaults["default_step"] if step is None else step
|
|
if step <= 0 or stop < start:
|
|
return {"success": False, "error": "invalid sweep range"}
|
|
values = list(range(start, stop + 1, step))
|
|
|
|
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
|
|
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
|
|
|
|
# Refuse to sweep a knob the driver is going to ignore.
|
|
effectiveness = _knob_effective(knob, profile, values, baseline_value)
|
|
if not effectiveness["effective"]:
|
|
return {
|
|
"success": False,
|
|
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
|
|
"effectiveness": effectiveness,
|
|
}
|
|
|
|
state.running = True
|
|
state.cancel = False
|
|
results: List[Dict[str, Any]] = []
|
|
t_start = time.time()
|
|
|
|
try:
|
|
# Load the model once up front so the first step does not pay the load cost.
|
|
await _decode_benchmark(model)
|
|
|
|
for value in values:
|
|
if state.cancel:
|
|
break
|
|
overclock_manager.apply_profile(profile, overrides={knob: value})
|
|
await asyncio.sleep(SETTLE_S)
|
|
step_started = time.time()
|
|
|
|
samples = []
|
|
for _ in range(max(repeats, 1)):
|
|
samples.append(await _decode_benchmark(model))
|
|
if state.cancel:
|
|
break
|
|
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
xids = _xid_since(step_started)
|
|
ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")]
|
|
temp = gpu.get("temperature_c", 0) or 0
|
|
|
|
instability = []
|
|
if xids:
|
|
instability.append(f"kernel Xid: {xids[0][:120]}")
|
|
if len(ok_samples) < len(samples):
|
|
instability.append("benchmark failed or produced degenerate output")
|
|
if temp >= TEMP_CEILING_C:
|
|
instability.append(f"temperature ceiling hit ({temp}°C)")
|
|
|
|
tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2)
|
|
row = {
|
|
"knob": knob,
|
|
"value": value,
|
|
"profile": profile,
|
|
"model": model,
|
|
"tokens_per_sec": tok_s,
|
|
"temp_c": temp,
|
|
"power_w": gpu.get("power_w"),
|
|
"clock_sm_mhz": gpu.get("clock_graphics_mhz"),
|
|
"clock_mem_mhz": gpu.get("clock_mem_mhz"),
|
|
"throttle_reasons": gpu.get("throttle_reasons"),
|
|
"stable": not instability,
|
|
"instability": "; ".join(instability) or None,
|
|
"samples": samples,
|
|
}
|
|
results.append(row)
|
|
telemetry_store.record_autotune({
|
|
"profile": profile, "knob": knob,
|
|
"core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"),
|
|
"mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"),
|
|
"tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"),
|
|
"stable": row["stable"], "instability": row["instability"],
|
|
"note": f"sweep {knob} {start}..{stop} step {step}",
|
|
})
|
|
state.current = {"knob": knob, "value": value, "stop": stop,
|
|
"tokens_per_sec": tok_s, "stable": row["stable"]}
|
|
logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, "
|
|
f"stable={row['stable']} {row['instability'] or ''}")
|
|
|
|
if not row["stable"]:
|
|
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
|
|
break
|
|
|
|
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
|
|
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
|
|
|
|
# Two different comparisons, and conflating them is how a sweep oversells itself.
|
|
# The first step of the range is not "baseline" unless it happens to be what the
|
|
# profile is already set to -- reporting "+102%" against the slowest value tried
|
|
# implies a speedup the user would not actually observe.
|
|
first_step = next((r for r in results if r["value"] == values[0]), None)
|
|
current = next((r for r in results if r["value"] == baseline_value), None)
|
|
gain_vs_first_step_pct = None
|
|
if best and first_step and first_step["tokens_per_sec"] > 0:
|
|
gain_vs_first_step_pct = round(
|
|
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
|
|
gain_vs_current_pct = None
|
|
if best and current and current["tokens_per_sec"] > 0:
|
|
gain_vs_current_pct = round(
|
|
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
|
|
|
|
applied = None
|
|
if apply_best and best:
|
|
overclock_manager.set_profile(profile, {knob: best["value"]})
|
|
applied = {knob: best["value"], "profile": profile}
|
|
logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'")
|
|
|
|
result = {
|
|
"success": True,
|
|
"knob": knob,
|
|
"profile": profile,
|
|
"model": model,
|
|
"range": {"start": start, "stop": stop, "step": step, "values": values},
|
|
"effectiveness": effectiveness,
|
|
"steps_run": len(results),
|
|
"duration_s": round(time.time() - t_start, 1),
|
|
"cancelled": state.cancel,
|
|
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
|
|
"clock_sm_mhz")} if best else None,
|
|
"current_profile_value": baseline_value,
|
|
"first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
|
|
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
|
|
"gain_vs_first_step_pct": gain_vs_first_step_pct,
|
|
"gain_vs_current_pct": gain_vs_current_pct,
|
|
"gain_note": ("compared against the profile's current setting"
|
|
if current is not None else
|
|
f"the profile's current value ({baseline_value}) was not in the "
|
|
f"swept range, so only the spread across tried values is shown"),
|
|
"applied_to_profile": applied,
|
|
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
|
|
for r in results if not r["stable"]), None),
|
|
"table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w",
|
|
"clock_mem_mhz", "clock_sm_mhz", "stable",
|
|
"instability")} for r in results],
|
|
}
|
|
state.last_result = result
|
|
return result
|
|
finally:
|
|
# Always hand the card back exactly as we found it.
|
|
state.running = False
|
|
state.current = None
|
|
try:
|
|
overclock_manager.apply_profile(profile)
|
|
logger.info(f"autotune restored profile '{profile}'")
|
|
except Exception as e:
|
|
logger.error(f"autotune failed to restore profile, forcing stock: {e}")
|
|
overclock_manager.restore_safe("autotune restore failed")
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
return {
|
|
"running": state.running,
|
|
"current": state.current,
|
|
"last_result": state.last_result,
|
|
"knobs": KNOBS,
|
|
"history": telemetry_store.autotune_history(100),
|
|
}
|
|
|
|
|
|
def cancel() -> Dict[str, Any]:
|
|
if not state.running:
|
|
return {"cancelled": False, "reason": "no sweep running"}
|
|
state.cancel = True
|
|
return {"cancelled": True}
|