Verify autotune knobs against hardware before sweeping

Sweeping a knob the driver ignores measures nothing but benchmark noise, and the
tuner would then confidently report 'best = the highest value tried'. On this box
(driver 595.84) nvidia-settings accepts GPUGraphicsClockOffset/GPUMemoryTransferRate
Offset and silently discards them: assigning 0 reports success and reads back 250.
The ollama profile's core_offset_mhz=35 and mem_offset_mhz=200 have therefore been
doing nothing.

- _knob_effective() applies a probe value and confirms the hardware actually moved
  before any sweep starts, choosing the candidate furthest from the current reading
  (probing with the maximum fails when the card already sits at its top clock).
- Adds discrete clock-lock knobs (lock_mem_mhz, lock_core_max) driven by the card's
  own supported-clock list, since -lmc/-lgc do work where offsets do not.
- Reasoning models return their output in 'thinking' with an empty 'response', which
  the degeneracy check was flagging as corruption. Token count is now the primary
  signal.
- Separates gain-vs-current-setting from gain-vs-slowest-value-tried. Reporting the
  latter as 'gain vs baseline' implied a +102% speedup that nobody would observe.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-08-28 09:10:06 -07:00
parent 5431144b2e
commit 63297dc49d
2 changed files with 157 additions and 21 deletions

View File

@@ -32,11 +32,96 @@ SETTLE_S = 2.5
TEMP_CEILING_C = 84.0 TEMP_CEILING_C = 84.0
KNOBS = { KNOBS = {
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100}, # Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25}, # attribute is accepted and then silently ignored -- assigning 0 reports success and
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
# checks before any sweep runs.
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
"kind": "offset", "verify": "mem_offset"},
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
"kind": "offset", "verify": "core_offset"},
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
"values": None}, # filled from the card's supported clock list
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
"values": None},
} }
def _supported_clocks(which: str = "mem") -> List[int]:
"""Discrete clock values the card will actually accept for -lmc / -lgc."""
try:
proc = subprocess.run(
["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
col = 0 if which == "mem" else 1
vals = set()
for line in proc.stdout.splitlines():
parts = [p.strip() for p in line.split(",")]
if len(parts) > col and parts[col].isdigit():
vals.add(int(parts[col]))
return sorted(vals)
except Exception as e:
logger.debug(f"supported clock query failed: {e}")
return []
def _read_hw(field: str) -> Optional[float]:
"""Read back the hardware state a knob is supposed to move."""
gpu = vram_arbitrator.get_gpu_hardware_stats()
if field == "clock_mem":
return gpu.get("clock_mem_mhz")
if field == "clock_sm":
return gpu.get("clock_graphics_mhz")
if field in ("mem_offset", "core_offset"):
r = overclock_manager._nvidia_settings(
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
for line in (r.get("out") or "").splitlines():
if "Attribute" in line and "):" in line:
try:
return float(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
return None
def _knob_effective(knob: str, profile: str, candidates: List[int],
baseline_value: int) -> Dict[str, Any]:
"""Verify a knob actually moves the hardware before we sweep it.
Without this the tuner happily reports "best = the highest value tried" from run-to-run
benchmark noise on a knob the driver is ignoring.
The probe value is chosen as the candidate *furthest* from where the hardware currently
sits. Probing with the maximum is not good enough: if the card already happens to be at
its top clock, setting it there again moves nothing and a perfectly good knob looks
broken.
"""
field = KNOBS[knob].get("verify")
before = _read_hw(field)
if before is not None and candidates:
probe_value = max(candidates, key=lambda v: abs(v - before))
else:
probe_value = candidates[-1] if candidates else 0
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
time.sleep(2.0)
after = _read_hw(field)
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
return {
"effective": bool(moved),
"field": field,
"before": before,
"after": after,
"probe_value": probe_value,
"detail": (f"{field} moved {before} -> {after}" if moved else
f"{field} stayed at {after} after setting {knob}={probe_value}; "
f"this driver accepts the setting and ignores it"),
}
def _xid_since(since_ts: float) -> List[str]: def _xid_since(since_ts: float) -> List[str]:
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal.""" """Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
try: try:
@@ -70,7 +155,9 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]:
data = resp.json() data = resp.json()
eval_ms = data.get("eval_duration", 0) / 1e6 eval_ms = data.get("eval_duration", 0) / 1e6
eval_count = data.get("eval_count", 0) eval_count = data.get("eval_count", 0)
text = data.get("response", "") or "" # Reasoning models put their output in `thinking` and leave `response` empty, so a
# check that only looked at `response` flagged every one of them as degenerate.
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
return { return {
"ok": True, "ok": True,
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0, "tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
@@ -79,8 +166,11 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]:
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2), "prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
"wall_ms": wall_ms, "wall_ms": wall_ms,
"response_chars": len(text), "response_chars": len(text),
# A model producing almost nothing, or pure repetition, is a corruption signal. # Corruption signals: the model stopped far short of the token budget, or it
"degenerate": eval_count < BENCH_TOKENS * 0.5 or len(set(text.split())) < 8, # produced text that is pure repetition. Token count is the primary signal --
# empty text alone is not enough, since output can arrive in other fields.
"degenerate": (eval_count < BENCH_TOKENS * 0.5
or (len(text) > 0 and len(set(text.split())) < 8)),
} }
@@ -123,13 +213,36 @@ async def sweep(knob: str = "mem_offset_mhz",
model = installed[0].get("name") model = installed[0].get("name")
defaults = KNOBS[knob] defaults = KNOBS[knob]
if defaults.get("kind") == "discrete":
supported = defaults.get("values") or _supported_clocks(
"mem" if knob == "lock_mem_mhz" else "gr")
if not supported:
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
values = [v for v in supported
if (start is None or v >= start) and (stop is None or v <= stop)]
if not values:
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
start, stop, step = values[0], values[-1], None
else:
start = defaults["default_start"] if start is None else start start = defaults["default_start"] if start is None else start
stop = defaults["default_stop"] if stop is None else stop stop = defaults["default_stop"] if stop is None else stop
step = defaults["default_step"] if step is None else step step = defaults["default_step"] if step is None else step
if step <= 0 or stop < start: if step <= 0 or stop < start:
return {"success": False, "error": "invalid sweep range"} return {"success": False, "error": "invalid sweep range"}
values = list(range(start, stop + 1, step))
baseline_cfg = overclock_manager.load_profiles().get(profile, {}) baseline_cfg = overclock_manager.load_profiles().get(profile, {})
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
# Refuse to sweep a knob the driver is going to ignore.
effectiveness = _knob_effective(knob, profile, values, baseline_value)
if not effectiveness["effective"]:
return {
"success": False,
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
"effectiveness": effectiveness,
}
state.running = True state.running = True
state.cancel = False state.cancel = False
results: List[Dict[str, Any]] = [] results: List[Dict[str, Any]] = []
@@ -139,8 +252,9 @@ async def sweep(knob: str = "mem_offset_mhz",
# Load the model once up front so the first step does not pay the load cost. # Load the model once up front so the first step does not pay the load cost.
await _decode_benchmark(model) await _decode_benchmark(model)
value = start for value in values:
while value <= stop and not state.cancel: if state.cancel:
break
overclock_manager.apply_profile(profile, overrides={knob: value}) overclock_manager.apply_profile(profile, overrides={knob: value})
await asyncio.sleep(SETTLE_S) await asyncio.sleep(SETTLE_S)
step_started = time.time() step_started = time.time()
@@ -197,14 +311,24 @@ async def sweep(knob: str = "mem_offset_mhz",
if not row["stable"]: if not row["stable"]:
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}") logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
break break
value += step
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0] stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
baseline = next((r for r in results if r["value"] == start), None)
gain_pct = None # Two different comparisons, and conflating them is how a sweep oversells itself.
if best and baseline and baseline["tokens_per_sec"] > 0: # The first step of the range is not "baseline" unless it happens to be what the
gain_pct = round((best["tokens_per_sec"] / baseline["tokens_per_sec"] - 1) * 100, 2) # profile is already set to -- reporting "+102%" against the slowest value tried
# implies a speedup the user would not actually observe.
first_step = next((r for r in results if r["value"] == values[0]), None)
current = next((r for r in results if r["value"] == baseline_value), None)
gain_vs_first_step_pct = None
if best and first_step and first_step["tokens_per_sec"] > 0:
gain_vs_first_step_pct = round(
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
gain_vs_current_pct = None
if best and current and current["tokens_per_sec"] > 0:
gain_vs_current_pct = round(
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
applied = None applied = None
if apply_best and best: if apply_best and best:
@@ -217,14 +341,22 @@ async def sweep(knob: str = "mem_offset_mhz",
"knob": knob, "knob": knob,
"profile": profile, "profile": profile,
"model": model, "model": model,
"range": {"start": start, "stop": stop, "step": step}, "range": {"start": start, "stop": stop, "step": step, "values": values},
"effectiveness": effectiveness,
"steps_run": len(results), "steps_run": len(results),
"duration_s": round(time.time() - t_start, 1), "duration_s": round(time.time() - t_start, 1),
"cancelled": state.cancel, "cancelled": state.cancel,
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz", "best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
"clock_sm_mhz")} if best else None, "clock_sm_mhz")} if best else None,
"baseline_tokens_per_sec": baseline["tokens_per_sec"] if baseline else None, "current_profile_value": baseline_value,
"gain_pct": gain_pct, "first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
"gain_vs_first_step_pct": gain_vs_first_step_pct,
"gain_vs_current_pct": gain_vs_current_pct,
"gain_note": ("compared against the profile's current setting"
if current is not None else
f"the profile's current value ({baseline_value}) was not in the "
f"swept range, so only the spread across tried values is shown"),
"applied_to_profile": applied, "applied_to_profile": applied,
"first_unstable": next(({"value": r["value"], "why": r["instability"]} "first_unstable": next(({"value": r["value"], "why": r["instability"]}
for r in results if not r["stable"]), None), for r in results if not r["stable"]), None),

View File

@@ -855,7 +855,11 @@ function renderSweep(d) {
const best = d.best const best = d.best
? `<div class="text-emerald-400 mb-2">★ Best stable: <span class="font-mono">${d.knob}=${d.best.value}</span> ? `<div class="text-emerald-400 mb-2">★ Best stable: <span class="font-mono">${d.knob}=${d.best.value}</span>
→ ${d.best.tokens_per_sec} tok/s at ${d.best.temp_c}°C → ${d.best.tokens_per_sec} tok/s at ${d.best.temp_c}°C
${d.gain_pct != null ? `<span class="text-slate-400">(${d.gain_pct > 0 ? '+' : ''}${d.gain_pct}% vs baseline)</span>` : ''} ${d.gain_vs_current_pct != null
? `<span class="text-slate-400">(${d.gain_vs_current_pct > 0 ? '+' : ''}${d.gain_vs_current_pct}% vs your current setting)</span>`
: d.gain_vs_first_step_pct != null
? `<span class="text-slate-500">(spread across tried values: ${d.gain_vs_first_step_pct > 0 ? '+' : ''}${d.gain_vs_first_step_pct}% vs slowest)</span>`
: ''}
${d.applied_to_profile ? '<span class="text-fuchsia-400">· saved to profile</span>' : ''}</div>` ${d.applied_to_profile ? '<span class="text-fuchsia-400">· saved to profile</span>' : ''}</div>`
: '<div class="text-amber-400 mb-2">No stable step produced throughput.</div>'; : '<div class="text-amber-400 mb-2">No stable step produced throughput.</div>';
const unstable = d.first_unstable const unstable = d.first_unstable