Verify autotune knobs against hardware before sweeping
Sweeping a knob the driver ignores measures nothing but benchmark noise, and the tuner would then confidently report 'best = the highest value tried'. On this box (driver 595.84) nvidia-settings accepts GPUGraphicsClockOffset/GPUMemoryTransferRate Offset and silently discards them: assigning 0 reports success and reads back 250. The ollama profile's core_offset_mhz=35 and mem_offset_mhz=200 have therefore been doing nothing. - _knob_effective() applies a probe value and confirms the hardware actually moved before any sweep starts, choosing the candidate furthest from the current reading (probing with the maximum fails when the card already sits at its top clock). - Adds discrete clock-lock knobs (lock_mem_mhz, lock_core_max) driven by the card's own supported-clock list, since -lmc/-lgc do work where offsets do not. - Reasoning models return their output in 'thinking' with an empty 'response', which the degeneracy check was flagging as corruption. Token count is now the primary signal. - Separates gain-vs-current-setting from gain-vs-slowest-value-tried. Reporting the latter as 'gain vs baseline' implied a +102% speedup that nobody would observe. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
162
autotune.py
162
autotune.py
@@ -32,11 +32,96 @@ SETTLE_S = 2.5
|
|||||||
TEMP_CEILING_C = 84.0
|
TEMP_CEILING_C = 84.0
|
||||||
|
|
||||||
KNOBS = {
|
KNOBS = {
|
||||||
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100},
|
# Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
|
||||||
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25},
|
# attribute is accepted and then silently ignored -- assigning 0 reports success and
|
||||||
|
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
|
||||||
|
# checks before any sweep runs.
|
||||||
|
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
|
||||||
|
"kind": "offset", "verify": "mem_offset"},
|
||||||
|
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
|
||||||
|
"kind": "offset", "verify": "core_offset"},
|
||||||
|
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
|
||||||
|
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
|
||||||
|
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
|
||||||
|
"values": None}, # filled from the card's supported clock list
|
||||||
|
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
|
||||||
|
"values": None},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _supported_clocks(which: str = "mem") -> List[int]:
|
||||||
|
"""Discrete clock values the card will actually accept for -lmc / -lgc."""
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}",
|
||||||
|
"--format=csv,noheader,nounits"],
|
||||||
|
capture_output=True, text=True, timeout=10)
|
||||||
|
col = 0 if which == "mem" else 1
|
||||||
|
vals = set()
|
||||||
|
for line in proc.stdout.splitlines():
|
||||||
|
parts = [p.strip() for p in line.split(",")]
|
||||||
|
if len(parts) > col and parts[col].isdigit():
|
||||||
|
vals.add(int(parts[col]))
|
||||||
|
return sorted(vals)
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"supported clock query failed: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def _read_hw(field: str) -> Optional[float]:
|
||||||
|
"""Read back the hardware state a knob is supposed to move."""
|
||||||
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
||||||
|
if field == "clock_mem":
|
||||||
|
return gpu.get("clock_mem_mhz")
|
||||||
|
if field == "clock_sm":
|
||||||
|
return gpu.get("clock_graphics_mhz")
|
||||||
|
if field in ("mem_offset", "core_offset"):
|
||||||
|
r = overclock_manager._nvidia_settings(
|
||||||
|
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
|
||||||
|
for line in (r.get("out") or "").splitlines():
|
||||||
|
if "Attribute" in line and "):" in line:
|
||||||
|
try:
|
||||||
|
return float(line.split("):")[-1].split(".")[0].strip())
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _knob_effective(knob: str, profile: str, candidates: List[int],
|
||||||
|
baseline_value: int) -> Dict[str, Any]:
|
||||||
|
"""Verify a knob actually moves the hardware before we sweep it.
|
||||||
|
|
||||||
|
Without this the tuner happily reports "best = the highest value tried" from run-to-run
|
||||||
|
benchmark noise on a knob the driver is ignoring.
|
||||||
|
|
||||||
|
The probe value is chosen as the candidate *furthest* from where the hardware currently
|
||||||
|
sits. Probing with the maximum is not good enough: if the card already happens to be at
|
||||||
|
its top clock, setting it there again moves nothing and a perfectly good knob looks
|
||||||
|
broken.
|
||||||
|
"""
|
||||||
|
field = KNOBS[knob].get("verify")
|
||||||
|
before = _read_hw(field)
|
||||||
|
if before is not None and candidates:
|
||||||
|
probe_value = max(candidates, key=lambda v: abs(v - before))
|
||||||
|
else:
|
||||||
|
probe_value = candidates[-1] if candidates else 0
|
||||||
|
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
|
||||||
|
time.sleep(2.0)
|
||||||
|
after = _read_hw(field)
|
||||||
|
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
|
||||||
|
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
|
||||||
|
return {
|
||||||
|
"effective": bool(moved),
|
||||||
|
"field": field,
|
||||||
|
"before": before,
|
||||||
|
"after": after,
|
||||||
|
"probe_value": probe_value,
|
||||||
|
"detail": (f"{field} moved {before} -> {after}" if moved else
|
||||||
|
f"{field} stayed at {after} after setting {knob}={probe_value}; "
|
||||||
|
f"this driver accepts the setting and ignores it"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def _xid_since(since_ts: float) -> List[str]:
|
def _xid_since(since_ts: float) -> List[str]:
|
||||||
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
|
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
|
||||||
try:
|
try:
|
||||||
@@ -70,7 +155,9 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]:
|
|||||||
data = resp.json()
|
data = resp.json()
|
||||||
eval_ms = data.get("eval_duration", 0) / 1e6
|
eval_ms = data.get("eval_duration", 0) / 1e6
|
||||||
eval_count = data.get("eval_count", 0)
|
eval_count = data.get("eval_count", 0)
|
||||||
text = data.get("response", "") or ""
|
# Reasoning models put their output in `thinking` and leave `response` empty, so a
|
||||||
|
# check that only looked at `response` flagged every one of them as degenerate.
|
||||||
|
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
|
||||||
return {
|
return {
|
||||||
"ok": True,
|
"ok": True,
|
||||||
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
|
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
|
||||||
@@ -79,8 +166,11 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]:
|
|||||||
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
|
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
|
||||||
"wall_ms": wall_ms,
|
"wall_ms": wall_ms,
|
||||||
"response_chars": len(text),
|
"response_chars": len(text),
|
||||||
# A model producing almost nothing, or pure repetition, is a corruption signal.
|
# Corruption signals: the model stopped far short of the token budget, or it
|
||||||
"degenerate": eval_count < BENCH_TOKENS * 0.5 or len(set(text.split())) < 8,
|
# produced text that is pure repetition. Token count is the primary signal --
|
||||||
|
# empty text alone is not enough, since output can arrive in other fields.
|
||||||
|
"degenerate": (eval_count < BENCH_TOKENS * 0.5
|
||||||
|
or (len(text) > 0 and len(set(text.split())) < 8)),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -123,13 +213,36 @@ async def sweep(knob: str = "mem_offset_mhz",
|
|||||||
model = installed[0].get("name")
|
model = installed[0].get("name")
|
||||||
|
|
||||||
defaults = KNOBS[knob]
|
defaults = KNOBS[knob]
|
||||||
|
if defaults.get("kind") == "discrete":
|
||||||
|
supported = defaults.get("values") or _supported_clocks(
|
||||||
|
"mem" if knob == "lock_mem_mhz" else "gr")
|
||||||
|
if not supported:
|
||||||
|
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
|
||||||
|
values = [v for v in supported
|
||||||
|
if (start is None or v >= start) and (stop is None or v <= stop)]
|
||||||
|
if not values:
|
||||||
|
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
|
||||||
|
start, stop, step = values[0], values[-1], None
|
||||||
|
else:
|
||||||
start = defaults["default_start"] if start is None else start
|
start = defaults["default_start"] if start is None else start
|
||||||
stop = defaults["default_stop"] if stop is None else stop
|
stop = defaults["default_stop"] if stop is None else stop
|
||||||
step = defaults["default_step"] if step is None else step
|
step = defaults["default_step"] if step is None else step
|
||||||
if step <= 0 or stop < start:
|
if step <= 0 or stop < start:
|
||||||
return {"success": False, "error": "invalid sweep range"}
|
return {"success": False, "error": "invalid sweep range"}
|
||||||
|
values = list(range(start, stop + 1, step))
|
||||||
|
|
||||||
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
|
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
|
||||||
|
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
|
||||||
|
|
||||||
|
# Refuse to sweep a knob the driver is going to ignore.
|
||||||
|
effectiveness = _knob_effective(knob, profile, values, baseline_value)
|
||||||
|
if not effectiveness["effective"]:
|
||||||
|
return {
|
||||||
|
"success": False,
|
||||||
|
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
|
||||||
|
"effectiveness": effectiveness,
|
||||||
|
}
|
||||||
|
|
||||||
state.running = True
|
state.running = True
|
||||||
state.cancel = False
|
state.cancel = False
|
||||||
results: List[Dict[str, Any]] = []
|
results: List[Dict[str, Any]] = []
|
||||||
@@ -139,8 +252,9 @@ async def sweep(knob: str = "mem_offset_mhz",
|
|||||||
# Load the model once up front so the first step does not pay the load cost.
|
# Load the model once up front so the first step does not pay the load cost.
|
||||||
await _decode_benchmark(model)
|
await _decode_benchmark(model)
|
||||||
|
|
||||||
value = start
|
for value in values:
|
||||||
while value <= stop and not state.cancel:
|
if state.cancel:
|
||||||
|
break
|
||||||
overclock_manager.apply_profile(profile, overrides={knob: value})
|
overclock_manager.apply_profile(profile, overrides={knob: value})
|
||||||
await asyncio.sleep(SETTLE_S)
|
await asyncio.sleep(SETTLE_S)
|
||||||
step_started = time.time()
|
step_started = time.time()
|
||||||
@@ -197,14 +311,24 @@ async def sweep(knob: str = "mem_offset_mhz",
|
|||||||
if not row["stable"]:
|
if not row["stable"]:
|
||||||
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
|
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
|
||||||
break
|
break
|
||||||
value += step
|
|
||||||
|
|
||||||
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
|
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
|
||||||
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
|
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
|
||||||
baseline = next((r for r in results if r["value"] == start), None)
|
|
||||||
gain_pct = None
|
# Two different comparisons, and conflating them is how a sweep oversells itself.
|
||||||
if best and baseline and baseline["tokens_per_sec"] > 0:
|
# The first step of the range is not "baseline" unless it happens to be what the
|
||||||
gain_pct = round((best["tokens_per_sec"] / baseline["tokens_per_sec"] - 1) * 100, 2)
|
# profile is already set to -- reporting "+102%" against the slowest value tried
|
||||||
|
# implies a speedup the user would not actually observe.
|
||||||
|
first_step = next((r for r in results if r["value"] == values[0]), None)
|
||||||
|
current = next((r for r in results if r["value"] == baseline_value), None)
|
||||||
|
gain_vs_first_step_pct = None
|
||||||
|
if best and first_step and first_step["tokens_per_sec"] > 0:
|
||||||
|
gain_vs_first_step_pct = round(
|
||||||
|
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
|
||||||
|
gain_vs_current_pct = None
|
||||||
|
if best and current and current["tokens_per_sec"] > 0:
|
||||||
|
gain_vs_current_pct = round(
|
||||||
|
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
|
||||||
|
|
||||||
applied = None
|
applied = None
|
||||||
if apply_best and best:
|
if apply_best and best:
|
||||||
@@ -217,14 +341,22 @@ async def sweep(knob: str = "mem_offset_mhz",
|
|||||||
"knob": knob,
|
"knob": knob,
|
||||||
"profile": profile,
|
"profile": profile,
|
||||||
"model": model,
|
"model": model,
|
||||||
"range": {"start": start, "stop": stop, "step": step},
|
"range": {"start": start, "stop": stop, "step": step, "values": values},
|
||||||
|
"effectiveness": effectiveness,
|
||||||
"steps_run": len(results),
|
"steps_run": len(results),
|
||||||
"duration_s": round(time.time() - t_start, 1),
|
"duration_s": round(time.time() - t_start, 1),
|
||||||
"cancelled": state.cancel,
|
"cancelled": state.cancel,
|
||||||
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
|
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
|
||||||
"clock_sm_mhz")} if best else None,
|
"clock_sm_mhz")} if best else None,
|
||||||
"baseline_tokens_per_sec": baseline["tokens_per_sec"] if baseline else None,
|
"current_profile_value": baseline_value,
|
||||||
"gain_pct": gain_pct,
|
"first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
|
||||||
|
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
|
||||||
|
"gain_vs_first_step_pct": gain_vs_first_step_pct,
|
||||||
|
"gain_vs_current_pct": gain_vs_current_pct,
|
||||||
|
"gain_note": ("compared against the profile's current setting"
|
||||||
|
if current is not None else
|
||||||
|
f"the profile's current value ({baseline_value}) was not in the "
|
||||||
|
f"swept range, so only the spread across tried values is shown"),
|
||||||
"applied_to_profile": applied,
|
"applied_to_profile": applied,
|
||||||
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
|
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
|
||||||
for r in results if not r["stable"]), None),
|
for r in results if not r["stable"]), None),
|
||||||
|
|||||||
@@ -855,7 +855,11 @@ function renderSweep(d) {
|
|||||||
const best = d.best
|
const best = d.best
|
||||||
? `<div class="text-emerald-400 mb-2">★ Best stable: <span class="font-mono">${d.knob}=${d.best.value}</span>
|
? `<div class="text-emerald-400 mb-2">★ Best stable: <span class="font-mono">${d.knob}=${d.best.value}</span>
|
||||||
→ ${d.best.tokens_per_sec} tok/s at ${d.best.temp_c}°C
|
→ ${d.best.tokens_per_sec} tok/s at ${d.best.temp_c}°C
|
||||||
${d.gain_pct != null ? `<span class="text-slate-400">(${d.gain_pct > 0 ? '+' : ''}${d.gain_pct}% vs baseline)</span>` : ''}
|
${d.gain_vs_current_pct != null
|
||||||
|
? `<span class="text-slate-400">(${d.gain_vs_current_pct > 0 ? '+' : ''}${d.gain_vs_current_pct}% vs your current setting)</span>`
|
||||||
|
: d.gain_vs_first_step_pct != null
|
||||||
|
? `<span class="text-slate-500">(spread across tried values: ${d.gain_vs_first_step_pct > 0 ? '+' : ''}${d.gain_vs_first_step_pct}% vs slowest)</span>`
|
||||||
|
: ''}
|
||||||
${d.applied_to_profile ? '<span class="text-fuchsia-400">· saved to profile</span>' : ''}</div>`
|
${d.applied_to_profile ? '<span class="text-fuchsia-400">· saved to profile</span>' : ''}</div>`
|
||||||
: '<div class="text-amber-400 mb-2">No stable step produced throughput.</div>';
|
: '<div class="text-amber-400 mb-2">No stable step produced throughput.</div>';
|
||||||
const unstable = d.first_unstable
|
const unstable = d.first_unstable
|
||||||
|
|||||||
Reference in New Issue
Block a user