diff --git a/autotune.py b/autotune.py index c819096..8dd6cf5 100644 --- a/autotune.py +++ b/autotune.py @@ -32,11 +32,96 @@ SETTLE_S = 2.5 TEMP_CEILING_C = 84.0 KNOBS = { - "mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100}, - "core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25}, + # Clock offsets go through nvidia-settings. On some drivers (595.84 here) the + # attribute is accepted and then silently ignored -- assigning 0 reports success and + # reads back 250 -- so a sweep of these can measure pure noise. _knob_effective() + # checks before any sweep runs. + "mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100, + "kind": "offset", "verify": "mem_offset"}, + "core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25, + "kind": "offset", "verify": "core_offset"}, + # Clock locks go through nvidia-smi and do work on the open/proprietary module alike. + # Memory clock is the lever that matters for LLM decode, which is bandwidth bound. + "lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem", + "values": None}, # filled from the card's supported clock list + "lock_core_max": {"kind": "discrete", "verify": "clock_sm", + "values": None}, } +def _supported_clocks(which: str = "mem") -> List[int]: + """Discrete clock values the card will actually accept for -lmc / -lgc.""" + try: + proc = subprocess.run( + ["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10) + col = 0 if which == "mem" else 1 + vals = set() + for line in proc.stdout.splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) > col and parts[col].isdigit(): + vals.add(int(parts[col])) + return sorted(vals) + except Exception as e: + logger.debug(f"supported clock query failed: {e}") + return [] + + +def _read_hw(field: str) -> Optional[float]: + """Read back the hardware state a knob is supposed to move.""" + gpu = vram_arbitrator.get_gpu_hardware_stats() + if field == "clock_mem": + return gpu.get("clock_mem_mhz") + if field == "clock_sm": + return gpu.get("clock_graphics_mhz") + if field in ("mem_offset", "core_offset"): + r = overclock_manager._nvidia_settings( + "-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]") + for line in (r.get("out") or "").splitlines(): + if "Attribute" in line and "):" in line: + try: + return float(line.split("):")[-1].split(".")[0].strip()) + except Exception: + pass + return None + + +def _knob_effective(knob: str, profile: str, candidates: List[int], + baseline_value: int) -> Dict[str, Any]: + """Verify a knob actually moves the hardware before we sweep it. + + Without this the tuner happily reports "best = the highest value tried" from run-to-run + benchmark noise on a knob the driver is ignoring. + + The probe value is chosen as the candidate *furthest* from where the hardware currently + sits. Probing with the maximum is not good enough: if the card already happens to be at + its top clock, setting it there again moves nothing and a perfectly good knob looks + broken. + """ + field = KNOBS[knob].get("verify") + before = _read_hw(field) + if before is not None and candidates: + probe_value = max(candidates, key=lambda v: abs(v - before)) + else: + probe_value = candidates[-1] if candidates else 0 + overclock_manager.apply_profile(profile, overrides={knob: probe_value}) + time.sleep(2.0) + after = _read_hw(field) + overclock_manager.apply_profile(profile, overrides={knob: baseline_value}) + moved = (before is not None and after is not None and abs(after - before) > 1e-6) + return { + "effective": bool(moved), + "field": field, + "before": before, + "after": after, + "probe_value": probe_value, + "detail": (f"{field} moved {before} -> {after}" if moved else + f"{field} stayed at {after} after setting {knob}={probe_value}; " + f"this driver accepts the setting and ignores it"), + } + + def _xid_since(since_ts: float) -> List[str]: """Look for NVIDIA Xid errors in the kernel log — the clearest instability signal.""" try: @@ -70,7 +155,9 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]: data = resp.json() eval_ms = data.get("eval_duration", 0) / 1e6 eval_count = data.get("eval_count", 0) - text = data.get("response", "") or "" + # Reasoning models put their output in `thinking` and leave `response` empty, so a + # check that only looked at `response` flagged every one of them as degenerate. + text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip() return { "ok": True, "tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0, @@ -79,8 +166,11 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]: "prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2), "wall_ms": wall_ms, "response_chars": len(text), - # A model producing almost nothing, or pure repetition, is a corruption signal. - "degenerate": eval_count < BENCH_TOKENS * 0.5 or len(set(text.split())) < 8, + # Corruption signals: the model stopped far short of the token budget, or it + # produced text that is pure repetition. Token count is the primary signal -- + # empty text alone is not enough, since output can arrive in other fields. + "degenerate": (eval_count < BENCH_TOKENS * 0.5 + or (len(text) > 0 and len(set(text.split())) < 8)), } @@ -123,13 +213,36 @@ async def sweep(knob: str = "mem_offset_mhz", model = installed[0].get("name") defaults = KNOBS[knob] - start = defaults["default_start"] if start is None else start - stop = defaults["default_stop"] if stop is None else stop - step = defaults["default_step"] if step is None else step - if step <= 0 or stop < start: - return {"success": False, "error": "invalid sweep range"} + if defaults.get("kind") == "discrete": + supported = defaults.get("values") or _supported_clocks( + "mem" if knob == "lock_mem_mhz" else "gr") + if not supported: + return {"success": False, "error": f"card reported no supported clocks for {knob}"} + values = [v for v in supported + if (start is None or v >= start) and (stop is None or v <= stop)] + if not values: + return {"success": False, "error": f"no supported values in range; card offers {supported}"} + start, stop, step = values[0], values[-1], None + else: + start = defaults["default_start"] if start is None else start + stop = defaults["default_stop"] if stop is None else stop + step = defaults["default_step"] if step is None else step + if step <= 0 or stop < start: + return {"success": False, "error": "invalid sweep range"} + values = list(range(start, stop + 1, step)) baseline_cfg = overclock_manager.load_profiles().get(profile, {}) + baseline_value = int(baseline_cfg.get(knob, 0) or 0) + + # Refuse to sweep a knob the driver is going to ignore. + effectiveness = _knob_effective(knob, profile, values, baseline_value) + if not effectiveness["effective"]: + return { + "success": False, + "error": f"'{knob}' does not move this GPU: {effectiveness['detail']}", + "effectiveness": effectiveness, + } + state.running = True state.cancel = False results: List[Dict[str, Any]] = [] @@ -139,8 +252,9 @@ async def sweep(knob: str = "mem_offset_mhz", # Load the model once up front so the first step does not pay the load cost. await _decode_benchmark(model) - value = start - while value <= stop and not state.cancel: + for value in values: + if state.cancel: + break overclock_manager.apply_profile(profile, overrides={knob: value}) await asyncio.sleep(SETTLE_S) step_started = time.time() @@ -197,14 +311,24 @@ async def sweep(knob: str = "mem_offset_mhz", if not row["stable"]: logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}") break - value += step stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0] best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None - baseline = next((r for r in results if r["value"] == start), None) - gain_pct = None - if best and baseline and baseline["tokens_per_sec"] > 0: - gain_pct = round((best["tokens_per_sec"] / baseline["tokens_per_sec"] - 1) * 100, 2) + + # Two different comparisons, and conflating them is how a sweep oversells itself. + # The first step of the range is not "baseline" unless it happens to be what the + # profile is already set to -- reporting "+102%" against the slowest value tried + # implies a speedup the user would not actually observe. + first_step = next((r for r in results if r["value"] == values[0]), None) + current = next((r for r in results if r["value"] == baseline_value), None) + gain_vs_first_step_pct = None + if best and first_step and first_step["tokens_per_sec"] > 0: + gain_vs_first_step_pct = round( + (best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2) + gain_vs_current_pct = None + if best and current and current["tokens_per_sec"] > 0: + gain_vs_current_pct = round( + (best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2) applied = None if apply_best and best: @@ -217,14 +341,22 @@ async def sweep(knob: str = "mem_offset_mhz", "knob": knob, "profile": profile, "model": model, - "range": {"start": start, "stop": stop, "step": step}, + "range": {"start": start, "stop": stop, "step": step, "values": values}, + "effectiveness": effectiveness, "steps_run": len(results), "duration_s": round(time.time() - t_start, 1), "cancelled": state.cancel, "best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz", "clock_sm_mhz")} if best else None, - "baseline_tokens_per_sec": baseline["tokens_per_sec"] if baseline else None, - "gain_pct": gain_pct, + "current_profile_value": baseline_value, + "first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None, + "current_tokens_per_sec": current["tokens_per_sec"] if current else None, + "gain_vs_first_step_pct": gain_vs_first_step_pct, + "gain_vs_current_pct": gain_vs_current_pct, + "gain_note": ("compared against the profile's current setting" + if current is not None else + f"the profile's current value ({baseline_value}) was not in the " + f"swept range, so only the spread across tried values is shown"), "applied_to_profile": applied, "first_unstable": next(({"value": r["value"], "why": r["instability"]} for r in results if not r["stable"]), None), diff --git a/static/app.js b/static/app.js index b388f61..98a7604 100644 --- a/static/app.js +++ b/static/app.js @@ -855,7 +855,11 @@ function renderSweep(d) { const best = d.best ? `
★ Best stable: ${d.knob}=${d.best.value} → ${d.best.tokens_per_sec} tok/s at ${d.best.temp_c}°C - ${d.gain_pct != null ? `(${d.gain_pct > 0 ? '+' : ''}${d.gain_pct}% vs baseline)` : ''} + ${d.gain_vs_current_pct != null + ? `(${d.gain_vs_current_pct > 0 ? '+' : ''}${d.gain_vs_current_pct}% vs your current setting)` + : d.gain_vs_first_step_pct != null + ? `(spread across tried values: ${d.gain_vs_first_step_pct > 0 ? '+' : ''}${d.gain_vs_first_step_pct}% vs slowest)` + : ''} ${d.applied_to_profile ? '· saved to profile' : ''}
` : '
No stable step produced throughput.
'; const unstable = d.first_unstable