"""Closed-loop overclock autotuner. The profiles in this repo were hand-tuned and had already drifted apart from the defaults in overclock_manager.py, with no record of which numbers were actually faster. This module answers that empirically: it walks a clock offset upward, measures real decode throughput at each step, watches for instability, and reports the highest setting that was both stable and fastest. Safety properties: * The original profile is always restored, including on exception or cancellation. * A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's render by yanking clocks mid-graph. * Every step is bounded by a temperature ceiling and checked for kernel Xid messages, and the sweep stops climbing the moment a step looks unstable. """ import asyncio import logging import subprocess import time from typing import Any, Dict, List, Optional import overclock_manager import telemetry_store import vram_arbitrator logger = logging.getLogger("autotune") BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging " "works in a modern operating system kernel.") BENCH_TOKENS = 160 SETTLE_S = 2.5 TEMP_CEILING_C = 84.0 KNOBS = { # Clock offsets go through nvidia-settings. On some drivers (595.84 here) the # attribute is accepted and then silently ignored -- assigning 0 reports success and # reads back 250 -- so a sweep of these can measure pure noise. _knob_effective() # checks before any sweep runs. "mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100, "kind": "offset", "verify": "mem_offset"}, "core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25, "kind": "offset", "verify": "core_offset"}, # Clock locks go through nvidia-smi and do work on the open/proprietary module alike. # Memory clock is the lever that matters for LLM decode, which is bandwidth bound. "lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem", "values": None}, # filled from the card's supported clock list "lock_core_max": {"kind": "discrete", "verify": "clock_sm", "values": None}, } def _supported_clocks(which: str = "mem") -> List[int]: """Discrete clock values the card will actually accept for -lmc / -lgc.""" try: proc = subprocess.run( ["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}", "--format=csv,noheader,nounits"], capture_output=True, text=True, timeout=10) col = 0 if which == "mem" else 1 vals = set() for line in proc.stdout.splitlines(): parts = [p.strip() for p in line.split(",")] if len(parts) > col and parts[col].isdigit(): vals.add(int(parts[col])) return sorted(vals) except Exception as e: logger.debug(f"supported clock query failed: {e}") return [] def _read_hw(field: str) -> Optional[float]: """Read back the hardware state a knob is supposed to move.""" gpu = vram_arbitrator.get_gpu_hardware_stats() if field == "clock_mem": return gpu.get("clock_mem_mhz") if field == "clock_sm": return gpu.get("clock_graphics_mhz") if field in ("mem_offset", "core_offset"): r = overclock_manager._nvidia_settings( "-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]") for line in (r.get("out") or "").splitlines(): if "Attribute" in line and "):" in line: try: return float(line.split("):")[-1].split(".")[0].strip()) except Exception: pass return None def _knob_effective(knob: str, profile: str, candidates: List[int], baseline_value: int) -> Dict[str, Any]: """Verify a knob actually moves the hardware before we sweep it. Without this the tuner happily reports "best = the highest value tried" from run-to-run benchmark noise on a knob the driver is ignoring. The probe value is chosen as the candidate *furthest* from where the hardware currently sits. Probing with the maximum is not good enough: if the card already happens to be at its top clock, setting it there again moves nothing and a perfectly good knob looks broken. """ field = KNOBS[knob].get("verify") before = _read_hw(field) if before is not None and candidates: probe_value = max(candidates, key=lambda v: abs(v - before)) else: probe_value = candidates[-1] if candidates else 0 overclock_manager.apply_profile(profile, overrides={knob: probe_value}) time.sleep(2.0) after = _read_hw(field) overclock_manager.apply_profile(profile, overrides={knob: baseline_value}) moved = (before is not None and after is not None and abs(after - before) > 1e-6) return { "effective": bool(moved), "field": field, "before": before, "after": after, "probe_value": probe_value, "detail": (f"{field} moved {before} -> {after}" if moved else f"{field} stayed at {after} after setting {knob}={probe_value}; " f"this driver accepts the setting and ignores it"), } def _xid_since(since_ts: float) -> List[str]: """Look for NVIDIA Xid errors in the kernel log — the clearest instability signal.""" try: since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts)) proc = subprocess.run( ["journalctl", "-k", "--since", since, "--no-pager", "-q"], capture_output=True, text=True, timeout=10, ) return [ln.strip() for ln in proc.stdout.splitlines() if "Xid" in ln or "NVRM:" in ln] except Exception as e: logger.debug(f"Xid check unavailable: {e}") return [] async def _decode_benchmark(model: str) -> Dict[str, Any]: """One fixed decode run. Throughput here is the thing being optimised.""" client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0) t0 = time.perf_counter() resp = await client.post("/api/generate", json={ "model": model, "prompt": BENCH_PROMPT, "stream": False, "keep_alive": "10m", "options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42}, }) wall_ms = round((time.perf_counter() - t0) * 1000, 2) if resp.status_code != 200: return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}", "wall_ms": wall_ms} data = resp.json() eval_ms = data.get("eval_duration", 0) / 1e6 eval_count = data.get("eval_count", 0) # Reasoning models put their output in `thinking` and leave `response` empty, so a # check that only looked at `response` flagged every one of them as degenerate. text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip() return { "ok": True, "tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0, "eval_count": eval_count, "eval_ms": round(eval_ms, 2), "prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2), "wall_ms": wall_ms, "response_chars": len(text), # Corruption signals: the model stopped far short of the token budget, or it # produced text that is pure repetition. Token count is the primary signal -- # empty text alone is not enough, since output can arrive in other fields. "degenerate": (eval_count < BENCH_TOKENS * 0.5 or (len(text) > 0 and len(set(text.split())) < 8)), } class SweepState: def __init__(self) -> None: self.running = False self.cancel = False self.current: Optional[Dict[str, Any]] = None self.last_result: Optional[Dict[str, Any]] = None state = SweepState() async def sweep(knob: str = "mem_offset_mhz", profile: str = "ollama", model: Optional[str] = None, start: Optional[int] = None, stop: Optional[int] = None, step: Optional[int] = None, repeats: int = 1, apply_best: bool = False) -> Dict[str, Any]: """Sweep one clock offset and return the fastest stable value.""" if knob not in KNOBS: return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"} if state.running: return {"success": False, "error": "a sweep is already running"} comfy = await vram_arbitrator.get_comfyui_live_state() if comfy.get("executing") or comfy.get("queue_remaining"): return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"} if not model: ollama = await vram_arbitrator.get_ollama_live_state() model = ollama.get("active_model_name") if not model: installed = ollama.get("installed_models") or [] if not installed: return {"success": False, "error": "no Ollama model available to benchmark"} model = installed[0].get("name") defaults = KNOBS[knob] if defaults.get("kind") == "discrete": supported = defaults.get("values") or _supported_clocks( "mem" if knob == "lock_mem_mhz" else "gr") if not supported: return {"success": False, "error": f"card reported no supported clocks for {knob}"} values = [v for v in supported if (start is None or v >= start) and (stop is None or v <= stop)] if not values: return {"success": False, "error": f"no supported values in range; card offers {supported}"} start, stop, step = values[0], values[-1], None else: start = defaults["default_start"] if start is None else start stop = defaults["default_stop"] if stop is None else stop step = defaults["default_step"] if step is None else step if step <= 0 or stop < start: return {"success": False, "error": "invalid sweep range"} values = list(range(start, stop + 1, step)) baseline_cfg = overclock_manager.load_profiles().get(profile, {}) baseline_value = int(baseline_cfg.get(knob, 0) or 0) # Refuse to sweep a knob the driver is going to ignore. effectiveness = _knob_effective(knob, profile, values, baseline_value) if not effectiveness["effective"]: return { "success": False, "error": f"'{knob}' does not move this GPU: {effectiveness['detail']}", "effectiveness": effectiveness, } state.running = True state.cancel = False results: List[Dict[str, Any]] = [] t_start = time.time() try: # Load the model once up front so the first step does not pay the load cost. await _decode_benchmark(model) for value in values: if state.cancel: break overclock_manager.apply_profile(profile, overrides={knob: value}) await asyncio.sleep(SETTLE_S) step_started = time.time() samples = [] for _ in range(max(repeats, 1)): samples.append(await _decode_benchmark(model)) if state.cancel: break gpu = vram_arbitrator.get_gpu_hardware_stats() xids = _xid_since(step_started) ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")] temp = gpu.get("temperature_c", 0) or 0 instability = [] if xids: instability.append(f"kernel Xid: {xids[0][:120]}") if len(ok_samples) < len(samples): instability.append("benchmark failed or produced degenerate output") if temp >= TEMP_CEILING_C: instability.append(f"temperature ceiling hit ({temp}°C)") tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2) row = { "knob": knob, "value": value, "profile": profile, "model": model, "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "clock_sm_mhz": gpu.get("clock_graphics_mhz"), "clock_mem_mhz": gpu.get("clock_mem_mhz"), "throttle_reasons": gpu.get("throttle_reasons"), "stable": not instability, "instability": "; ".join(instability) or None, "samples": samples, } results.append(row) telemetry_store.record_autotune({ "profile": profile, "knob": knob, "core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"), "mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"), "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "stable": row["stable"], "instability": row["instability"], "note": f"sweep {knob} {start}..{stop} step {step}", }) state.current = {"knob": knob, "value": value, "stop": stop, "tokens_per_sec": tok_s, "stable": row["stable"]} logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, " f"stable={row['stable']} {row['instability'] or ''}") if not row["stable"]: logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}") break stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0] best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None # Two different comparisons, and conflating them is how a sweep oversells itself. # The first step of the range is not "baseline" unless it happens to be what the # profile is already set to -- reporting "+102%" against the slowest value tried # implies a speedup the user would not actually observe. first_step = next((r for r in results if r["value"] == values[0]), None) current = next((r for r in results if r["value"] == baseline_value), None) gain_vs_first_step_pct = None if best and first_step and first_step["tokens_per_sec"] > 0: gain_vs_first_step_pct = round( (best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2) gain_vs_current_pct = None if best and current and current["tokens_per_sec"] > 0: gain_vs_current_pct = round( (best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2) applied = None if apply_best and best: overclock_manager.set_profile(profile, {knob: best["value"]}) applied = {knob: best["value"], "profile": profile} logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'") result = { "success": True, "knob": knob, "profile": profile, "model": model, "range": {"start": start, "stop": stop, "step": step, "values": values}, "effectiveness": effectiveness, "steps_run": len(results), "duration_s": round(time.time() - t_start, 1), "cancelled": state.cancel, "best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz", "clock_sm_mhz")} if best else None, "current_profile_value": baseline_value, "first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None, "current_tokens_per_sec": current["tokens_per_sec"] if current else None, "gain_vs_first_step_pct": gain_vs_first_step_pct, "gain_vs_current_pct": gain_vs_current_pct, "gain_note": ("compared against the profile's current setting" if current is not None else f"the profile's current value ({baseline_value}) was not in the " f"swept range, so only the spread across tried values is shown"), "applied_to_profile": applied, "first_unstable": next(({"value": r["value"], "why": r["instability"]} for r in results if not r["stable"]), None), "table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w", "clock_mem_mhz", "clock_sm_mhz", "stable", "instability")} for r in results], } state.last_result = result return result finally: # Always hand the card back exactly as we found it. state.running = False state.current = None try: overclock_manager.apply_profile(profile) logger.info(f"autotune restored profile '{profile}'") except Exception as e: logger.error(f"autotune failed to restore profile, forcing stock: {e}") overclock_manager.restore_safe("autotune restore failed") def get_status() -> Dict[str, Any]: return { "running": state.running, "current": state.current, "last_result": state.last_result, "knobs": KNOBS, "history": telemetry_store.autotune_history(100), } def cancel() -> Dict[str, Any]: if not state.running: return {"cancelled": False, "reason": "no sweep running"} state.cancel = True return {"cancelled": True}