"""Closed-loop overclock autotuner. The profiles in this repo were hand-tuned and had already drifted apart from the defaults in overclock_manager.py, with no record of which numbers were actually faster. This module answers that empirically: it walks a clock offset upward, measures real decode throughput at each step, watches for instability, and reports the highest setting that was both stable and fastest. Safety properties: * The original profile is always restored, including on exception or cancellation. * A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's render by yanking clocks mid-graph. * Every step is bounded by a temperature ceiling and checked for kernel Xid messages, and the sweep stops climbing the moment a step looks unstable. """ import asyncio import logging import subprocess import time from typing import Any, Dict, List, Optional import overclock_manager import telemetry_store import vram_arbitrator logger = logging.getLogger("autotune") BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging " "works in a modern operating system kernel.") BENCH_TOKENS = 160 SETTLE_S = 2.5 TEMP_CEILING_C = 84.0 KNOBS = { "mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100}, "core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25}, } def _xid_since(since_ts: float) -> List[str]: """Look for NVIDIA Xid errors in the kernel log — the clearest instability signal.""" try: since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts)) proc = subprocess.run( ["journalctl", "-k", "--since", since, "--no-pager", "-q"], capture_output=True, text=True, timeout=10, ) return [ln.strip() for ln in proc.stdout.splitlines() if "Xid" in ln or "NVRM:" in ln] except Exception as e: logger.debug(f"Xid check unavailable: {e}") return [] async def _decode_benchmark(model: str) -> Dict[str, Any]: """One fixed decode run. Throughput here is the thing being optimised.""" client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0) t0 = time.perf_counter() resp = await client.post("/api/generate", json={ "model": model, "prompt": BENCH_PROMPT, "stream": False, "keep_alive": "10m", "options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42}, }) wall_ms = round((time.perf_counter() - t0) * 1000, 2) if resp.status_code != 200: return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}", "wall_ms": wall_ms} data = resp.json() eval_ms = data.get("eval_duration", 0) / 1e6 eval_count = data.get("eval_count", 0) text = data.get("response", "") or "" return { "ok": True, "tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0, "eval_count": eval_count, "eval_ms": round(eval_ms, 2), "prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2), "wall_ms": wall_ms, "response_chars": len(text), # A model producing almost nothing, or pure repetition, is a corruption signal. "degenerate": eval_count < BENCH_TOKENS * 0.5 or len(set(text.split())) < 8, } class SweepState: def __init__(self) -> None: self.running = False self.cancel = False self.current: Optional[Dict[str, Any]] = None self.last_result: Optional[Dict[str, Any]] = None state = SweepState() async def sweep(knob: str = "mem_offset_mhz", profile: str = "ollama", model: Optional[str] = None, start: Optional[int] = None, stop: Optional[int] = None, step: Optional[int] = None, repeats: int = 1, apply_best: bool = False) -> Dict[str, Any]: """Sweep one clock offset and return the fastest stable value.""" if knob not in KNOBS: return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"} if state.running: return {"success": False, "error": "a sweep is already running"} comfy = await vram_arbitrator.get_comfyui_live_state() if comfy.get("executing") or comfy.get("queue_remaining"): return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"} if not model: ollama = await vram_arbitrator.get_ollama_live_state() model = ollama.get("active_model_name") if not model: installed = ollama.get("installed_models") or [] if not installed: return {"success": False, "error": "no Ollama model available to benchmark"} model = installed[0].get("name") defaults = KNOBS[knob] start = defaults["default_start"] if start is None else start stop = defaults["default_stop"] if stop is None else stop step = defaults["default_step"] if step is None else step if step <= 0 or stop < start: return {"success": False, "error": "invalid sweep range"} baseline_cfg = overclock_manager.load_profiles().get(profile, {}) state.running = True state.cancel = False results: List[Dict[str, Any]] = [] t_start = time.time() try: # Load the model once up front so the first step does not pay the load cost. await _decode_benchmark(model) value = start while value <= stop and not state.cancel: overclock_manager.apply_profile(profile, overrides={knob: value}) await asyncio.sleep(SETTLE_S) step_started = time.time() samples = [] for _ in range(max(repeats, 1)): samples.append(await _decode_benchmark(model)) if state.cancel: break gpu = vram_arbitrator.get_gpu_hardware_stats() xids = _xid_since(step_started) ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")] temp = gpu.get("temperature_c", 0) or 0 instability = [] if xids: instability.append(f"kernel Xid: {xids[0][:120]}") if len(ok_samples) < len(samples): instability.append("benchmark failed or produced degenerate output") if temp >= TEMP_CEILING_C: instability.append(f"temperature ceiling hit ({temp}°C)") tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2) row = { "knob": knob, "value": value, "profile": profile, "model": model, "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "clock_sm_mhz": gpu.get("clock_graphics_mhz"), "clock_mem_mhz": gpu.get("clock_mem_mhz"), "throttle_reasons": gpu.get("throttle_reasons"), "stable": not instability, "instability": "; ".join(instability) or None, "samples": samples, } results.append(row) telemetry_store.record_autotune({ "profile": profile, "knob": knob, "core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"), "mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"), "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "stable": row["stable"], "instability": row["instability"], "note": f"sweep {knob} {start}..{stop} step {step}", }) state.current = {"knob": knob, "value": value, "stop": stop, "tokens_per_sec": tok_s, "stable": row["stable"]} logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, " f"stable={row['stable']} {row['instability'] or ''}") if not row["stable"]: logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}") break value += step stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0] best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None baseline = next((r for r in results if r["value"] == start), None) gain_pct = None if best and baseline and baseline["tokens_per_sec"] > 0: gain_pct = round((best["tokens_per_sec"] / baseline["tokens_per_sec"] - 1) * 100, 2) applied = None if apply_best and best: overclock_manager.set_profile(profile, {knob: best["value"]}) applied = {knob: best["value"], "profile": profile} logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'") result = { "success": True, "knob": knob, "profile": profile, "model": model, "range": {"start": start, "stop": stop, "step": step}, "steps_run": len(results), "duration_s": round(time.time() - t_start, 1), "cancelled": state.cancel, "best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz", "clock_sm_mhz")} if best else None, "baseline_tokens_per_sec": baseline["tokens_per_sec"] if baseline else None, "gain_pct": gain_pct, "applied_to_profile": applied, "first_unstable": next(({"value": r["value"], "why": r["instability"]} for r in results if not r["stable"]), None), "table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w", "clock_mem_mhz", "clock_sm_mhz", "stable", "instability")} for r in results], } state.last_result = result return result finally: # Always hand the card back exactly as we found it. state.running = False state.current = None try: overclock_manager.apply_profile(profile) logger.info(f"autotune restored profile '{profile}'") except Exception as e: logger.error(f"autotune failed to restore profile, forcing stock: {e}") overclock_manager.restore_safe("autotune restore failed") def get_status() -> Dict[str, Any]: return { "running": state.running, "current": state.current, "last_result": state.last_result, "knobs": KNOBS, "history": telemetry_store.autotune_history(100), } def cancel() -> Dict[str, Any]: if not state.running: return {"cancelled": False, "reason": "no sweep running"} state.cancel = True return {"cancelled": True}