"""Closed-loop overclock autotuner. The profiles in this repo were hand-tuned and had already drifted apart from the defaults in overclock_manager.py, with no record of which numbers were actually faster. This module answers that empirically: it walks a clock offset upward, measures real decode throughput at each step, watches for instability, and reports the highest setting that was both stable and fastest. Safety properties: * The original profile is always restored, including on exception or cancellation. * A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's render by yanking clocks mid-graph. * Every step is bounded by a temperature ceiling and checked for kernel Xid messages, and the sweep stops climbing the moment a step looks unstable. """ import asyncio import logging import random import subprocess import time from typing import Any, Dict, List, Optional import overclock_manager import telemetry_store import vram_arbitrator logger = logging.getLogger("autotune") BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging " "works in a modern operating system kernel.") BENCH_TOKENS = 160 SETTLE_S = 2.5 TEMP_CEILING_C = 84.0 KNOBS = { # Clock offsets go through nvidia-settings. On some drivers (595.84 here) the # attribute is accepted and then silently ignored -- assigning 0 reports success and # reads back 250 -- so a sweep of these can measure pure noise. _knob_effective() # checks before any sweep runs. "mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100, "kind": "offset", "verify": "mem_offset"}, "core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25, "kind": "offset", "verify": "core_offset"}, # Clock locks go through nvidia-smi and do work on the open/proprietary module alike. # Memory clock is the lever that matters for LLM decode, which is bandwidth bound. "lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem", "values": None}, # filled from the card's supported clock list "lock_core_max": {"kind": "discrete", "verify": "clock_sm", "values": None}, # The one lever this driver definitely honours. Worth knowing whether the extra # watts actually buy throughput, or just heat and fan noise. "power_limit_w": {"kind": "discrete", "verify": "power_limit", "values": None, "no_unlocked": True}, } def _supported_clocks(which: str = "mem") -> List[int]: """Discrete clock values the card will actually accept for -lmc / -lgc. Always queried as the mem,gr pair: asking for a single field returns one column, and reading index 1 from it silently yields an empty list rather than an error. """ try: proc = subprocess.run( ["nvidia-smi", "--query-supported-clocks=mem,gr", "--format=csv,noheader,nounits"], capture_output=True, text=True, timeout=15) rows = [] for line in proc.stdout.splitlines(): parts = [p.strip() for p in line.split(",")] if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit(): rows.append((int(parts[0]), int(parts[1]))) if not rows: return [] if which == "mem": return sorted({m for m, _ in rows}) # Graphics clocks are enumerated per memory clock; take the list for the highest # memory clock, which is the one any real workload runs at. top_mem = max(m for m, _ in rows) return sorted({g for m, g in rows if m == top_mem}) except Exception as e: logger.debug(f"supported clock query failed: {e}") return [] def _supported_power_limits(steps: int = 5) -> List[int]: """Power limits between the card's minimum and maximum, in even increments.""" try: proc = subprocess.run( ["nvidia-smi", "--query-gpu=power.min_limit,power.max_limit,power.default_limit", "--format=csv,noheader,nounits"], capture_output=True, text=True, timeout=10) parts = [p.strip() for p in proc.stdout.strip().split(",")] lo, hi, default = (int(float(parts[0])), int(float(parts[1])), int(float(parts[2]))) except Exception as e: logger.debug(f"power limit query failed: {e}") return [] # Start at 60% of max -- below that the card is not doing useful work for these # workloads -- and always include the stock default as a reference point. lo = max(lo, int(hi * 0.6)) span = hi - lo vals = {lo + round(i * span / (steps - 1)) for i in range(steps)} vals.add(default) return sorted(v for v in vals if lo <= v <= hi) def _subsample(values: List[int], max_steps: int) -> List[int]: """Evenly spaced subset, always keeping the endpoints. The card enumerates ~194 graphics clocks in 15 MHz increments; benchmarking every one would take hours and tell us nothing that a handful of well-spread points does not. """ if len(values) <= max_steps: return values idx = [round(i * (len(values) - 1) / (max_steps - 1)) for i in range(max_steps)] return sorted({values[i] for i in idx}) def _read_hw(field: str) -> Optional[float]: """Read back the hardware state a knob is supposed to move.""" gpu = vram_arbitrator.get_gpu_hardware_stats() if field == "clock_mem": return gpu.get("clock_mem_mhz") if field == "clock_sm": return gpu.get("clock_graphics_mhz") if field == "power_limit": return gpu.get("power_limit_w") if field in ("mem_offset", "core_offset"): r = overclock_manager._nvidia_settings( "-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]") for line in (r.get("out") or "").splitlines(): if "Attribute" in line and "):" in line: try: return float(line.split("):")[-1].split(".")[0].strip()) except Exception: pass return None def _knob_effective(knob: str, profile: str, candidates: List[int], baseline_value: int) -> Dict[str, Any]: """Verify a knob actually moves the hardware before we sweep it. Without this the tuner happily reports "best = the highest value tried" from run-to-run benchmark noise on a knob the driver is ignoring. The probe value is chosen as the candidate *furthest* from where the hardware currently sits. Probing with the maximum is not good enough: if the card already happens to be at its top clock, setting it there again moves nothing and a perfectly good knob looks broken. """ field = KNOBS[knob].get("verify") before = _read_hw(field) if before is not None and candidates: probe_value = max(candidates, key=lambda v: abs(v - before)) else: probe_value = candidates[-1] if candidates else 0 overclock_manager.apply_profile(profile, overrides={knob: probe_value}) time.sleep(2.0) after = _read_hw(field) overclock_manager.apply_profile(profile, overrides={knob: baseline_value}) moved = (before is not None and after is not None and abs(after - before) > 1e-6) return { "effective": bool(moved), "field": field, "before": before, "after": after, "probe_value": probe_value, "detail": (f"{field} moved {before} -> {after}" if moved else f"{field} stayed at {after} after setting {knob}={probe_value}; " f"this driver accepts the setting and ignores it"), } def _xid_since(since_ts: float) -> List[str]: """Look for NVIDIA Xid errors in the kernel log — the clearest instability signal.""" try: since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts)) proc = subprocess.run( ["journalctl", "-k", "--since", since, "--no-pager", "-q"], capture_output=True, text=True, timeout=10, ) return [ln.strip() for ln in proc.stdout.splitlines() if "Xid" in ln or "NVRM:" in ln] except Exception as e: logger.debug(f"Xid check unavailable: {e}") return [] async def _decode_benchmark(model: str) -> Dict[str, Any]: """One fixed decode run. Throughput here is the thing being optimised.""" client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0) t0 = time.perf_counter() resp = await client.post("/api/generate", json={ "model": model, "prompt": BENCH_PROMPT, "stream": False, "keep_alive": "10m", "options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42}, }) wall_ms = round((time.perf_counter() - t0) * 1000, 2) if resp.status_code != 200: return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}", "wall_ms": wall_ms} data = resp.json() eval_ms = data.get("eval_duration", 0) / 1e6 eval_count = data.get("eval_count", 0) # Reasoning models put their output in `thinking` and leave `response` empty, so a # check that only looked at `response` flagged every one of them as degenerate. text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip() return { "ok": True, "tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0, "eval_count": eval_count, "eval_ms": round(eval_ms, 2), "prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2), "wall_ms": wall_ms, "response_chars": len(text), # Corruption signals: the model stopped far short of the token budget, or it # produced text that is pure repetition. Token count is the primary signal -- # empty text alone is not enough, since output can arrive in other fields. "degenerate": (eval_count < BENCH_TOKENS * 0.5 or (len(text) > 0 and len(set(text.split())) < 8)), } # A fixed SDXL txt2img graph. Deterministic seed/steps/resolution so every step of a # sweep does identical work and the only variable is the clock. PreviewImage rather than # SaveImage keeps benchmark runs out of the user's output gallery. COMFY_BENCH_CKPT = "sd_xl_base_1.0.safetensors" COMFY_BENCH_STEPS = 20 COMFY_BENCH_SIZE = 1024 def _comfy_workflow(ckpt: str = COMFY_BENCH_CKPT, seed: Optional[int] = None) -> Dict[str, Any]: # The seed must vary per run. ComfyUI caches by node inputs, so a fixed seed makes the # second and later benchmarks return in ~1ms without executing anything at all. The # cost of the graph is identical regardless of seed, so this costs no comparability. seed = random.randint(1, 2**31) if seed is None else seed return { "1": {"class_type": "CheckpointLoaderSimple", "inputs": {"ckpt_name": ckpt}}, "2": {"class_type": "CLIPTextEncode", "inputs": {"clip": ["1", 1], "text": "a detailed photograph of a mountain range at sunrise"}}, "3": {"class_type": "CLIPTextEncode", "inputs": {"clip": ["1", 1], "text": "blurry, low quality"}}, "4": {"class_type": "EmptyLatentImage", "inputs": {"width": COMFY_BENCH_SIZE, "height": COMFY_BENCH_SIZE, "batch_size": 1}}, "5": {"class_type": "KSampler", "inputs": {"model": ["1", 0], "positive": ["2", 0], "negative": ["3", 0], "latent_image": ["4", 0], "seed": seed, "steps": COMFY_BENCH_STEPS, "cfg": 7.0, "sampler_name": "euler", "scheduler": "normal", "denoise": 1.0}}, "6": {"class_type": "VAEDecode", "inputs": {"samples": ["5", 0], "vae": ["1", 2]}}, "7": {"class_type": "PreviewImage", "inputs": {"images": ["6", 0]}}, } async def _diffusion_benchmark(ckpt: str = COMFY_BENCH_CKPT, timeout_s: float = 300.0) -> Dict[str, Any]: """Queue one fixed SDXL graph and time it. This is the compute-bound counterpart to the decode benchmark, and the only way to tell whether the 'comfy' profile helps.""" client = vram_arbitrator._client(vram_arbitrator.COMFY_API_BASE, 30.0) t0 = time.perf_counter() try: resp = await client.post("/prompt", json={"prompt": _comfy_workflow(ckpt), "client_id": "hyperswap-autotune"}) if resp.status_code != 200: return {"ok": False, "error": f"queue failed HTTP {resp.status_code}: {resp.text[:200]}"} prompt_id = resp.json().get("prompt_id") except Exception as e: return {"ok": False, "error": f"queue failed: {e}"} while (time.perf_counter() - t0) < timeout_s: await asyncio.sleep(0.25) try: h = await client.get(f"/history/{prompt_id}") if h.status_code != 200: continue entry = (h.json() or {}).get(prompt_id) if not entry: continue status = entry.get("status", {}) if status.get("status_str") == "error" or not status.get("completed", True): if status.get("status_str") == "error": return {"ok": False, "error": "ComfyUI reported an execution error", "wall_ms": round((time.perf_counter() - t0) * 1000, 2)} if status.get("completed"): wall = time.perf_counter() - t0 # ComfyUI stamps execution_start/success in the status messages; the delta # between them excludes our polling overhead and the queue wait. stamps = {} for msg in status.get("messages", []): if isinstance(msg, list) and len(msg) >= 2 and isinstance(msg[1], dict): if "timestamp" in msg[1]: stamps[msg[0]] = msg[1]["timestamp"] exec_ms = None if "execution_start" in stamps and "execution_success" in stamps: exec_ms = round(stamps["execution_success"] - stamps["execution_start"], 2) effective_ms = exec_ms or wall * 1000 # A graph that "finished" implausibly fast was served from ComfyUI's cache # rather than executed; treat it as an invalid sample, not a record score. cached = effective_ms < 250 return { "ok": not cached, "error": "result served from ComfyUI cache, not executed" if cached else None, "wall_ms": round(wall * 1000, 2), "exec_ms": exec_ms, "steps": COMFY_BENCH_STEPS, "it_per_sec": round(COMFY_BENCH_STEPS / (effective_ms / 1000), 3), "degenerate": cached, } except Exception: continue return {"ok": False, "error": f"diffusion benchmark timed out after {timeout_s}s"} class SweepState: def __init__(self) -> None: self.running = False self.cancel = False self.current: Optional[Dict[str, Any]] = None self.last_result: Optional[Dict[str, Any]] = None state = SweepState() async def sweep(knob: str = "mem_offset_mhz", profile: str = "ollama", workload: str = "auto", model: Optional[str] = None, start: Optional[int] = None, stop: Optional[int] = None, step: Optional[int] = None, repeats: int = 1, max_steps: int = 6, include_unlocked: bool = True, apply_best: bool = False) -> Dict[str, Any]: """Sweep one clock offset and return the fastest stable value.""" if knob not in KNOBS: return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"} if state.running: return {"success": False, "error": "a sweep is already running"} comfy = await vram_arbitrator.get_comfyui_live_state() if comfy.get("executing") or comfy.get("queue_remaining"): return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"} # 'auto': tune the workload the profile is actually for. if workload == "auto": workload = "comfy" if profile == "comfy" else "ollama" if workload not in ("ollama", "comfy"): return {"success": False, "error": "workload must be 'ollama', 'comfy' or 'auto'"} if workload == "comfy" and not comfy.get("online"): return {"success": False, "error": "ComfyUI is not reachable; cannot run a diffusion sweep"} if workload == "comfy": model = model or COMFY_BENCH_CKPT benchmark = lambda: _diffusion_benchmark(model) metric = "it_per_sec" else: benchmark = lambda: _decode_benchmark(model) metric = "tokens_per_sec" if workload == "ollama" and not model: ollama = await vram_arbitrator.get_ollama_live_state() model = ollama.get("active_model_name") if not model: installed = ollama.get("installed_models") or [] if not installed: return {"success": False, "error": "no Ollama model available to benchmark"} model = installed[0].get("name") benchmark = lambda: _decode_benchmark(model) defaults = KNOBS[knob] if defaults.get("kind") == "discrete": if knob == "power_limit_w": supported = _supported_power_limits() else: supported = defaults.get("values") or _supported_clocks( "mem" if knob == "lock_mem_mhz" else "gr") if not supported: return {"success": False, "error": f"card reported no supported clocks for {knob}"} values = [v for v in supported if (start is None or v >= start) and (stop is None or v <= stop)] if not values: return {"success": False, "error": f"no supported values in range; card offers {supported}"} values = _subsample(values, max_steps or 6) # 0 means "no lock at all". That is the honest control for a profile whose whole # premise is that locking the clock beats letting the card boost on its own. if include_unlocked and not defaults.get("no_unlocked"): values = [0] + values start, stop, step = values[0], values[-1], None else: start = defaults["default_start"] if start is None else start stop = defaults["default_stop"] if stop is None else stop step = defaults["default_step"] if step is None else step if step <= 0 or stop < start: return {"success": False, "error": "invalid sweep range"} values = list(range(start, stop + 1, step)) baseline_cfg = overclock_manager.load_profiles().get(profile, {}) baseline_value = int(baseline_cfg.get(knob, 0) or 0) # Refuse to sweep a knob the driver is going to ignore. effectiveness = _knob_effective(knob, profile, [v for v in values if v] or values, baseline_value) if not effectiveness["effective"]: return { "success": False, "error": f"'{knob}' does not move this GPU: {effectiveness['detail']}", "effectiveness": effectiveness, } state.running = True state.cancel = False results: List[Dict[str, Any]] = [] t_start = time.time() try: # Warm-up: load weights once up front so the first step does not pay the load cost. vram_arbitrator.arbitrator.suspend_oc("autotune sweep") await benchmark() for value in values: if state.cancel: break overclock_manager.apply_profile(profile, overrides={knob: value}) await asyncio.sleep(SETTLE_S) step_started = time.time() samples = [] for _ in range(max(repeats, 1)): samples.append(await benchmark()) if state.cancel: break gpu = vram_arbitrator.get_gpu_hardware_stats() xids = _xid_since(step_started) ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")] temp = gpu.get("temperature_c", 0) or 0 instability = [] if xids: instability.append(f"kernel Xid: {xids[0][:120]}") if len(ok_samples) < len(samples): instability.append("benchmark failed or produced degenerate output") if temp >= TEMP_CEILING_C: instability.append(f"temperature ceiling hit ({temp}°C)") tok_s = round(max((s.get(metric, 0.0) for s in ok_samples), default=0.0), 2) row = { "knob": knob, "value": value, "profile": profile, "workload": workload, "metric": metric, "model": model, "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "clock_sm_mhz": gpu.get("clock_graphics_mhz"), "clock_mem_mhz": gpu.get("clock_mem_mhz"), "throttle_reasons": gpu.get("throttle_reasons"), "stable": not instability, "instability": "; ".join(instability) or None, "samples": samples, } results.append(row) telemetry_store.record_autotune({ "profile": profile, "knob": knob, "core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"), "mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"), "tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"), "stable": row["stable"], "instability": row["instability"], "note": f"sweep {knob} {start}..{stop} step {step}", }) state.current = {"knob": knob, "value": value, "stop": stop, "tokens_per_sec": tok_s, "stable": row["stable"]} logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, " f"stable={row['stable']} {row['instability'] or ''}") if not row["stable"]: logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}") break stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0] best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None # Two different comparisons, and conflating them is how a sweep oversells itself. # The first step of the range is not "baseline" unless it happens to be what the # profile is already set to -- reporting "+102%" against the slowest value tried # implies a speedup the user would not actually observe. first_step = next((r for r in results if r["value"] == values[0]), None) current = next((r for r in results if r["value"] == baseline_value), None) gain_vs_first_step_pct = None if best and first_step and first_step["tokens_per_sec"] > 0: gain_vs_first_step_pct = round( (best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2) gain_vs_current_pct = None if best and current and current["tokens_per_sec"] > 0: gain_vs_current_pct = round( (best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2) applied = None if apply_best and best: overclock_manager.set_profile(profile, {knob: best["value"]}) applied = {knob: best["value"], "profile": profile} logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'") result = { "success": True, "knob": knob, "profile": profile, "workload": workload, "metric": metric, "model": model, "range": {"start": start, "stop": stop, "step": step, "values": values}, "effectiveness": effectiveness, "steps_run": len(results), "duration_s": round(time.time() - t_start, 1), "cancelled": state.cancel, "best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz", "clock_sm_mhz")} if best else None, "current_profile_value": baseline_value, "first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None, "current_tokens_per_sec": current["tokens_per_sec"] if current else None, "gain_vs_first_step_pct": gain_vs_first_step_pct, "gain_vs_current_pct": gain_vs_current_pct, "gain_note": ("compared against the profile's current setting" if current is not None else f"the profile's current value ({baseline_value}) was not in the " f"swept range, so only the spread across tried values is shown"), "applied_to_profile": applied, "first_unstable": next(({"value": r["value"], "why": r["instability"]} for r in results if not r["stable"]), None), "table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w", "clock_mem_mhz", "clock_sm_mhz", "stable", "instability")} for r in results], } state.last_result = result return result finally: # Always hand the card back exactly as we found it. state.running = False state.current = None vram_arbitrator.arbitrator.resume_oc(profile) try: overclock_manager.apply_profile(profile) logger.info(f"autotune restored profile '{profile}'") except Exception as e: logger.error(f"autotune failed to restore profile, forcing stock: {e}") overclock_manager.restore_safe("autotune restore failed") def get_status() -> Dict[str, Any]: return { "running": state.running, "current": state.current, "last_result": state.last_result, "knobs": KNOBS, "history": telemetry_store.autotune_history(100), } def cancel() -> Dict[str, Any]: if not state.running: return {"cancelled": False, "reason": "no sweep running"} state.cancel = True return {"cancelled": True}