The profiles were hand-written and had never been checked against the hardware. Adding a ComfyUI benchmark alongside the existing decode one made the compute side measurable for the first time, and most of what the profiles configured turned out to do nothing. Measured on this card (RTX 4080 SUPER, driver 595.84): - LLM decode is not power-bound: 73.0-73.5 tok/s flat from 222W to 370W, with the card never drawing more than 224W at any limit. The ollama profile's 370W did nothing. - Diffusion is power-bound: 5.48 it/s @222W rising to 6.71 @370W, so comfy's 370W is worth a real +2.8% over the 320W stock default. - Clock locks did nothing for either workload: 72.6 tok/s locked at 11251MHz vs 72.7 unlocked; 6.77 it/s locked at 3105MHz vs 6.73 unlocked, and 6.78 at 2400MHz. - Memory bandwidth is still the decode bottleneck (5001MHz halves throughput to 35.9 tok/s), confirming the profile's premise -- the card just gets there unaided. - Fans: 48,435 samples show 81C all-time max and zero thermal throttle events, while the ollama profile held 49.6C average by running fans at 87%. All profiles now use automatic fans and let the thermal governor escalate on demand. Code changes supporting that: - _diffusion_benchmark() queues a fixed SDXL graph via ComfyUI's API. The seed must vary per run: ComfyUI caches by node inputs, so a fixed seed returned in ~1ms without executing. Implausibly fast results are now rejected as cache hits rather than recorded as record scores. - The arbitrator's automatic profile switching is suspended during a sweep. A diffusion benchmark trips trigger_comfy_priority, which reapplies the whole profile and would silently overwrite the clock being measured. - _supported_clocks() queries the mem,gr pair; asking for a single field returned one column and reading index 1 yielded an empty list rather than an error. Graphics clocks are subsampled (the card enumerates 194 of them) and lock sweeps include an explicit unlocked control step. - offsets_supported() probes once and apply_profile skips inert offset levers with an explanation instead of pretending they applied. - Profiles carry a 'measured' field recording the evidence behind each setting. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
571 lines
26 KiB
Python
571 lines
26 KiB
Python
"""Closed-loop overclock autotuner.
|
|
|
|
The profiles in this repo were hand-tuned and had already drifted apart from the defaults
|
|
in overclock_manager.py, with no record of which numbers were actually faster. This module
|
|
answers that empirically: it walks a clock offset upward, measures real decode throughput
|
|
at each step, watches for instability, and reports the highest setting that was both
|
|
stable and fastest.
|
|
|
|
Safety properties:
|
|
* The original profile is always restored, including on exception or cancellation.
|
|
* A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's
|
|
render by yanking clocks mid-graph.
|
|
* Every step is bounded by a temperature ceiling and checked for kernel Xid messages,
|
|
and the sweep stops climbing the moment a step looks unstable.
|
|
"""
|
|
import asyncio
|
|
import logging
|
|
import random
|
|
import subprocess
|
|
import time
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
import overclock_manager
|
|
import telemetry_store
|
|
import vram_arbitrator
|
|
|
|
logger = logging.getLogger("autotune")
|
|
|
|
BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging "
|
|
"works in a modern operating system kernel.")
|
|
BENCH_TOKENS = 160
|
|
SETTLE_S = 2.5
|
|
TEMP_CEILING_C = 84.0
|
|
|
|
KNOBS = {
|
|
# Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
|
|
# attribute is accepted and then silently ignored -- assigning 0 reports success and
|
|
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
|
|
# checks before any sweep runs.
|
|
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
|
|
"kind": "offset", "verify": "mem_offset"},
|
|
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
|
|
"kind": "offset", "verify": "core_offset"},
|
|
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
|
|
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
|
|
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
|
|
"values": None}, # filled from the card's supported clock list
|
|
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
|
|
"values": None},
|
|
# The one lever this driver definitely honours. Worth knowing whether the extra
|
|
# watts actually buy throughput, or just heat and fan noise.
|
|
"power_limit_w": {"kind": "discrete", "verify": "power_limit",
|
|
"values": None, "no_unlocked": True},
|
|
}
|
|
|
|
|
|
def _supported_clocks(which: str = "mem") -> List[int]:
|
|
"""Discrete clock values the card will actually accept for -lmc / -lgc.
|
|
|
|
Always queried as the mem,gr pair: asking for a single field returns one column, and
|
|
reading index 1 from it silently yields an empty list rather than an error.
|
|
"""
|
|
try:
|
|
proc = subprocess.run(
|
|
["nvidia-smi", "--query-supported-clocks=mem,gr", "--format=csv,noheader,nounits"],
|
|
capture_output=True, text=True, timeout=15)
|
|
rows = []
|
|
for line in proc.stdout.splitlines():
|
|
parts = [p.strip() for p in line.split(",")]
|
|
if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit():
|
|
rows.append((int(parts[0]), int(parts[1])))
|
|
if not rows:
|
|
return []
|
|
if which == "mem":
|
|
return sorted({m for m, _ in rows})
|
|
# Graphics clocks are enumerated per memory clock; take the list for the highest
|
|
# memory clock, which is the one any real workload runs at.
|
|
top_mem = max(m for m, _ in rows)
|
|
return sorted({g for m, g in rows if m == top_mem})
|
|
except Exception as e:
|
|
logger.debug(f"supported clock query failed: {e}")
|
|
return []
|
|
|
|
|
|
def _supported_power_limits(steps: int = 5) -> List[int]:
|
|
"""Power limits between the card's minimum and maximum, in even increments."""
|
|
try:
|
|
proc = subprocess.run(
|
|
["nvidia-smi", "--query-gpu=power.min_limit,power.max_limit,power.default_limit",
|
|
"--format=csv,noheader,nounits"],
|
|
capture_output=True, text=True, timeout=10)
|
|
parts = [p.strip() for p in proc.stdout.strip().split(",")]
|
|
lo, hi, default = (int(float(parts[0])), int(float(parts[1])), int(float(parts[2])))
|
|
except Exception as e:
|
|
logger.debug(f"power limit query failed: {e}")
|
|
return []
|
|
# Start at 60% of max -- below that the card is not doing useful work for these
|
|
# workloads -- and always include the stock default as a reference point.
|
|
lo = max(lo, int(hi * 0.6))
|
|
span = hi - lo
|
|
vals = {lo + round(i * span / (steps - 1)) for i in range(steps)}
|
|
vals.add(default)
|
|
return sorted(v for v in vals if lo <= v <= hi)
|
|
|
|
|
|
def _subsample(values: List[int], max_steps: int) -> List[int]:
|
|
"""Evenly spaced subset, always keeping the endpoints.
|
|
|
|
The card enumerates ~194 graphics clocks in 15 MHz increments; benchmarking every one
|
|
would take hours and tell us nothing that a handful of well-spread points does not.
|
|
"""
|
|
if len(values) <= max_steps:
|
|
return values
|
|
idx = [round(i * (len(values) - 1) / (max_steps - 1)) for i in range(max_steps)]
|
|
return sorted({values[i] for i in idx})
|
|
|
|
|
|
def _read_hw(field: str) -> Optional[float]:
|
|
"""Read back the hardware state a knob is supposed to move."""
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
if field == "clock_mem":
|
|
return gpu.get("clock_mem_mhz")
|
|
if field == "clock_sm":
|
|
return gpu.get("clock_graphics_mhz")
|
|
if field == "power_limit":
|
|
return gpu.get("power_limit_w")
|
|
if field in ("mem_offset", "core_offset"):
|
|
r = overclock_manager._nvidia_settings(
|
|
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
|
|
for line in (r.get("out") or "").splitlines():
|
|
if "Attribute" in line and "):" in line:
|
|
try:
|
|
return float(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def _knob_effective(knob: str, profile: str, candidates: List[int],
|
|
baseline_value: int) -> Dict[str, Any]:
|
|
"""Verify a knob actually moves the hardware before we sweep it.
|
|
|
|
Without this the tuner happily reports "best = the highest value tried" from run-to-run
|
|
benchmark noise on a knob the driver is ignoring.
|
|
|
|
The probe value is chosen as the candidate *furthest* from where the hardware currently
|
|
sits. Probing with the maximum is not good enough: if the card already happens to be at
|
|
its top clock, setting it there again moves nothing and a perfectly good knob looks
|
|
broken.
|
|
"""
|
|
field = KNOBS[knob].get("verify")
|
|
before = _read_hw(field)
|
|
if before is not None and candidates:
|
|
probe_value = max(candidates, key=lambda v: abs(v - before))
|
|
else:
|
|
probe_value = candidates[-1] if candidates else 0
|
|
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
|
|
time.sleep(2.0)
|
|
after = _read_hw(field)
|
|
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
|
|
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
|
|
return {
|
|
"effective": bool(moved),
|
|
"field": field,
|
|
"before": before,
|
|
"after": after,
|
|
"probe_value": probe_value,
|
|
"detail": (f"{field} moved {before} -> {after}" if moved else
|
|
f"{field} stayed at {after} after setting {knob}={probe_value}; "
|
|
f"this driver accepts the setting and ignores it"),
|
|
}
|
|
|
|
|
|
def _xid_since(since_ts: float) -> List[str]:
|
|
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
|
|
try:
|
|
since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts))
|
|
proc = subprocess.run(
|
|
["journalctl", "-k", "--since", since, "--no-pager", "-q"],
|
|
capture_output=True, text=True, timeout=10,
|
|
)
|
|
return [ln.strip() for ln in proc.stdout.splitlines()
|
|
if "Xid" in ln or "NVRM:" in ln]
|
|
except Exception as e:
|
|
logger.debug(f"Xid check unavailable: {e}")
|
|
return []
|
|
|
|
|
|
async def _decode_benchmark(model: str) -> Dict[str, Any]:
|
|
"""One fixed decode run. Throughput here is the thing being optimised."""
|
|
client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0)
|
|
t0 = time.perf_counter()
|
|
resp = await client.post("/api/generate", json={
|
|
"model": model,
|
|
"prompt": BENCH_PROMPT,
|
|
"stream": False,
|
|
"keep_alive": "10m",
|
|
"options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42},
|
|
})
|
|
wall_ms = round((time.perf_counter() - t0) * 1000, 2)
|
|
if resp.status_code != 200:
|
|
return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}",
|
|
"wall_ms": wall_ms}
|
|
data = resp.json()
|
|
eval_ms = data.get("eval_duration", 0) / 1e6
|
|
eval_count = data.get("eval_count", 0)
|
|
# Reasoning models put their output in `thinking` and leave `response` empty, so a
|
|
# check that only looked at `response` flagged every one of them as degenerate.
|
|
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
|
|
return {
|
|
"ok": True,
|
|
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
|
|
"eval_count": eval_count,
|
|
"eval_ms": round(eval_ms, 2),
|
|
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
|
|
"wall_ms": wall_ms,
|
|
"response_chars": len(text),
|
|
# Corruption signals: the model stopped far short of the token budget, or it
|
|
# produced text that is pure repetition. Token count is the primary signal --
|
|
# empty text alone is not enough, since output can arrive in other fields.
|
|
"degenerate": (eval_count < BENCH_TOKENS * 0.5
|
|
or (len(text) > 0 and len(set(text.split())) < 8)),
|
|
}
|
|
|
|
|
|
# A fixed SDXL txt2img graph. Deterministic seed/steps/resolution so every step of a
|
|
# sweep does identical work and the only variable is the clock. PreviewImage rather than
|
|
# SaveImage keeps benchmark runs out of the user's output gallery.
|
|
COMFY_BENCH_CKPT = "sd_xl_base_1.0.safetensors"
|
|
COMFY_BENCH_STEPS = 20
|
|
COMFY_BENCH_SIZE = 1024
|
|
|
|
|
|
def _comfy_workflow(ckpt: str = COMFY_BENCH_CKPT, seed: Optional[int] = None) -> Dict[str, Any]:
|
|
# The seed must vary per run. ComfyUI caches by node inputs, so a fixed seed makes the
|
|
# second and later benchmarks return in ~1ms without executing anything at all. The
|
|
# cost of the graph is identical regardless of seed, so this costs no comparability.
|
|
seed = random.randint(1, 2**31) if seed is None else seed
|
|
return {
|
|
"1": {"class_type": "CheckpointLoaderSimple", "inputs": {"ckpt_name": ckpt}},
|
|
"2": {"class_type": "CLIPTextEncode",
|
|
"inputs": {"clip": ["1", 1],
|
|
"text": "a detailed photograph of a mountain range at sunrise"}},
|
|
"3": {"class_type": "CLIPTextEncode",
|
|
"inputs": {"clip": ["1", 1], "text": "blurry, low quality"}},
|
|
"4": {"class_type": "EmptyLatentImage",
|
|
"inputs": {"width": COMFY_BENCH_SIZE, "height": COMFY_BENCH_SIZE, "batch_size": 1}},
|
|
"5": {"class_type": "KSampler",
|
|
"inputs": {"model": ["1", 0], "positive": ["2", 0], "negative": ["3", 0],
|
|
"latent_image": ["4", 0], "seed": seed, "steps": COMFY_BENCH_STEPS,
|
|
"cfg": 7.0, "sampler_name": "euler", "scheduler": "normal",
|
|
"denoise": 1.0}},
|
|
"6": {"class_type": "VAEDecode", "inputs": {"samples": ["5", 0], "vae": ["1", 2]}},
|
|
"7": {"class_type": "PreviewImage", "inputs": {"images": ["6", 0]}},
|
|
}
|
|
|
|
|
|
async def _diffusion_benchmark(ckpt: str = COMFY_BENCH_CKPT,
|
|
timeout_s: float = 300.0) -> Dict[str, Any]:
|
|
"""Queue one fixed SDXL graph and time it. This is the compute-bound counterpart to
|
|
the decode benchmark, and the only way to tell whether the 'comfy' profile helps."""
|
|
client = vram_arbitrator._client(vram_arbitrator.COMFY_API_BASE, 30.0)
|
|
t0 = time.perf_counter()
|
|
try:
|
|
resp = await client.post("/prompt", json={"prompt": _comfy_workflow(ckpt),
|
|
"client_id": "hyperswap-autotune"})
|
|
if resp.status_code != 200:
|
|
return {"ok": False, "error": f"queue failed HTTP {resp.status_code}: {resp.text[:200]}"}
|
|
prompt_id = resp.json().get("prompt_id")
|
|
except Exception as e:
|
|
return {"ok": False, "error": f"queue failed: {e}"}
|
|
|
|
while (time.perf_counter() - t0) < timeout_s:
|
|
await asyncio.sleep(0.25)
|
|
try:
|
|
h = await client.get(f"/history/{prompt_id}")
|
|
if h.status_code != 200:
|
|
continue
|
|
entry = (h.json() or {}).get(prompt_id)
|
|
if not entry:
|
|
continue
|
|
status = entry.get("status", {})
|
|
if status.get("status_str") == "error" or not status.get("completed", True):
|
|
if status.get("status_str") == "error":
|
|
return {"ok": False, "error": "ComfyUI reported an execution error",
|
|
"wall_ms": round((time.perf_counter() - t0) * 1000, 2)}
|
|
if status.get("completed"):
|
|
wall = time.perf_counter() - t0
|
|
# ComfyUI stamps execution_start/success in the status messages; the delta
|
|
# between them excludes our polling overhead and the queue wait.
|
|
stamps = {}
|
|
for msg in status.get("messages", []):
|
|
if isinstance(msg, list) and len(msg) >= 2 and isinstance(msg[1], dict):
|
|
if "timestamp" in msg[1]:
|
|
stamps[msg[0]] = msg[1]["timestamp"]
|
|
exec_ms = None
|
|
if "execution_start" in stamps and "execution_success" in stamps:
|
|
exec_ms = round(stamps["execution_success"] - stamps["execution_start"], 2)
|
|
effective_ms = exec_ms or wall * 1000
|
|
# A graph that "finished" implausibly fast was served from ComfyUI's cache
|
|
# rather than executed; treat it as an invalid sample, not a record score.
|
|
cached = effective_ms < 250
|
|
return {
|
|
"ok": not cached,
|
|
"error": "result served from ComfyUI cache, not executed" if cached else None,
|
|
"wall_ms": round(wall * 1000, 2),
|
|
"exec_ms": exec_ms,
|
|
"steps": COMFY_BENCH_STEPS,
|
|
"it_per_sec": round(COMFY_BENCH_STEPS / (effective_ms / 1000), 3),
|
|
"degenerate": cached,
|
|
}
|
|
except Exception:
|
|
continue
|
|
return {"ok": False, "error": f"diffusion benchmark timed out after {timeout_s}s"}
|
|
|
|
|
|
class SweepState:
|
|
def __init__(self) -> None:
|
|
self.running = False
|
|
self.cancel = False
|
|
self.current: Optional[Dict[str, Any]] = None
|
|
self.last_result: Optional[Dict[str, Any]] = None
|
|
|
|
|
|
state = SweepState()
|
|
|
|
|
|
async def sweep(knob: str = "mem_offset_mhz",
|
|
profile: str = "ollama",
|
|
workload: str = "auto",
|
|
model: Optional[str] = None,
|
|
start: Optional[int] = None,
|
|
stop: Optional[int] = None,
|
|
step: Optional[int] = None,
|
|
repeats: int = 1,
|
|
max_steps: int = 6,
|
|
include_unlocked: bool = True,
|
|
apply_best: bool = False) -> Dict[str, Any]:
|
|
"""Sweep one clock offset and return the fastest stable value."""
|
|
if knob not in KNOBS:
|
|
return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"}
|
|
if state.running:
|
|
return {"success": False, "error": "a sweep is already running"}
|
|
|
|
comfy = await vram_arbitrator.get_comfyui_live_state()
|
|
if comfy.get("executing") or comfy.get("queue_remaining"):
|
|
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
|
|
|
|
# 'auto': tune the workload the profile is actually for.
|
|
if workload == "auto":
|
|
workload = "comfy" if profile == "comfy" else "ollama"
|
|
if workload not in ("ollama", "comfy"):
|
|
return {"success": False, "error": "workload must be 'ollama', 'comfy' or 'auto'"}
|
|
if workload == "comfy" and not comfy.get("online"):
|
|
return {"success": False, "error": "ComfyUI is not reachable; cannot run a diffusion sweep"}
|
|
|
|
if workload == "comfy":
|
|
model = model or COMFY_BENCH_CKPT
|
|
benchmark = lambda: _diffusion_benchmark(model)
|
|
metric = "it_per_sec"
|
|
else:
|
|
benchmark = lambda: _decode_benchmark(model)
|
|
metric = "tokens_per_sec"
|
|
|
|
if workload == "ollama" and not model:
|
|
ollama = await vram_arbitrator.get_ollama_live_state()
|
|
model = ollama.get("active_model_name")
|
|
if not model:
|
|
installed = ollama.get("installed_models") or []
|
|
if not installed:
|
|
return {"success": False, "error": "no Ollama model available to benchmark"}
|
|
model = installed[0].get("name")
|
|
benchmark = lambda: _decode_benchmark(model)
|
|
|
|
defaults = KNOBS[knob]
|
|
if defaults.get("kind") == "discrete":
|
|
if knob == "power_limit_w":
|
|
supported = _supported_power_limits()
|
|
else:
|
|
supported = defaults.get("values") or _supported_clocks(
|
|
"mem" if knob == "lock_mem_mhz" else "gr")
|
|
if not supported:
|
|
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
|
|
values = [v for v in supported
|
|
if (start is None or v >= start) and (stop is None or v <= stop)]
|
|
if not values:
|
|
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
|
|
values = _subsample(values, max_steps or 6)
|
|
# 0 means "no lock at all". That is the honest control for a profile whose whole
|
|
# premise is that locking the clock beats letting the card boost on its own.
|
|
if include_unlocked and not defaults.get("no_unlocked"):
|
|
values = [0] + values
|
|
start, stop, step = values[0], values[-1], None
|
|
else:
|
|
start = defaults["default_start"] if start is None else start
|
|
stop = defaults["default_stop"] if stop is None else stop
|
|
step = defaults["default_step"] if step is None else step
|
|
if step <= 0 or stop < start:
|
|
return {"success": False, "error": "invalid sweep range"}
|
|
values = list(range(start, stop + 1, step))
|
|
|
|
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
|
|
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
|
|
|
|
# Refuse to sweep a knob the driver is going to ignore.
|
|
effectiveness = _knob_effective(knob, profile, [v for v in values if v] or values,
|
|
baseline_value)
|
|
if not effectiveness["effective"]:
|
|
return {
|
|
"success": False,
|
|
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
|
|
"effectiveness": effectiveness,
|
|
}
|
|
|
|
state.running = True
|
|
state.cancel = False
|
|
results: List[Dict[str, Any]] = []
|
|
t_start = time.time()
|
|
|
|
try:
|
|
# Warm-up: load weights once up front so the first step does not pay the load cost.
|
|
vram_arbitrator.arbitrator.suspend_oc("autotune sweep")
|
|
await benchmark()
|
|
|
|
for value in values:
|
|
if state.cancel:
|
|
break
|
|
overclock_manager.apply_profile(profile, overrides={knob: value})
|
|
await asyncio.sleep(SETTLE_S)
|
|
step_started = time.time()
|
|
|
|
samples = []
|
|
for _ in range(max(repeats, 1)):
|
|
samples.append(await benchmark())
|
|
if state.cancel:
|
|
break
|
|
|
|
gpu = vram_arbitrator.get_gpu_hardware_stats()
|
|
xids = _xid_since(step_started)
|
|
ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")]
|
|
temp = gpu.get("temperature_c", 0) or 0
|
|
|
|
instability = []
|
|
if xids:
|
|
instability.append(f"kernel Xid: {xids[0][:120]}")
|
|
if len(ok_samples) < len(samples):
|
|
instability.append("benchmark failed or produced degenerate output")
|
|
if temp >= TEMP_CEILING_C:
|
|
instability.append(f"temperature ceiling hit ({temp}°C)")
|
|
|
|
tok_s = round(max((s.get(metric, 0.0) for s in ok_samples), default=0.0), 2)
|
|
row = {
|
|
"knob": knob,
|
|
"value": value,
|
|
"profile": profile,
|
|
"workload": workload,
|
|
"metric": metric,
|
|
"model": model,
|
|
"tokens_per_sec": tok_s,
|
|
"temp_c": temp,
|
|
"power_w": gpu.get("power_w"),
|
|
"clock_sm_mhz": gpu.get("clock_graphics_mhz"),
|
|
"clock_mem_mhz": gpu.get("clock_mem_mhz"),
|
|
"throttle_reasons": gpu.get("throttle_reasons"),
|
|
"stable": not instability,
|
|
"instability": "; ".join(instability) or None,
|
|
"samples": samples,
|
|
}
|
|
results.append(row)
|
|
telemetry_store.record_autotune({
|
|
"profile": profile, "knob": knob,
|
|
"core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"),
|
|
"mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"),
|
|
"tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"),
|
|
"stable": row["stable"], "instability": row["instability"],
|
|
"note": f"sweep {knob} {start}..{stop} step {step}",
|
|
})
|
|
state.current = {"knob": knob, "value": value, "stop": stop,
|
|
"tokens_per_sec": tok_s, "stable": row["stable"]}
|
|
logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, "
|
|
f"stable={row['stable']} {row['instability'] or ''}")
|
|
|
|
if not row["stable"]:
|
|
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
|
|
break
|
|
|
|
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
|
|
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
|
|
|
|
# Two different comparisons, and conflating them is how a sweep oversells itself.
|
|
# The first step of the range is not "baseline" unless it happens to be what the
|
|
# profile is already set to -- reporting "+102%" against the slowest value tried
|
|
# implies a speedup the user would not actually observe.
|
|
first_step = next((r for r in results if r["value"] == values[0]), None)
|
|
current = next((r for r in results if r["value"] == baseline_value), None)
|
|
gain_vs_first_step_pct = None
|
|
if best and first_step and first_step["tokens_per_sec"] > 0:
|
|
gain_vs_first_step_pct = round(
|
|
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
|
|
gain_vs_current_pct = None
|
|
if best and current and current["tokens_per_sec"] > 0:
|
|
gain_vs_current_pct = round(
|
|
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
|
|
|
|
applied = None
|
|
if apply_best and best:
|
|
overclock_manager.set_profile(profile, {knob: best["value"]})
|
|
applied = {knob: best["value"], "profile": profile}
|
|
logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'")
|
|
|
|
result = {
|
|
"success": True,
|
|
"knob": knob,
|
|
"profile": profile,
|
|
"workload": workload,
|
|
"metric": metric,
|
|
"model": model,
|
|
"range": {"start": start, "stop": stop, "step": step, "values": values},
|
|
"effectiveness": effectiveness,
|
|
"steps_run": len(results),
|
|
"duration_s": round(time.time() - t_start, 1),
|
|
"cancelled": state.cancel,
|
|
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
|
|
"clock_sm_mhz")} if best else None,
|
|
"current_profile_value": baseline_value,
|
|
"first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
|
|
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
|
|
"gain_vs_first_step_pct": gain_vs_first_step_pct,
|
|
"gain_vs_current_pct": gain_vs_current_pct,
|
|
"gain_note": ("compared against the profile's current setting"
|
|
if current is not None else
|
|
f"the profile's current value ({baseline_value}) was not in the "
|
|
f"swept range, so only the spread across tried values is shown"),
|
|
"applied_to_profile": applied,
|
|
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
|
|
for r in results if not r["stable"]), None),
|
|
"table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w",
|
|
"clock_mem_mhz", "clock_sm_mhz", "stable",
|
|
"instability")} for r in results],
|
|
}
|
|
state.last_result = result
|
|
return result
|
|
finally:
|
|
# Always hand the card back exactly as we found it.
|
|
state.running = False
|
|
state.current = None
|
|
vram_arbitrator.arbitrator.resume_oc(profile)
|
|
try:
|
|
overclock_manager.apply_profile(profile)
|
|
logger.info(f"autotune restored profile '{profile}'")
|
|
except Exception as e:
|
|
logger.error(f"autotune failed to restore profile, forcing stock: {e}")
|
|
overclock_manager.restore_safe("autotune restore failed")
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
return {
|
|
"running": state.running,
|
|
"current": state.current,
|
|
"last_result": state.last_result,
|
|
"knobs": KNOBS,
|
|
"history": telemetry_store.autotune_history(100),
|
|
}
|
|
|
|
|
|
def cancel() -> Dict[str, Any]:
|
|
if not state.running:
|
|
return {"cancelled": False, "reason": "no sweep running"}
|
|
state.cancel = True
|
|
return {"cancelled": True}
|