Tune profiles from measurement; add a diffusion benchmark to close the loop

The profiles were hand-written and had never been checked against the hardware. Adding
a ComfyUI benchmark alongside the existing decode one made the compute side measurable
for the first time, and most of what the profiles configured turned out to do nothing.

Measured on this card (RTX 4080 SUPER, driver 595.84):

- LLM decode is not power-bound: 73.0-73.5 tok/s flat from 222W to 370W, with the card
  never drawing more than 224W at any limit. The ollama profile's 370W did nothing.
- Diffusion is power-bound: 5.48 it/s @222W rising to 6.71 @370W, so comfy's 370W is
  worth a real +2.8% over the 320W stock default.
- Clock locks did nothing for either workload: 72.6 tok/s locked at 11251MHz vs 72.7
  unlocked; 6.77 it/s locked at 3105MHz vs 6.73 unlocked, and 6.78 at 2400MHz.
- Memory bandwidth is still the decode bottleneck (5001MHz halves throughput to 35.9
  tok/s), confirming the profile's premise -- the card just gets there unaided.
- Fans: 48,435 samples show 81C all-time max and zero thermal throttle events, while
  the ollama profile held 49.6C average by running fans at 87%. All profiles now use
  automatic fans and let the thermal governor escalate on demand.

Code changes supporting that:
- _diffusion_benchmark() queues a fixed SDXL graph via ComfyUI's API. The seed must
  vary per run: ComfyUI caches by node inputs, so a fixed seed returned in ~1ms without
  executing. Implausibly fast results are now rejected as cache hits rather than
  recorded as record scores.
- The arbitrator's automatic profile switching is suspended during a sweep. A diffusion
  benchmark trips trigger_comfy_priority, which reapplies the whole profile and would
  silently overwrite the clock being measured.
- _supported_clocks() queries the mem,gr pair; asking for a single field returned one
  column and reading index 1 yielded an empty list rather than an error. Graphics clocks
  are subsampled (the card enumerates 194 of them) and lock sweeps include an explicit
  unlocked control step.
- offsets_supported() probes once and apply_profile skips inert offset levers with an
  explanation instead of pretending they applied.
- Profiles carry a 'measured' field recording the evidence behind each setting.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-08-28 11:18:35 -07:00
parent a30444ef8e
commit c689ec8711
6 changed files with 373 additions and 56 deletions

View File

@@ -15,6 +15,7 @@ Safety properties:
"""
import asyncio
import logging
import random
import subprocess
import time
from typing import Any, Dict, List, Optional
@@ -46,28 +47,74 @@ KNOBS = {
"values": None}, # filled from the card's supported clock list
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
"values": None},
# The one lever this driver definitely honours. Worth knowing whether the extra
# watts actually buy throughput, or just heat and fan noise.
"power_limit_w": {"kind": "discrete", "verify": "power_limit",
"values": None, "no_unlocked": True},
}
def _supported_clocks(which: str = "mem") -> List[int]:
"""Discrete clock values the card will actually accept for -lmc / -lgc."""
"""Discrete clock values the card will actually accept for -lmc / -lgc.
Always queried as the mem,gr pair: asking for a single field returns one column, and
reading index 1 from it silently yields an empty list rather than an error.
"""
try:
proc = subprocess.run(
["nvidia-smi", f"--query-supported-clocks={'mem' if which == 'mem' else 'gr'}",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
col = 0 if which == "mem" else 1
vals = set()
["nvidia-smi", "--query-supported-clocks=mem,gr", "--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=15)
rows = []
for line in proc.stdout.splitlines():
parts = [p.strip() for p in line.split(",")]
if len(parts) > col and parts[col].isdigit():
vals.add(int(parts[col]))
return sorted(vals)
if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit():
rows.append((int(parts[0]), int(parts[1])))
if not rows:
return []
if which == "mem":
return sorted({m for m, _ in rows})
# Graphics clocks are enumerated per memory clock; take the list for the highest
# memory clock, which is the one any real workload runs at.
top_mem = max(m for m, _ in rows)
return sorted({g for m, g in rows if m == top_mem})
except Exception as e:
logger.debug(f"supported clock query failed: {e}")
return []
def _supported_power_limits(steps: int = 5) -> List[int]:
"""Power limits between the card's minimum and maximum, in even increments."""
try:
proc = subprocess.run(
["nvidia-smi", "--query-gpu=power.min_limit,power.max_limit,power.default_limit",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
parts = [p.strip() for p in proc.stdout.strip().split(",")]
lo, hi, default = (int(float(parts[0])), int(float(parts[1])), int(float(parts[2])))
except Exception as e:
logger.debug(f"power limit query failed: {e}")
return []
# Start at 60% of max -- below that the card is not doing useful work for these
# workloads -- and always include the stock default as a reference point.
lo = max(lo, int(hi * 0.6))
span = hi - lo
vals = {lo + round(i * span / (steps - 1)) for i in range(steps)}
vals.add(default)
return sorted(v for v in vals if lo <= v <= hi)
def _subsample(values: List[int], max_steps: int) -> List[int]:
"""Evenly spaced subset, always keeping the endpoints.
The card enumerates ~194 graphics clocks in 15 MHz increments; benchmarking every one
would take hours and tell us nothing that a handful of well-spread points does not.
"""
if len(values) <= max_steps:
return values
idx = [round(i * (len(values) - 1) / (max_steps - 1)) for i in range(max_steps)]
return sorted({values[i] for i in idx})
def _read_hw(field: str) -> Optional[float]:
"""Read back the hardware state a knob is supposed to move."""
gpu = vram_arbitrator.get_gpu_hardware_stats()
@@ -75,6 +122,8 @@ def _read_hw(field: str) -> Optional[float]:
return gpu.get("clock_mem_mhz")
if field == "clock_sm":
return gpu.get("clock_graphics_mhz")
if field == "power_limit":
return gpu.get("power_limit_w")
if field in ("mem_offset", "core_offset"):
r = overclock_manager._nvidia_settings(
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
@@ -174,6 +223,97 @@ async def _decode_benchmark(model: str) -> Dict[str, Any]:
}
# A fixed SDXL txt2img graph. Deterministic seed/steps/resolution so every step of a
# sweep does identical work and the only variable is the clock. PreviewImage rather than
# SaveImage keeps benchmark runs out of the user's output gallery.
COMFY_BENCH_CKPT = "sd_xl_base_1.0.safetensors"
COMFY_BENCH_STEPS = 20
COMFY_BENCH_SIZE = 1024
def _comfy_workflow(ckpt: str = COMFY_BENCH_CKPT, seed: Optional[int] = None) -> Dict[str, Any]:
# The seed must vary per run. ComfyUI caches by node inputs, so a fixed seed makes the
# second and later benchmarks return in ~1ms without executing anything at all. The
# cost of the graph is identical regardless of seed, so this costs no comparability.
seed = random.randint(1, 2**31) if seed is None else seed
return {
"1": {"class_type": "CheckpointLoaderSimple", "inputs": {"ckpt_name": ckpt}},
"2": {"class_type": "CLIPTextEncode",
"inputs": {"clip": ["1", 1],
"text": "a detailed photograph of a mountain range at sunrise"}},
"3": {"class_type": "CLIPTextEncode",
"inputs": {"clip": ["1", 1], "text": "blurry, low quality"}},
"4": {"class_type": "EmptyLatentImage",
"inputs": {"width": COMFY_BENCH_SIZE, "height": COMFY_BENCH_SIZE, "batch_size": 1}},
"5": {"class_type": "KSampler",
"inputs": {"model": ["1", 0], "positive": ["2", 0], "negative": ["3", 0],
"latent_image": ["4", 0], "seed": seed, "steps": COMFY_BENCH_STEPS,
"cfg": 7.0, "sampler_name": "euler", "scheduler": "normal",
"denoise": 1.0}},
"6": {"class_type": "VAEDecode", "inputs": {"samples": ["5", 0], "vae": ["1", 2]}},
"7": {"class_type": "PreviewImage", "inputs": {"images": ["6", 0]}},
}
async def _diffusion_benchmark(ckpt: str = COMFY_BENCH_CKPT,
timeout_s: float = 300.0) -> Dict[str, Any]:
"""Queue one fixed SDXL graph and time it. This is the compute-bound counterpart to
the decode benchmark, and the only way to tell whether the 'comfy' profile helps."""
client = vram_arbitrator._client(vram_arbitrator.COMFY_API_BASE, 30.0)
t0 = time.perf_counter()
try:
resp = await client.post("/prompt", json={"prompt": _comfy_workflow(ckpt),
"client_id": "hyperswap-autotune"})
if resp.status_code != 200:
return {"ok": False, "error": f"queue failed HTTP {resp.status_code}: {resp.text[:200]}"}
prompt_id = resp.json().get("prompt_id")
except Exception as e:
return {"ok": False, "error": f"queue failed: {e}"}
while (time.perf_counter() - t0) < timeout_s:
await asyncio.sleep(0.25)
try:
h = await client.get(f"/history/{prompt_id}")
if h.status_code != 200:
continue
entry = (h.json() or {}).get(prompt_id)
if not entry:
continue
status = entry.get("status", {})
if status.get("status_str") == "error" or not status.get("completed", True):
if status.get("status_str") == "error":
return {"ok": False, "error": "ComfyUI reported an execution error",
"wall_ms": round((time.perf_counter() - t0) * 1000, 2)}
if status.get("completed"):
wall = time.perf_counter() - t0
# ComfyUI stamps execution_start/success in the status messages; the delta
# between them excludes our polling overhead and the queue wait.
stamps = {}
for msg in status.get("messages", []):
if isinstance(msg, list) and len(msg) >= 2 and isinstance(msg[1], dict):
if "timestamp" in msg[1]:
stamps[msg[0]] = msg[1]["timestamp"]
exec_ms = None
if "execution_start" in stamps and "execution_success" in stamps:
exec_ms = round(stamps["execution_success"] - stamps["execution_start"], 2)
effective_ms = exec_ms or wall * 1000
# A graph that "finished" implausibly fast was served from ComfyUI's cache
# rather than executed; treat it as an invalid sample, not a record score.
cached = effective_ms < 250
return {
"ok": not cached,
"error": "result served from ComfyUI cache, not executed" if cached else None,
"wall_ms": round(wall * 1000, 2),
"exec_ms": exec_ms,
"steps": COMFY_BENCH_STEPS,
"it_per_sec": round(COMFY_BENCH_STEPS / (effective_ms / 1000), 3),
"degenerate": cached,
}
except Exception:
continue
return {"ok": False, "error": f"diffusion benchmark timed out after {timeout_s}s"}
class SweepState:
def __init__(self) -> None:
self.running = False
@@ -187,11 +327,14 @@ state = SweepState()
async def sweep(knob: str = "mem_offset_mhz",
profile: str = "ollama",
workload: str = "auto",
model: Optional[str] = None,
start: Optional[int] = None,
stop: Optional[int] = None,
step: Optional[int] = None,
repeats: int = 1,
max_steps: int = 6,
include_unlocked: bool = True,
apply_best: bool = False) -> Dict[str, Any]:
"""Sweep one clock offset and return the fastest stable value."""
if knob not in KNOBS:
@@ -203,7 +346,23 @@ async def sweep(knob: str = "mem_offset_mhz",
if comfy.get("executing") or comfy.get("queue_remaining"):
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
if not model:
# 'auto': tune the workload the profile is actually for.
if workload == "auto":
workload = "comfy" if profile == "comfy" else "ollama"
if workload not in ("ollama", "comfy"):
return {"success": False, "error": "workload must be 'ollama', 'comfy' or 'auto'"}
if workload == "comfy" and not comfy.get("online"):
return {"success": False, "error": "ComfyUI is not reachable; cannot run a diffusion sweep"}
if workload == "comfy":
model = model or COMFY_BENCH_CKPT
benchmark = lambda: _diffusion_benchmark(model)
metric = "it_per_sec"
else:
benchmark = lambda: _decode_benchmark(model)
metric = "tokens_per_sec"
if workload == "ollama" and not model:
ollama = await vram_arbitrator.get_ollama_live_state()
model = ollama.get("active_model_name")
if not model:
@@ -211,17 +370,26 @@ async def sweep(knob: str = "mem_offset_mhz",
if not installed:
return {"success": False, "error": "no Ollama model available to benchmark"}
model = installed[0].get("name")
benchmark = lambda: _decode_benchmark(model)
defaults = KNOBS[knob]
if defaults.get("kind") == "discrete":
supported = defaults.get("values") or _supported_clocks(
"mem" if knob == "lock_mem_mhz" else "gr")
if knob == "power_limit_w":
supported = _supported_power_limits()
else:
supported = defaults.get("values") or _supported_clocks(
"mem" if knob == "lock_mem_mhz" else "gr")
if not supported:
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
values = [v for v in supported
if (start is None or v >= start) and (stop is None or v <= stop)]
if not values:
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
values = _subsample(values, max_steps or 6)
# 0 means "no lock at all". That is the honest control for a profile whose whole
# premise is that locking the clock beats letting the card boost on its own.
if include_unlocked and not defaults.get("no_unlocked"):
values = [0] + values
start, stop, step = values[0], values[-1], None
else:
start = defaults["default_start"] if start is None else start
@@ -235,7 +403,8 @@ async def sweep(knob: str = "mem_offset_mhz",
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
# Refuse to sweep a knob the driver is going to ignore.
effectiveness = _knob_effective(knob, profile, values, baseline_value)
effectiveness = _knob_effective(knob, profile, [v for v in values if v] or values,
baseline_value)
if not effectiveness["effective"]:
return {
"success": False,
@@ -249,8 +418,9 @@ async def sweep(knob: str = "mem_offset_mhz",
t_start = time.time()
try:
# Load the model once up front so the first step does not pay the load cost.
await _decode_benchmark(model)
# Warm-up: load weights once up front so the first step does not pay the load cost.
vram_arbitrator.arbitrator.suspend_oc("autotune sweep")
await benchmark()
for value in values:
if state.cancel:
@@ -261,7 +431,7 @@ async def sweep(knob: str = "mem_offset_mhz",
samples = []
for _ in range(max(repeats, 1)):
samples.append(await _decode_benchmark(model))
samples.append(await benchmark())
if state.cancel:
break
@@ -278,11 +448,13 @@ async def sweep(knob: str = "mem_offset_mhz",
if temp >= TEMP_CEILING_C:
instability.append(f"temperature ceiling hit ({temp}°C)")
tok_s = round(max((s["tokens_per_sec"] for s in ok_samples), default=0.0), 2)
tok_s = round(max((s.get(metric, 0.0) for s in ok_samples), default=0.0), 2)
row = {
"knob": knob,
"value": value,
"profile": profile,
"workload": workload,
"metric": metric,
"model": model,
"tokens_per_sec": tok_s,
"temp_c": temp,
@@ -340,6 +512,8 @@ async def sweep(knob: str = "mem_offset_mhz",
"success": True,
"knob": knob,
"profile": profile,
"workload": workload,
"metric": metric,
"model": model,
"range": {"start": start, "stop": stop, "step": step, "values": values},
"effectiveness": effectiveness,
@@ -370,6 +544,7 @@ async def sweep(knob: str = "mem_offset_mhz",
# Always hand the card back exactly as we found it.
state.running = False
state.current = None
vram_arbitrator.arbitrator.resume_oc(profile)
try:
overclock_manager.apply_profile(profile)
logger.info(f"autotune restored profile '{profile}'")