Files
gpu-program-swapper/autotune.py
drjones 868d82794d Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-01 13:40:30 -07:00

575 lines
26 KiB
Python

"""Closed-loop overclock autotuner.
The profiles in this repo were hand-tuned and had already drifted apart from the defaults
in overclock_manager.py, with no record of which numbers were actually faster. This module
answers that empirically: it walks a clock offset upward, measures real decode throughput
at each step, watches for instability, and reports the highest setting that was both
stable and fastest.
Safety properties:
* The original profile is always restored, including on exception or cancellation.
* A sweep refuses to start while ComfyUI is executing, so it cannot corrupt someone's
render by yanking clocks mid-graph.
* Every step is bounded by a temperature ceiling and checked for kernel Xid messages,
and the sweep stops climbing the moment a step looks unstable.
"""
import asyncio
import logging
import random
import subprocess
import time
from typing import Any, Dict, List, Optional
import overclock_manager
import telemetry_store
import vram_arbitrator
logger = logging.getLogger("autotune")
BENCH_PROMPT = ("Write a detailed technical explanation of how virtual memory paging "
"works in a modern operating system kernel.")
BENCH_TOKENS = 160
SETTLE_S = 2.5
TEMP_CEILING_C = 84.0
KNOBS = {
# Clock offsets go through nvidia-settings. On some drivers (595.84 here) the
# attribute is accepted and then silently ignored -- assigning 0 reports success and
# reads back 250 -- so a sweep of these can measure pure noise. _knob_effective()
# checks before any sweep runs.
"mem_offset_mhz": {"default_start": 0, "default_stop": 1000, "default_step": 100,
"kind": "offset", "verify": "mem_offset"},
"core_offset_mhz": {"default_start": 0, "default_stop": 300, "default_step": 25,
"kind": "offset", "verify": "core_offset"},
# Clock locks go through nvidia-smi and do work on the open/proprietary module alike.
# Memory clock is the lever that matters for LLM decode, which is bandwidth bound.
"lock_mem_mhz": {"kind": "discrete", "verify": "clock_mem",
"values": None}, # filled from the card's supported clock list
"lock_core_max": {"kind": "discrete", "verify": "clock_sm",
"values": None},
# The one lever this driver definitely honours. Worth knowing whether the extra
# watts actually buy throughput, or just heat and fan noise.
"power_limit_w": {"kind": "discrete", "verify": "power_limit",
"values": None, "no_unlocked": True},
}
def _supported_clocks(which: str = "mem") -> List[int]:
"""Discrete clock values the card will actually accept for -lmc / -lgc.
Always queried as the mem,gr pair: asking for a single field returns one column, and
reading index 1 from it silently yields an empty list rather than an error.
"""
try:
proc = subprocess.run(
["nvidia-smi", "--query-supported-clocks=mem,gr", "--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=15)
rows = []
for line in proc.stdout.splitlines():
parts = [p.strip() for p in line.split(",")]
if len(parts) >= 2 and parts[0].isdigit() and parts[1].isdigit():
rows.append((int(parts[0]), int(parts[1])))
if not rows:
return []
if which == "mem":
return sorted({m for m, _ in rows})
# Graphics clocks are enumerated per memory clock; take the list for the highest
# memory clock, which is the one any real workload runs at.
top_mem = max(m for m, _ in rows)
return sorted({g for m, g in rows if m == top_mem})
except Exception as e:
logger.debug(f"supported clock query failed: {e}")
return []
def _supported_power_limits(steps: int = 5) -> List[int]:
"""Power limits between the card's minimum and maximum, in even increments."""
try:
proc = subprocess.run(
["nvidia-smi", "--query-gpu=power.min_limit,power.max_limit,power.default_limit",
"--format=csv,noheader,nounits"],
capture_output=True, text=True, timeout=10)
parts = [p.strip() for p in proc.stdout.strip().split(",")]
lo, hi, default = (int(float(parts[0])), int(float(parts[1])), int(float(parts[2])))
except Exception as e:
logger.debug(f"power limit query failed: {e}")
return []
# Start at 60% of max -- below that the card is not doing useful work for these
# workloads -- and always include the stock default as a reference point.
lo = max(lo, int(hi * 0.6))
span = hi - lo
vals = {lo + round(i * span / (steps - 1)) for i in range(steps)}
vals.add(default)
return sorted(v for v in vals if lo <= v <= hi)
def _subsample(values: List[int], max_steps: int) -> List[int]:
"""Evenly spaced subset, always keeping the endpoints.
The card enumerates ~194 graphics clocks in 15 MHz increments; benchmarking every one
would take hours and tell us nothing that a handful of well-spread points does not.
"""
if len(values) <= max_steps:
return values
if max_steps < 2:
# One step cannot span a range; take the top, which is what a caller asking for
# a single sample almost certainly wants. Guards a ZeroDivisionError below.
return values[-1:]
idx = [round(i * (len(values) - 1) / (max_steps - 1)) for i in range(max_steps)]
return sorted({values[i] for i in idx})
def _read_hw(field: str) -> Optional[float]:
"""Read back the hardware state a knob is supposed to move."""
gpu = vram_arbitrator.get_gpu_hardware_stats()
if field == "clock_mem":
return gpu.get("clock_mem_mhz")
if field == "clock_sm":
return gpu.get("clock_graphics_mhz")
if field == "power_limit":
return gpu.get("power_limit_w")
if field in ("mem_offset", "core_offset"):
r = overclock_manager._nvidia_settings(
"-q", f"[gpu:0]/{'GPUMemoryTransferRateOffset' if field == 'mem_offset' else 'GPUGraphicsClockOffset'}[3]")
for line in (r.get("out") or "").splitlines():
if "Attribute" in line and "):" in line:
try:
return float(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
return None
def _knob_effective(knob: str, profile: str, candidates: List[int],
baseline_value: int) -> Dict[str, Any]:
"""Verify a knob actually moves the hardware before we sweep it.
Without this the tuner happily reports "best = the highest value tried" from run-to-run
benchmark noise on a knob the driver is ignoring.
The probe value is chosen as the candidate *furthest* from where the hardware currently
sits. Probing with the maximum is not good enough: if the card already happens to be at
its top clock, setting it there again moves nothing and a perfectly good knob looks
broken.
"""
field = KNOBS[knob].get("verify")
before = _read_hw(field)
if before is not None and candidates:
probe_value = max(candidates, key=lambda v: abs(v - before))
else:
probe_value = candidates[-1] if candidates else 0
overclock_manager.apply_profile(profile, overrides={knob: probe_value})
time.sleep(2.0)
after = _read_hw(field)
overclock_manager.apply_profile(profile, overrides={knob: baseline_value})
moved = (before is not None and after is not None and abs(after - before) > 1e-6)
return {
"effective": bool(moved),
"field": field,
"before": before,
"after": after,
"probe_value": probe_value,
"detail": (f"{field} moved {before} -> {after}" if moved else
f"{field} stayed at {after} after setting {knob}={probe_value}; "
f"this driver accepts the setting and ignores it"),
}
def _xid_since(since_ts: float) -> List[str]:
"""Look for NVIDIA Xid errors in the kernel log — the clearest instability signal."""
try:
since = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(since_ts))
proc = subprocess.run(
["journalctl", "-k", "--since", since, "--no-pager", "-q"],
capture_output=True, text=True, timeout=10,
)
return [ln.strip() for ln in proc.stdout.splitlines()
if "Xid" in ln or "NVRM:" in ln]
except Exception as e:
logger.debug(f"Xid check unavailable: {e}")
return []
async def _decode_benchmark(model: str) -> Dict[str, Any]:
"""One fixed decode run. Throughput here is the thing being optimised."""
client = vram_arbitrator._client(vram_arbitrator.OLLAMA_API_BASE, 300.0)
t0 = time.perf_counter()
resp = await client.post("/api/generate", json={
"model": model,
"prompt": BENCH_PROMPT,
"stream": False,
"keep_alive": "10m",
"options": {"num_predict": BENCH_TOKENS, "temperature": 0.0, "seed": 42},
})
wall_ms = round((time.perf_counter() - t0) * 1000, 2)
if resp.status_code != 200:
return {"ok": False, "error": f"HTTP {resp.status_code}: {resp.text[:200]}",
"wall_ms": wall_ms}
data = resp.json()
eval_ms = data.get("eval_duration", 0) / 1e6
eval_count = data.get("eval_count", 0)
# Reasoning models put their output in `thinking` and leave `response` empty, so a
# check that only looked at `response` flagged every one of them as degenerate.
text = ((data.get("response") or "") + " " + (data.get("thinking") or "")).strip()
return {
"ok": True,
"tokens_per_sec": round(eval_count / (eval_ms / 1000), 2) if eval_ms > 0 else 0.0,
"eval_count": eval_count,
"eval_ms": round(eval_ms, 2),
"prompt_eval_ms": round(data.get("prompt_eval_duration", 0) / 1e6, 2),
"wall_ms": wall_ms,
"response_chars": len(text),
# Corruption signals: the model stopped far short of the token budget, or it
# produced text that is pure repetition. Token count is the primary signal --
# empty text alone is not enough, since output can arrive in other fields.
"degenerate": (eval_count < BENCH_TOKENS * 0.5
or (len(text) > 0 and len(set(text.split())) < 8)),
}
# A fixed SDXL txt2img graph. Deterministic seed/steps/resolution so every step of a
# sweep does identical work and the only variable is the clock. PreviewImage rather than
# SaveImage keeps benchmark runs out of the user's output gallery.
COMFY_BENCH_CKPT = "sd_xl_base_1.0.safetensors"
COMFY_BENCH_STEPS = 20
COMFY_BENCH_SIZE = 1024
def _comfy_workflow(ckpt: str = COMFY_BENCH_CKPT, seed: Optional[int] = None) -> Dict[str, Any]:
# The seed must vary per run. ComfyUI caches by node inputs, so a fixed seed makes the
# second and later benchmarks return in ~1ms without executing anything at all. The
# cost of the graph is identical regardless of seed, so this costs no comparability.
seed = random.randint(1, 2**31) if seed is None else seed
return {
"1": {"class_type": "CheckpointLoaderSimple", "inputs": {"ckpt_name": ckpt}},
"2": {"class_type": "CLIPTextEncode",
"inputs": {"clip": ["1", 1],
"text": "a detailed photograph of a mountain range at sunrise"}},
"3": {"class_type": "CLIPTextEncode",
"inputs": {"clip": ["1", 1], "text": "blurry, low quality"}},
"4": {"class_type": "EmptyLatentImage",
"inputs": {"width": COMFY_BENCH_SIZE, "height": COMFY_BENCH_SIZE, "batch_size": 1}},
"5": {"class_type": "KSampler",
"inputs": {"model": ["1", 0], "positive": ["2", 0], "negative": ["3", 0],
"latent_image": ["4", 0], "seed": seed, "steps": COMFY_BENCH_STEPS,
"cfg": 7.0, "sampler_name": "euler", "scheduler": "normal",
"denoise": 1.0}},
"6": {"class_type": "VAEDecode", "inputs": {"samples": ["5", 0], "vae": ["1", 2]}},
"7": {"class_type": "PreviewImage", "inputs": {"images": ["6", 0]}},
}
async def _diffusion_benchmark(ckpt: str = COMFY_BENCH_CKPT,
timeout_s: float = 300.0) -> Dict[str, Any]:
"""Queue one fixed SDXL graph and time it. This is the compute-bound counterpart to
the decode benchmark, and the only way to tell whether the 'comfy' profile helps."""
client = vram_arbitrator._client(vram_arbitrator.COMFY_API_BASE, 30.0)
t0 = time.perf_counter()
try:
resp = await client.post("/prompt", json={"prompt": _comfy_workflow(ckpt),
"client_id": "hyperswap-autotune"})
if resp.status_code != 200:
return {"ok": False, "error": f"queue failed HTTP {resp.status_code}: {resp.text[:200]}"}
prompt_id = resp.json().get("prompt_id")
except Exception as e:
return {"ok": False, "error": f"queue failed: {e}"}
while (time.perf_counter() - t0) < timeout_s:
await asyncio.sleep(0.25)
try:
h = await client.get(f"/history/{prompt_id}")
if h.status_code != 200:
continue
entry = (h.json() or {}).get(prompt_id)
if not entry:
continue
status = entry.get("status", {})
if status.get("status_str") == "error" or not status.get("completed", True):
if status.get("status_str") == "error":
return {"ok": False, "error": "ComfyUI reported an execution error",
"wall_ms": round((time.perf_counter() - t0) * 1000, 2)}
if status.get("completed"):
wall = time.perf_counter() - t0
# ComfyUI stamps execution_start/success in the status messages; the delta
# between them excludes our polling overhead and the queue wait.
stamps = {}
for msg in status.get("messages", []):
if isinstance(msg, list) and len(msg) >= 2 and isinstance(msg[1], dict):
if "timestamp" in msg[1]:
stamps[msg[0]] = msg[1]["timestamp"]
exec_ms = None
if "execution_start" in stamps and "execution_success" in stamps:
exec_ms = round(stamps["execution_success"] - stamps["execution_start"], 2)
effective_ms = exec_ms or wall * 1000
# A graph that "finished" implausibly fast was served from ComfyUI's cache
# rather than executed; treat it as an invalid sample, not a record score.
cached = effective_ms < 250
return {
"ok": not cached,
"error": "result served from ComfyUI cache, not executed" if cached else None,
"wall_ms": round(wall * 1000, 2),
"exec_ms": exec_ms,
"steps": COMFY_BENCH_STEPS,
"it_per_sec": round(COMFY_BENCH_STEPS / (effective_ms / 1000), 3),
"degenerate": cached,
}
except Exception:
continue
return {"ok": False, "error": f"diffusion benchmark timed out after {timeout_s}s"}
class SweepState:
def __init__(self) -> None:
self.running = False
self.cancel = False
self.current: Optional[Dict[str, Any]] = None
self.last_result: Optional[Dict[str, Any]] = None
state = SweepState()
async def sweep(knob: str = "mem_offset_mhz",
profile: str = "ollama",
workload: str = "auto",
model: Optional[str] = None,
start: Optional[int] = None,
stop: Optional[int] = None,
step: Optional[int] = None,
repeats: int = 1,
max_steps: int = 6,
include_unlocked: bool = True,
apply_best: bool = False) -> Dict[str, Any]:
"""Sweep one clock offset and return the fastest stable value."""
if knob not in KNOBS:
return {"success": False, "error": f"unknown knob '{knob}'; try {list(KNOBS)}"}
if state.running:
return {"success": False, "error": "a sweep is already running"}
comfy = await vram_arbitrator.get_comfyui_live_state()
if comfy.get("executing") or comfy.get("queue_remaining"):
return {"success": False, "error": "ComfyUI is busy; refusing to change clocks mid-render"}
# 'auto': tune the workload the profile is actually for.
if workload == "auto":
workload = "comfy" if profile == "comfy" else "ollama"
if workload not in ("ollama", "comfy"):
return {"success": False, "error": "workload must be 'ollama', 'comfy' or 'auto'"}
if workload == "comfy" and not comfy.get("online"):
return {"success": False, "error": "ComfyUI is not reachable; cannot run a diffusion sweep"}
if workload == "comfy":
model = model or COMFY_BENCH_CKPT
benchmark = lambda: _diffusion_benchmark(model)
metric = "it_per_sec"
else:
benchmark = lambda: _decode_benchmark(model)
metric = "tokens_per_sec"
if workload == "ollama" and not model:
ollama = await vram_arbitrator.get_ollama_live_state()
model = ollama.get("active_model_name")
if not model:
installed = ollama.get("installed_models") or []
if not installed:
return {"success": False, "error": "no Ollama model available to benchmark"}
model = installed[0].get("name")
benchmark = lambda: _decode_benchmark(model)
defaults = KNOBS[knob]
if defaults.get("kind") == "discrete":
if knob == "power_limit_w":
supported = _supported_power_limits()
else:
supported = defaults.get("values") or _supported_clocks(
"mem" if knob == "lock_mem_mhz" else "gr")
if not supported:
return {"success": False, "error": f"card reported no supported clocks for {knob}"}
values = [v for v in supported
if (start is None or v >= start) and (stop is None or v <= stop)]
if not values:
return {"success": False, "error": f"no supported values in range; card offers {supported}"}
values = _subsample(values, max_steps or 6)
# 0 means "no lock at all". That is the honest control for a profile whose whole
# premise is that locking the clock beats letting the card boost on its own.
if include_unlocked and not defaults.get("no_unlocked"):
values = [0] + values
start, stop, step = values[0], values[-1], None
else:
start = defaults["default_start"] if start is None else start
stop = defaults["default_stop"] if stop is None else stop
step = defaults["default_step"] if step is None else step
if step <= 0 or stop < start:
return {"success": False, "error": "invalid sweep range"}
values = list(range(start, stop + 1, step))
baseline_cfg = overclock_manager.load_profiles().get(profile, {})
baseline_value = int(baseline_cfg.get(knob, 0) or 0)
# Refuse to sweep a knob the driver is going to ignore.
effectiveness = _knob_effective(knob, profile, [v for v in values if v] or values,
baseline_value)
if not effectiveness["effective"]:
return {
"success": False,
"error": f"'{knob}' does not move this GPU: {effectiveness['detail']}",
"effectiveness": effectiveness,
}
state.running = True
state.cancel = False
results: List[Dict[str, Any]] = []
t_start = time.time()
try:
# Warm-up: load weights once up front so the first step does not pay the load cost.
vram_arbitrator.arbitrator.suspend_oc("autotune sweep")
await benchmark()
for value in values:
if state.cancel:
break
overclock_manager.apply_profile(profile, overrides={knob: value})
await asyncio.sleep(SETTLE_S)
step_started = time.time()
samples = []
for _ in range(max(repeats, 1)):
samples.append(await benchmark())
if state.cancel:
break
gpu = vram_arbitrator.get_gpu_hardware_stats()
xids = _xid_since(step_started)
ok_samples = [s for s in samples if s.get("ok") and not s.get("degenerate")]
temp = gpu.get("temperature_c", 0) or 0
instability = []
if xids:
instability.append(f"kernel Xid: {xids[0][:120]}")
if len(ok_samples) < len(samples):
instability.append("benchmark failed or produced degenerate output")
if temp >= TEMP_CEILING_C:
instability.append(f"temperature ceiling hit ({temp}°C)")
tok_s = round(max((s.get(metric, 0.0) for s in ok_samples), default=0.0), 2)
row = {
"knob": knob,
"value": value,
"profile": profile,
"workload": workload,
"metric": metric,
"model": model,
"tokens_per_sec": tok_s,
"temp_c": temp,
"power_w": gpu.get("power_w"),
"clock_sm_mhz": gpu.get("clock_graphics_mhz"),
"clock_mem_mhz": gpu.get("clock_mem_mhz"),
"throttle_reasons": gpu.get("throttle_reasons"),
"stable": not instability,
"instability": "; ".join(instability) or None,
"samples": samples,
}
results.append(row)
telemetry_store.record_autotune({
"profile": profile, "knob": knob,
"core_offset_mhz": value if knob == "core_offset_mhz" else baseline_cfg.get("core_offset_mhz"),
"mem_offset_mhz": value if knob == "mem_offset_mhz" else baseline_cfg.get("mem_offset_mhz"),
"tokens_per_sec": tok_s, "temp_c": temp, "power_w": gpu.get("power_w"),
"stable": row["stable"], "instability": row["instability"],
"note": f"sweep {knob} {start}..{stop} step {step}",
})
state.current = {"knob": knob, "value": value, "stop": stop,
"tokens_per_sec": tok_s, "stable": row["stable"]}
logger.info(f"autotune {knob}={value}: {tok_s} tok/s, {temp}°C, "
f"stable={row['stable']} {row['instability'] or ''}")
if not row["stable"]:
logger.warning(f"autotune stopping climb at {knob}={value}: {row['instability']}")
break
stable = [r for r in results if r["stable"] and r["tokens_per_sec"] > 0]
best = max(stable, key=lambda r: r["tokens_per_sec"]) if stable else None
# Two different comparisons, and conflating them is how a sweep oversells itself.
# The first step of the range is not "baseline" unless it happens to be what the
# profile is already set to -- reporting "+102%" against the slowest value tried
# implies a speedup the user would not actually observe.
first_step = next((r for r in results if r["value"] == values[0]), None)
current = next((r for r in results if r["value"] == baseline_value), None)
gain_vs_first_step_pct = None
if best and first_step and first_step["tokens_per_sec"] > 0:
gain_vs_first_step_pct = round(
(best["tokens_per_sec"] / first_step["tokens_per_sec"] - 1) * 100, 2)
gain_vs_current_pct = None
if best and current and current["tokens_per_sec"] > 0:
gain_vs_current_pct = round(
(best["tokens_per_sec"] / current["tokens_per_sec"] - 1) * 100, 2)
applied = None
if apply_best and best:
overclock_manager.set_profile(profile, {knob: best["value"]})
applied = {knob: best["value"], "profile": profile}
logger.info(f"autotune wrote {knob}={best['value']} into profile '{profile}'")
result = {
"success": True,
"knob": knob,
"profile": profile,
"workload": workload,
"metric": metric,
"model": model,
"range": {"start": start, "stop": stop, "step": step, "values": values},
"effectiveness": effectiveness,
"steps_run": len(results),
"duration_s": round(time.time() - t_start, 1),
"cancelled": state.cancel,
"best": {k: best[k] for k in ("value", "tokens_per_sec", "temp_c", "clock_mem_mhz",
"clock_sm_mhz")} if best else None,
"current_profile_value": baseline_value,
"first_step_tokens_per_sec": first_step["tokens_per_sec"] if first_step else None,
"current_tokens_per_sec": current["tokens_per_sec"] if current else None,
"gain_vs_first_step_pct": gain_vs_first_step_pct,
"gain_vs_current_pct": gain_vs_current_pct,
"gain_note": ("compared against the profile's current setting"
if current is not None else
f"the profile's current value ({baseline_value}) was not in the "
f"swept range, so only the spread across tried values is shown"),
"applied_to_profile": applied,
"first_unstable": next(({"value": r["value"], "why": r["instability"]}
for r in results if not r["stable"]), None),
"table": [{k: r[k] for k in ("value", "tokens_per_sec", "temp_c", "power_w",
"clock_mem_mhz", "clock_sm_mhz", "stable",
"instability")} for r in results],
}
state.last_result = result
return result
finally:
# Always hand the card back exactly as we found it.
state.running = False
state.current = None
vram_arbitrator.arbitrator.resume_oc(profile)
try:
overclock_manager.apply_profile(profile)
logger.info(f"autotune restored profile '{profile}'")
except Exception as e:
logger.error(f"autotune failed to restore profile, forcing stock: {e}")
overclock_manager.restore_safe("autotune restore failed")
def get_status() -> Dict[str, Any]:
return {
"running": state.running,
"current": state.current,
"last_result": state.last_result,
"knobs": KNOBS,
"history": telemetry_store.autotune_history(100),
}
def cancel() -> Dict[str, Any]:
if not state.running:
return {"cancelled": False, "reason": "no sweep running"}
state.cancel = True
return {"cancelled": True}