Files
gpu-program-swapper/overclock_manager.py
drjones 5431144b2e Add barrier-confirmed yielding, measured residency, persistence and closed-loop tuning
Nine changes, in rough order of how much they affect real behaviour:

1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload;
   measured here, the HTTP call returns in 63ms while the driver takes a further
   77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up
   allocating into VRAM that is still occupied. instant_free_ollama_vram() polls
   NVML until the allocation is actually gone and reports request/confirm split.

2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full
   checkpoint reload on each workflow iteration. It is held for 30s of genuinely
   empty queue, with an immediate purge when Ollama actually asks for the memory.

3. Cache-hit classification uses achieved bandwidth (size / load duration) rather
   than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read
   at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit.

4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB
   resident on a box with 46GB of page cache: the kernel only permits page-cache
   introspection on files you own, and the Ollama blobs are owned by uid ollama,
   for which mincore answers "all resident" instead of failing. Uses cachestat(2)
   where permitted and a randomised read-rate probe elsewhere, labelling which was
   used. Fixed-offset probing was self-fulfilling, so windows are random and cold
   ones are returned with FADV_DONTNEED.

5. Warming is budgeted and ranked by recency/frequency instead of reading every
   file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first.

6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a
   50-entry in-memory deque, so /api/analytics/profiles can finally answer whether
   an overclock profile actually delivers more tok/s.

7. Thermal governor walks the overclock back on sustained heat or hardware
   throttling, with hysteresis, fed from the existing sampler.

8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid
   errors and degenerate output, and restores the profile in a finally block.

9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost.
   Nothing previously undid a locked clock or a manually pinned fan.

Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than
every client re-running the whole snapshot; wall-clock timestamps in place of the
event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-28 08:57:35 -07:00

401 lines
15 KiB
Python

"""
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
Lever hierarchy (what actually works on this box):
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
Profiles are application-specific:
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
- comfy : diffusion is compute bound -> lock core clock high + max power
- balanced: stock boost, power unlocked only
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
"""
import json
import logging
import os
import shutil
import subprocess
from typing import Dict, Any, Optional, List
logger = logging.getLogger("overclock_manager")
_BASE = os.path.dirname(os.path.abspath(__file__))
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
NVIDIA_SMI = "nvidia-smi"
NVIDIA_SETTINGS = "nvidia-settings"
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
"ollama": {
"label": "Ollama — LLM decode (memory-bandwidth bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 11501,
"fan_mode": "manual",
"fan_speed_pct": 60,
},
"comfy": {
"label": "ComfyUI — diffusion (core-compute bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 2900,
"lock_core_max": 3105,
"lock_mem_mhz": 0,
"fan_mode": "manual",
"fan_speed_pct": 75,
},
"balanced": {
"label": "Balanced — stock boost, power unlocked",
"power_limit_w": 370,
"core_offset_mhz": 0,
"mem_offset_mhz": 0,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 0,
"fan_mode": "auto",
"fan_speed_pct": 0,
},
}
ACTIVE_PROFILE = "balanced"
_LAST_RESULT: Dict[str, Any] = {}
FAN_MANUAL = False
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
full = list(cmd)
if use_sudo:
full = ["sudo", "-n"] + full
try:
proc = subprocess.run(
full, capture_output=True, text=True, timeout=timeout
)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except subprocess.TimeoutExpired:
return {"rc": -1, "out": "", "err": "timeout"}
except FileNotFoundError as e:
return {"rc": -1, "out": "", "err": f"not found: {e}"}
def _smi(*args: str) -> Dict[str, Any]:
return _sh([NVIDIA_SMI, *args], use_sudo=True)
def _nvidia_settings(*args: str) -> Dict[str, Any]:
"""Run nvidia-settings against the headless X display that owns the GPU."""
env = os.environ.copy()
env["DISPLAY"] = HEADLESS_DISPLAY
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
try:
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except Exception as e:
return {"rc": -1, "out": "", "err": str(e)}
def load_profiles() -> Dict[str, Dict[str, Any]]:
"""Load profiles from disk, falling back to defaults and merging new keys."""
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
if os.path.exists(CONFIG_PATH):
try:
with open(CONFIG_PATH) as f:
stored = json.load(f)
for name, cfg in stored.items():
if name in profiles:
profiles[name].update(cfg)
else:
profiles[name] = cfg
except Exception as e:
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
return profiles
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
try:
with open(CONFIG_PATH, "w") as f:
json.dump(profiles, f, indent=2)
return True
except Exception as e:
logger.error(f"save_profiles failed: {e}")
return False
def get_profiles() -> Dict[str, Dict[str, Any]]:
return load_profiles()
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'"}
profiles[name].update(cfg)
ok = save_profiles(profiles)
return {"success": ok, "profiles": profiles if ok else None}
def _apply_power_limit(watts: int) -> Dict[str, Any]:
r = _smi("-pl", str(watts))
ok = r["rc"] == 0
return {"applied": ok, "detail": r.get("out") or r.get("err")}
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
if core_min == 0 and core_max == 0:
r = _smi("-rgc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lgc", f"{core_min},{core_max}")
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
if mem_mhz == 0:
r = _smi("-rmc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lmc", str(mem_mhz))
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
if core_mhz == 0 and mem_mhz == 0:
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
if r["rc"] != 0:
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
# Read back to confirm the driver actually persisted the offsets.
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
applied_core = applied_mem = None
for line in q["out"].splitlines():
line = line.strip()
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
try:
applied_core = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
try:
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
supported = (applied_core is not None and applied_core != 0) or \
(applied_mem is not None and applied_mem != 0)
return {
"applied": supported,
"supported": supported,
"readback_core": applied_core,
"readback_mem": applied_mem,
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
}
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""Apply a named overclock profile to the GPU. Returns a full result report.
`overrides` lets the thermal governor and the autotuner apply a modified version of a
profile (a derated offset, a probe clock) without mutating what is stored on disk.
"""
global ACTIVE_PROFILE, _LAST_RESULT
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
cfg = dict(profiles[name])
if overrides:
cfg.update(overrides)
fan_mode = cfg.get("fan_mode", "auto")
fan_speed = int(cfg.get("fan_speed_pct", 0))
result = {
"success": True,
"profile": name,
"label": cfg.get("label", name),
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
"offsets": _apply_offsets(int(cfg.get("core_offset_mhz", 0)), int(cfg.get("mem_offset_mhz", 0))),
"fan": apply_fan_control(fan_mode, fan_speed),
}
result["gpu"] = get_gpu_state()
result["fan_status"] = get_fan_status()
result["overrides"] = overrides or {}
_STATE_CACHE["value"] = None
_FAN_CACHE["value"] = None
ACTIVE_PROFILE = name
_LAST_RESULT = result
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
return result
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
STATE_TTL_S = 2.0
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
"""Read back live GPU clocks/power/limits via nvidia-smi.
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
"""
import time as _time
if not force and _STATE_CACHE["value"] is not None and \
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
return _STATE_CACHE["value"]
state: Dict[str, Any] = {}
r = _smi(
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
"temperature.gpu,power.draw,fan.speed",
"--format=csv,noheader,nounits",
)
if r["rc"] == 0 and r["out"]:
parts = [p.strip() for p in r["out"].split(",")]
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
"temp_c", "power_draw_w", "fan_pct"]
for i, k in enumerate(keys):
if i < len(parts):
try:
state[k] = float(parts[i])
except ValueError:
state[k] = parts[i]
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
return state
def is_headless_x_running() -> bool:
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
return r["rc"] == 0
def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]:
global FAN_MANUAL
if mode == "auto":
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if ok:
FAN_MANUAL = False
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")}
speed_pct = max(30, min(100, int(speed_pct)))
r = _nvidia_settings(
"-a", "[gpu:0]/GPUFanControlState=1",
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
)
ok = (r["rc"] == 0)
if ok:
FAN_MANUAL = True
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
def set_fan_speed(percent: int) -> Dict[str, Any]:
"""Set manual GPU fan target speed (30-100%)."""
return apply_fan_control(mode="manual", speed_pct=percent)
def set_fan_auto() -> Dict[str, Any]:
"""Return GPU fan to automatic control."""
global FAN_MANUAL
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if ok:
FAN_MANUAL = False
_FAN_CACHE["value"] = None
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
def get_fan_status(force: bool = False) -> Dict[str, Any]:
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
import time as _time
if not force and _FAN_CACHE["value"] is not None and \
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
return _FAN_CACHE["value"]
global FAN_MANUAL
target = None
manual = FAN_MANUAL
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
if r.get("rc") == 0 and r.get("out"):
for line in r["out"].splitlines():
if "GPUFanControlState" in line and ":" in line:
try:
val = int(line.split("):")[-1].split(".")[0].strip())
manual = (val == 1)
except Exception:
pass
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
try:
target = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
return result
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
This matters because every lever here is sticky. If the service dies while a profile is
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
manual PWM was last set. Nothing was undoing that.
"""
logger.warning(f"Restoring GPU to safe stock state ({reason})")
result = {
"reason": reason,
"clock_lock": _apply_clock_lock(0, 0),
"mem_lock": _apply_mem_lock(0),
"offsets": _apply_offsets(0, 0),
"fan": set_fan_auto(),
}
# Hand the power limit back to the card's own default rather than assuming 370 W.
state = get_gpu_state()
default_w = state.get("power_default_w")
if isinstance(default_w, (int, float)) and default_w > 0:
result["power_limit"] = _apply_power_limit(int(default_w))
global ACTIVE_PROFILE
ACTIVE_PROFILE = "stock"
return result
def get_status() -> Dict[str, Any]:
"""Full overclock status for the dashboard."""
return {
"active_profile": ACTIVE_PROFILE,
"profiles": load_profiles(),
"gpu": get_gpu_state(),
"fan": get_fan_status(),
"headless_x_running": is_headless_x_running(),
"headless_display": HEADLESS_DISPLAY,
"last_result": _LAST_RESULT,
}
if __name__ == "__main__":
import sys
logging.basicConfig(level=logging.INFO)
if len(sys.argv) > 1:
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
else:
print(json.dumps(get_status(), indent=2, default=str))