Files
gpu-program-swapper/overclock_manager.py
drjones 2948b0b440 Detect fan-mode drift; document health checks, VRAM accounting and tests
The reconciler only compared the power limit, so a profile whose fan setting never
applied stayed wrong indefinitely. The headless X server that owns the GPU can be
starting when this unit does; in-process retries cover a short delay, but if X
arrives later nothing noticed that the fan mode had never been set. profile_drift()
now compares fan mode too, gated on fan control having worked at least once so the
check does not fire forever on a machine without it.

README documents the self-check endpoint, the ollama/comfy/desktop/unmanaged
bucketing and why unreclaimable VRAM is reported separately, the SSE trimming, and
a tests section listing the measured constants the suite pins.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 19:17:16 -07:00

568 lines
22 KiB
Python

"""
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
Lever hierarchy (what actually works on this box):
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
Profiles are application-specific:
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
- comfy : diffusion is compute bound -> lock core clock high + max power
- balanced: stock boost, power unlocked only
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
"""
import json
import logging
import os
import shutil
import subprocess
import time
from typing import Dict, Any, Optional, List
logger = logging.getLogger("overclock_manager")
_BASE = os.path.dirname(os.path.abspath(__file__))
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
NVIDIA_SMI = "nvidia-smi"
NVIDIA_SETTINGS = "nvidia-settings"
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
"ollama": {
"label": "Ollama — LLM decode (memory-bandwidth bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 11501,
"fan_mode": "manual",
"fan_speed_pct": 60,
},
"comfy": {
"label": "ComfyUI — diffusion (core-compute bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 2900,
"lock_core_max": 3105,
"lock_mem_mhz": 0,
"fan_mode": "manual",
"fan_speed_pct": 75,
},
"balanced": {
"label": "Balanced — stock boost, power unlocked",
"power_limit_w": 370,
"core_offset_mhz": 0,
"mem_offset_mhz": 0,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 0,
"fan_mode": "auto",
"fan_speed_pct": 0,
},
}
_OFFSETS_SUPPORTED: Optional[bool] = None
def offsets_supported(recheck: bool = False) -> bool:
"""Whether nvidia-settings clock offsets actually take effect on this driver.
Driver 595.84 accepts GPUGraphicsClockOffset/GPUMemoryTransferRateOffset and silently
discards them: assigning 0 returns success and the attribute still reads back its old
value. Profiles carrying core_offset_mhz/mem_offset_mhz were therefore configuring
nothing. Probed once and cached.
"""
global _OFFSETS_SUPPORTED
if _OFFSETS_SUPPORTED is not None and not recheck:
return _OFFSETS_SUPPORTED
def _read() -> Optional[int]:
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]")
for line in (q.get("out") or "").splitlines():
if "Attribute" in line and "):" in line:
try:
return int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
return None
before = _read()
if before is None:
_OFFSETS_SUPPORTED = False
return False
probe = before + 25
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={probe}")
after = _read()
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={before}")
_OFFSETS_SUPPORTED = (after is not None and after != before)
if not _OFFSETS_SUPPORTED:
logger.warning("Clock offsets are not honoured by this driver "
f"(set {probe}, read back {after}); profile offset fields are inert.")
return _OFFSETS_SUPPORTED
ACTIVE_PROFILE = "balanced"
_LAST_RESULT: Dict[str, Any] = {}
FAN_MANUAL = False
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
full = list(cmd)
if use_sudo:
full = ["sudo", "-n"] + full
try:
proc = subprocess.run(
full, capture_output=True, text=True, timeout=timeout
)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except subprocess.TimeoutExpired:
return {"rc": -1, "out": "", "err": "timeout"}
except FileNotFoundError as e:
return {"rc": -1, "out": "", "err": f"not found: {e}"}
def _smi(*args: str) -> Dict[str, Any]:
return _sh([NVIDIA_SMI, *args], use_sudo=True)
def _nvidia_settings(*args: str) -> Dict[str, Any]:
"""Run nvidia-settings against the headless X display that owns the GPU."""
env = os.environ.copy()
env["DISPLAY"] = HEADLESS_DISPLAY
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
try:
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except Exception as e:
return {"rc": -1, "out": "", "err": str(e)}
def load_profiles() -> Dict[str, Dict[str, Any]]:
"""Load profiles from disk, falling back to defaults and merging new keys."""
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
if os.path.exists(CONFIG_PATH):
try:
with open(CONFIG_PATH) as f:
stored = json.load(f)
for name, cfg in stored.items():
if name in profiles:
profiles[name].update(cfg)
else:
profiles[name] = cfg
except Exception as e:
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
return profiles
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
try:
with open(CONFIG_PATH, "w") as f:
json.dump(profiles, f, indent=2)
return True
except Exception as e:
logger.error(f"save_profiles failed: {e}")
return False
def get_profiles() -> Dict[str, Dict[str, Any]]:
return load_profiles()
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'"}
profiles[name].update(cfg)
ok = save_profiles(profiles)
return {"success": ok, "profiles": profiles if ok else None}
def _apply_power_limit(watts: int) -> Dict[str, Any]:
r = _smi("-pl", str(watts))
ok = r["rc"] == 0
return {"applied": ok, "detail": r.get("out") or r.get("err")}
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
if core_min == 0 and core_max == 0:
r = _smi("-rgc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lgc", f"{core_min},{core_max}")
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
if mem_mhz == 0:
r = _smi("-rmc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lmc", str(mem_mhz))
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
if core_mhz == 0 and mem_mhz == 0:
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
if r["rc"] != 0:
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
# Read back to confirm the driver actually persisted the offsets.
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
applied_core = applied_mem = None
for line in q["out"].splitlines():
line = line.strip()
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
try:
applied_core = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
try:
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
supported = (applied_core is not None and applied_core != 0) or \
(applied_mem is not None and applied_mem != 0)
return {
"applied": supported,
"supported": supported,
"readback_core": applied_core,
"readback_mem": applied_mem,
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
}
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""Apply a named overclock profile to the GPU. Returns a full result report.
`overrides` lets the thermal governor and the autotuner apply a modified version of a
profile (a derated offset, a probe clock) without mutating what is stored on disk.
"""
global ACTIVE_PROFILE, _LAST_RESULT
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
cfg = dict(profiles[name])
if overrides:
cfg.update(overrides)
fan_mode = cfg.get("fan_mode", "auto")
fan_speed = int(cfg.get("fan_speed_pct", 0))
result = {
"success": True,
"profile": name,
"label": cfg.get("label", name),
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
"offsets": (_apply_offsets(int(cfg.get("core_offset_mhz", 0)),
int(cfg.get("mem_offset_mhz", 0)))
if offsets_supported() else
{"applied": False, "supported": False,
"detail": "skipped: this driver accepts clock offsets and ignores them"}),
"fan": apply_fan_control(fan_mode, fan_speed),
}
# Invalidate BEFORE reading back. These caches exist so the dashboard's polling does
# not fork sudo every few seconds, but reading through them here reported the
# pre-change value: a profile that had just moved the card 370W -> 320W came back
# claiming 370W, so the API contradicted nvidia-smi.
_STATE_CACHE["value"] = None
_FAN_CACHE["value"] = None
result["gpu"] = get_gpu_state(force=True)
result["fan_status"] = get_fan_status(force=True)
result["overrides"] = overrides or {}
# Say plainly whether the card ended up where the profile asked it to.
intended_w = int(cfg.get("power_limit_w", 0))
actual_w = result["gpu"].get("power_limit_w")
result["verified"] = {
"power_limit_intended_w": intended_w,
"power_limit_actual_w": actual_w,
"power_limit_ok": (actual_w is None or intended_w == 0
or abs(float(actual_w) - intended_w) < 1.0),
"fan_mode_intended": fan_mode,
"fan_mode_actual": result["fan_status"].get("mode"),
"fan_ok": result["fan"].get("applied", False),
}
if not result["verified"]["power_limit_ok"]:
logger.warning(f"Profile '{name}' asked for {intended_w}W but the card reports "
f"{actual_w}W")
if not result["verified"]["fan_ok"]:
logger.warning(f"Profile '{name}' could not set fans: "
f"{result['fan'].get('detail')}")
global _APPLIED_ONCE, _FAN_AVAILABLE
if result["verified"]["power_limit_ok"]:
_APPLIED_ONCE = True
if result["verified"]["fan_ok"]:
_FAN_AVAILABLE = True
ACTIVE_PROFILE = name
_LAST_RESULT = result
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
return result
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
STATE_TTL_S = 2.0
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
"""Read back live GPU clocks/power/limits via nvidia-smi.
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
"""
import time as _time
if not force and _STATE_CACHE["value"] is not None and \
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
return _STATE_CACHE["value"]
state: Dict[str, Any] = {}
r = _smi(
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
"temperature.gpu,power.draw,fan.speed",
"--format=csv,noheader,nounits",
)
if r["rc"] == 0 and r["out"]:
parts = [p.strip() for p in r["out"].split(",")]
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
"temp_c", "power_draw_w", "fan_pct"]
for i, k in enumerate(keys):
if i < len(parts):
try:
state[k] = float(parts[i])
except ValueError:
state[k] = parts[i]
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
return state
def is_headless_x_running() -> bool:
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
return r["rc"] == 0
FAN_RETRY_ATTEMPTS = 5
FAN_RETRY_DELAY_S = 2.0
def _fan_target_missing(result: Dict[str, Any]) -> bool:
"""True when nvidia-settings could not see the GPU at all.
On boot this service can start before the headless X server on :8 that owns the GPU
is accepting connections, and the fan assignment fails with 'Error resolving target
specification'. Nothing retried, so the fans were simply never configured for the
whole session and the failure was only visible deep in a log line.
"""
text = ((result.get("err") or "") + (result.get("out") or "")).lower()
return ("error resolving target" in text or "no targets match" in text
or "cannot open display" in text)
def apply_fan_control(mode: str, speed_pct: int, _attempt: int = 0) -> Dict[str, Any]:
global FAN_MANUAL
if mode == "auto":
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if not ok and _fan_target_missing(r) and _attempt < FAN_RETRY_ATTEMPTS:
logger.info(f"Fan control target not ready (attempt {_attempt + 1}/"
f"{FAN_RETRY_ATTEMPTS}); X on {HEADLESS_DISPLAY} may still be "
f"starting — retrying in {FAN_RETRY_DELAY_S}s")
time.sleep(FAN_RETRY_DELAY_S)
return apply_fan_control(mode, speed_pct, _attempt + 1)
if ok:
FAN_MANUAL = False
_FAN_CACHE["value"] = None
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err"),
"attempts": _attempt + 1}
speed_pct = max(30, min(100, int(speed_pct)))
r = _nvidia_settings(
"-a", "[gpu:0]/GPUFanControlState=1",
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
)
ok = (r["rc"] == 0)
if ok:
FAN_MANUAL = True
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
def set_fan_speed(percent: int) -> Dict[str, Any]:
"""Set manual GPU fan target speed (30-100%)."""
return apply_fan_control(mode="manual", speed_pct=percent)
def set_fan_auto() -> Dict[str, Any]:
"""Return GPU fan to automatic control."""
global FAN_MANUAL
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if ok:
FAN_MANUAL = False
_FAN_CACHE["value"] = None
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
def get_fan_status(force: bool = False) -> Dict[str, Any]:
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
import time as _time
if not force and _FAN_CACHE["value"] is not None and \
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
return _FAN_CACHE["value"]
global FAN_MANUAL
target = None
manual = FAN_MANUAL
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
if r.get("rc") == 0 and r.get("out"):
for line in r["out"].splitlines():
if "GPUFanControlState" in line and ":" in line:
try:
val = int(line.split("):")[-1].split(".")[0].strip())
manual = (val == 1)
except Exception:
pass
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
try:
target = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
return result
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
This matters because every lever here is sticky. If the service dies while a profile is
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
manual PWM was last set. Nothing was undoing that.
"""
logger.warning(f"Restoring GPU to safe stock state ({reason})")
result = {
"reason": reason,
"clock_lock": _apply_clock_lock(0, 0),
"mem_lock": _apply_mem_lock(0),
"offsets": _apply_offsets(0, 0),
"fan": set_fan_auto(),
}
# Hand the power limit back to the card's own default rather than assuming 370 W.
state = get_gpu_state()
default_w = state.get("power_default_w")
if isinstance(default_w, (int, float)) and default_w > 0:
result["power_limit"] = _apply_power_limit(int(default_w))
global ACTIVE_PROFILE
ACTIVE_PROFILE = "stock"
return result
_APPLIED_ONCE = False
# Set once fan control has worked at least once, so drift checks do not fire forever on
# a machine that simply has no fan control available.
_FAN_AVAILABLE = False
def profile_drift() -> Dict[str, Any]:
"""Compare what the active profile asks for against what the card actually reports.
ACTIVE_PROFILE defaults to "balanced" at import, which is indistinguishable from
"balanced was successfully applied" -- so a startup apply that failed left the app
confidently reporting a profile it had never put on the hardware. This makes the
difference visible instead.
"""
profiles = load_profiles()
cfg = profiles.get(ACTIVE_PROFILE, {})
state = get_gpu_state()
intended = int(cfg.get("power_limit_w", 0) or 0)
actual = state.get("power_limit_w")
power_drift = bool(intended and actual is not None
and abs(float(actual) - intended) >= 1.0)
# Fan mode is checked too. The headless X server that owns the GPU can still be
# starting when this unit does, and the fan assignment then fails; the in-process
# retries cover a short delay, but if X arrives later nothing else would ever notice
# that the profile's fan setting was never applied.
fan_intended = cfg.get("fan_mode", "auto")
fan_actual = None
fan_drift = False
if _FAN_AVAILABLE:
fan_actual = get_fan_status().get("mode")
fan_drift = bool(fan_actual and fan_actual != fan_intended)
drifted = power_drift or fan_drift or not _APPLIED_ONCE
reasons = []
if not _APPLIED_ONCE:
reasons.append("no profile has been successfully applied since startup")
if power_drift:
reasons.append(f"card reports {actual}W, profile asks {intended}W")
if fan_drift:
reasons.append(f"fans are {fan_actual}, profile asks {fan_intended}")
return {
"profile": ACTIVE_PROFILE,
"applied_since_start": _APPLIED_ONCE,
"power_limit_intended_w": intended,
"power_limit_actual_w": actual,
"fan_mode_intended": fan_intended,
"fan_mode_actual": fan_actual,
"fan_available": _FAN_AVAILABLE,
"drifted": drifted,
"reason": "; ".join(reasons) or None,
}
def reconcile_profile() -> Dict[str, Any]:
"""Re-apply the active profile if the hardware has drifted away from it."""
drift = profile_drift()
if not drift["drifted"]:
return {"reconciled": False, "drift": drift}
logger.warning(f"Overclock drift detected — re-applying '{ACTIVE_PROFILE}': "
f"{drift['reason']}")
res = apply_profile(ACTIVE_PROFILE)
return {"reconciled": True, "drift": drift, "result": res.get("verified")}
def get_status() -> Dict[str, Any]:
"""Full overclock status for the dashboard."""
return {
"active_profile": ACTIVE_PROFILE,
"drift": profile_drift(),
"offsets_supported": offsets_supported(),
"effective_levers": (["power_limit", "clock_lock", "mem_lock", "fan"]
+ (["offsets"] if offsets_supported() else [])),
"profiles": load_profiles(),
"gpu": get_gpu_state(),
"fan": get_fan_status(),
"headless_x_running": is_headless_x_running(),
"headless_display": HEADLESS_DISPLAY,
"last_result": _LAST_RESULT,
}
if __name__ == "__main__":
import sys
logging.basicConfig(level=logging.INFO)
if len(sys.argv) > 1:
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
else:
print(json.dumps(get_status(), indent=2, default=str))