Nine changes, in rough order of how much they affect real behaviour: 1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload; measured here, the HTTP call returns in 63ms while the driver takes a further 77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up allocating into VRAM that is still occupied. instant_free_ollama_vram() polls NVML until the allocation is actually gone and reports request/confirm split. 2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full checkpoint reload on each workflow iteration. It is held for 30s of genuinely empty queue, with an immediate purge when Ollama actually asks for the memory. 3. Cache-hit classification uses achieved bandwidth (size / load duration) rather than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit. 4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB resident on a box with 46GB of page cache: the kernel only permits page-cache introspection on files you own, and the Ollama blobs are owned by uid ollama, for which mincore answers "all resident" instead of failing. Uses cachestat(2) where permitted and a randomised read-rate probe elsewhere, labelling which was used. Fixed-offset probing was self-fulfilling, so windows are random and cold ones are returned with FADV_DONTNEED. 5. Warming is budgeted and ranked by recency/frequency instead of reading every file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first. 6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a 50-entry in-memory deque, so /api/analytics/profiles can finally answer whether an overclock profile actually delivers more tok/s. 7. Thermal governor walks the overclock back on sustained heat or hardware throttling, with hysteresis, fed from the existing sampler. 8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid errors and degenerate output, and restores the profile in a finally block. 9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost. Nothing previously undid a locked clock or a manually pinned fan. Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than every client re-running the whole snapshot; wall-clock timestamps in place of the event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
401 lines
15 KiB
Python
401 lines
15 KiB
Python
"""
|
|
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
|
|
|
|
Lever hierarchy (what actually works on this box):
|
|
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
|
|
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
|
|
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
|
|
|
|
Profiles are application-specific:
|
|
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
|
|
- comfy : diffusion is compute bound -> lock core clock high + max power
|
|
- balanced: stock boost, power unlocked only
|
|
|
|
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
|
|
"""
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
from typing import Dict, Any, Optional, List
|
|
|
|
logger = logging.getLogger("overclock_manager")
|
|
|
|
_BASE = os.path.dirname(os.path.abspath(__file__))
|
|
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
|
|
|
|
NVIDIA_SMI = "nvidia-smi"
|
|
NVIDIA_SETTINGS = "nvidia-settings"
|
|
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
|
|
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
|
|
|
|
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
|
|
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
|
|
"ollama": {
|
|
"label": "Ollama — LLM decode (memory-bandwidth bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 11501,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 60,
|
|
},
|
|
"comfy": {
|
|
"label": "ComfyUI — diffusion (core-compute bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 2900,
|
|
"lock_core_max": 3105,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 75,
|
|
},
|
|
"balanced": {
|
|
"label": "Balanced — stock boost, power unlocked",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 0,
|
|
"mem_offset_mhz": 0,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "auto",
|
|
"fan_speed_pct": 0,
|
|
},
|
|
}
|
|
|
|
ACTIVE_PROFILE = "balanced"
|
|
_LAST_RESULT: Dict[str, Any] = {}
|
|
FAN_MANUAL = False
|
|
|
|
|
|
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
|
|
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
|
|
full = list(cmd)
|
|
if use_sudo:
|
|
full = ["sudo", "-n"] + full
|
|
try:
|
|
proc = subprocess.run(
|
|
full, capture_output=True, text=True, timeout=timeout
|
|
)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except subprocess.TimeoutExpired:
|
|
return {"rc": -1, "out": "", "err": "timeout"}
|
|
except FileNotFoundError as e:
|
|
return {"rc": -1, "out": "", "err": f"not found: {e}"}
|
|
|
|
|
|
def _smi(*args: str) -> Dict[str, Any]:
|
|
return _sh([NVIDIA_SMI, *args], use_sudo=True)
|
|
|
|
|
|
def _nvidia_settings(*args: str) -> Dict[str, Any]:
|
|
"""Run nvidia-settings against the headless X display that owns the GPU."""
|
|
env = os.environ.copy()
|
|
env["DISPLAY"] = HEADLESS_DISPLAY
|
|
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
|
|
try:
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except Exception as e:
|
|
return {"rc": -1, "out": "", "err": str(e)}
|
|
|
|
|
|
def load_profiles() -> Dict[str, Dict[str, Any]]:
|
|
"""Load profiles from disk, falling back to defaults and merging new keys."""
|
|
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
|
|
if os.path.exists(CONFIG_PATH):
|
|
try:
|
|
with open(CONFIG_PATH) as f:
|
|
stored = json.load(f)
|
|
for name, cfg in stored.items():
|
|
if name in profiles:
|
|
profiles[name].update(cfg)
|
|
else:
|
|
profiles[name] = cfg
|
|
except Exception as e:
|
|
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
|
|
return profiles
|
|
|
|
|
|
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
|
|
try:
|
|
with open(CONFIG_PATH, "w") as f:
|
|
json.dump(profiles, f, indent=2)
|
|
return True
|
|
except Exception as e:
|
|
logger.error(f"save_profiles failed: {e}")
|
|
return False
|
|
|
|
|
|
def get_profiles() -> Dict[str, Dict[str, Any]]:
|
|
return load_profiles()
|
|
|
|
|
|
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'"}
|
|
profiles[name].update(cfg)
|
|
ok = save_profiles(profiles)
|
|
return {"success": ok, "profiles": profiles if ok else None}
|
|
|
|
|
|
def _apply_power_limit(watts: int) -> Dict[str, Any]:
|
|
r = _smi("-pl", str(watts))
|
|
ok = r["rc"] == 0
|
|
return {"applied": ok, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
|
|
if core_min == 0 and core_max == 0:
|
|
r = _smi("-rgc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lgc", f"{core_min},{core_max}")
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
|
|
if mem_mhz == 0:
|
|
r = _smi("-rmc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lmc", str(mem_mhz))
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
|
|
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
|
|
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
|
|
if core_mhz == 0 and mem_mhz == 0:
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
|
|
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
|
|
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
|
|
|
|
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
|
|
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
|
|
if r["rc"] != 0:
|
|
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
|
|
|
|
# Read back to confirm the driver actually persisted the offsets.
|
|
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
|
|
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
|
|
applied_core = applied_mem = None
|
|
for line in q["out"].splitlines():
|
|
line = line.strip()
|
|
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_core = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
|
|
supported = (applied_core is not None and applied_core != 0) or \
|
|
(applied_mem is not None and applied_mem != 0)
|
|
return {
|
|
"applied": supported,
|
|
"supported": supported,
|
|
"readback_core": applied_core,
|
|
"readback_mem": applied_mem,
|
|
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
|
|
}
|
|
|
|
|
|
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
"""Apply a named overclock profile to the GPU. Returns a full result report.
|
|
|
|
`overrides` lets the thermal governor and the autotuner apply a modified version of a
|
|
profile (a derated offset, a probe clock) without mutating what is stored on disk.
|
|
"""
|
|
global ACTIVE_PROFILE, _LAST_RESULT
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
|
|
|
|
cfg = dict(profiles[name])
|
|
if overrides:
|
|
cfg.update(overrides)
|
|
fan_mode = cfg.get("fan_mode", "auto")
|
|
fan_speed = int(cfg.get("fan_speed_pct", 0))
|
|
|
|
result = {
|
|
"success": True,
|
|
"profile": name,
|
|
"label": cfg.get("label", name),
|
|
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
|
|
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
|
|
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
|
|
"offsets": _apply_offsets(int(cfg.get("core_offset_mhz", 0)), int(cfg.get("mem_offset_mhz", 0))),
|
|
"fan": apply_fan_control(fan_mode, fan_speed),
|
|
}
|
|
result["gpu"] = get_gpu_state()
|
|
result["fan_status"] = get_fan_status()
|
|
result["overrides"] = overrides or {}
|
|
|
|
_STATE_CACHE["value"] = None
|
|
_FAN_CACHE["value"] = None
|
|
ACTIVE_PROFILE = name
|
|
_LAST_RESULT = result
|
|
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
|
|
return result
|
|
|
|
|
|
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
STATE_TTL_S = 2.0
|
|
|
|
|
|
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
|
|
"""Read back live GPU clocks/power/limits via nvidia-smi.
|
|
|
|
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
|
|
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
|
|
"""
|
|
import time as _time
|
|
if not force and _STATE_CACHE["value"] is not None and \
|
|
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
|
|
return _STATE_CACHE["value"]
|
|
state: Dict[str, Any] = {}
|
|
r = _smi(
|
|
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
|
|
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
|
|
"temperature.gpu,power.draw,fan.speed",
|
|
"--format=csv,noheader,nounits",
|
|
)
|
|
if r["rc"] == 0 and r["out"]:
|
|
parts = [p.strip() for p in r["out"].split(",")]
|
|
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
|
|
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
|
|
"temp_c", "power_draw_w", "fan_pct"]
|
|
for i, k in enumerate(keys):
|
|
if i < len(parts):
|
|
try:
|
|
state[k] = float(parts[i])
|
|
except ValueError:
|
|
state[k] = parts[i]
|
|
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
|
|
return state
|
|
|
|
|
|
def is_headless_x_running() -> bool:
|
|
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
|
|
return r["rc"] == 0
|
|
|
|
|
|
def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]:
|
|
global FAN_MANUAL
|
|
if mode == "auto":
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")}
|
|
|
|
speed_pct = max(30, min(100, int(speed_pct)))
|
|
r = _nvidia_settings(
|
|
"-a", "[gpu:0]/GPUFanControlState=1",
|
|
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
|
|
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
|
|
)
|
|
ok = (r["rc"] == 0)
|
|
if ok:
|
|
FAN_MANUAL = True
|
|
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def set_fan_speed(percent: int) -> Dict[str, Any]:
|
|
"""Set manual GPU fan target speed (30-100%)."""
|
|
return apply_fan_control(mode="manual", speed_pct=percent)
|
|
|
|
|
|
def set_fan_auto() -> Dict[str, Any]:
|
|
"""Return GPU fan to automatic control."""
|
|
global FAN_MANUAL
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
_FAN_CACHE["value"] = None
|
|
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def get_fan_status(force: bool = False) -> Dict[str, Any]:
|
|
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
|
|
import time as _time
|
|
if not force and _FAN_CACHE["value"] is not None and \
|
|
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
|
|
return _FAN_CACHE["value"]
|
|
global FAN_MANUAL
|
|
target = None
|
|
manual = FAN_MANUAL
|
|
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
|
|
if r.get("rc") == 0 and r.get("out"):
|
|
for line in r["out"].splitlines():
|
|
if "GPUFanControlState" in line and ":" in line:
|
|
try:
|
|
val = int(line.split("):")[-1].split(".")[0].strip())
|
|
manual = (val == 1)
|
|
except Exception:
|
|
pass
|
|
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
|
|
try:
|
|
target = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
|
|
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
|
|
return result
|
|
|
|
|
|
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
|
|
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
|
|
|
|
This matters because every lever here is sticky. If the service dies while a profile is
|
|
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
|
|
manual PWM was last set. Nothing was undoing that.
|
|
"""
|
|
logger.warning(f"Restoring GPU to safe stock state ({reason})")
|
|
result = {
|
|
"reason": reason,
|
|
"clock_lock": _apply_clock_lock(0, 0),
|
|
"mem_lock": _apply_mem_lock(0),
|
|
"offsets": _apply_offsets(0, 0),
|
|
"fan": set_fan_auto(),
|
|
}
|
|
# Hand the power limit back to the card's own default rather than assuming 370 W.
|
|
state = get_gpu_state()
|
|
default_w = state.get("power_default_w")
|
|
if isinstance(default_w, (int, float)) and default_w > 0:
|
|
result["power_limit"] = _apply_power_limit(int(default_w))
|
|
global ACTIVE_PROFILE
|
|
ACTIVE_PROFILE = "stock"
|
|
return result
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
"""Full overclock status for the dashboard."""
|
|
return {
|
|
"active_profile": ACTIVE_PROFILE,
|
|
"profiles": load_profiles(),
|
|
"gpu": get_gpu_state(),
|
|
"fan": get_fan_status(),
|
|
"headless_x_running": is_headless_x_running(),
|
|
"headless_display": HEADLESS_DISPLAY,
|
|
"last_result": _LAST_RESULT,
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
logging.basicConfig(level=logging.INFO)
|
|
if len(sys.argv) > 1:
|
|
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
|
|
else:
|
|
print(json.dumps(get_status(), indent=2, default=str))
|