The profiles were hand-written and had never been checked against the hardware. Adding a ComfyUI benchmark alongside the existing decode one made the compute side measurable for the first time, and most of what the profiles configured turned out to do nothing. Measured on this card (RTX 4080 SUPER, driver 595.84): - LLM decode is not power-bound: 73.0-73.5 tok/s flat from 222W to 370W, with the card never drawing more than 224W at any limit. The ollama profile's 370W did nothing. - Diffusion is power-bound: 5.48 it/s @222W rising to 6.71 @370W, so comfy's 370W is worth a real +2.8% over the 320W stock default. - Clock locks did nothing for either workload: 72.6 tok/s locked at 11251MHz vs 72.7 unlocked; 6.77 it/s locked at 3105MHz vs 6.73 unlocked, and 6.78 at 2400MHz. - Memory bandwidth is still the decode bottleneck (5001MHz halves throughput to 35.9 tok/s), confirming the profile's premise -- the card just gets there unaided. - Fans: 48,435 samples show 81C all-time max and zero thermal throttle events, while the ollama profile held 49.6C average by running fans at 87%. All profiles now use automatic fans and let the thermal governor escalate on demand. Code changes supporting that: - _diffusion_benchmark() queues a fixed SDXL graph via ComfyUI's API. The seed must vary per run: ComfyUI caches by node inputs, so a fixed seed returned in ~1ms without executing. Implausibly fast results are now rejected as cache hits rather than recorded as record scores. - The arbitrator's automatic profile switching is suspended during a sweep. A diffusion benchmark trips trigger_comfy_priority, which reapplies the whole profile and would silently overwrite the clock being measured. - _supported_clocks() queries the mem,gr pair; asking for a single field returned one column and reading index 1 yielded an empty list rather than an error. Graphics clocks are subsampled (the card enumerates 194 of them) and lock sweeps include an explicit unlocked control step. - offsets_supported() probes once and apply_profile skips inert offset levers with an explanation instead of pretending they applied. - Profiles carry a 'measured' field recording the evidence behind each setting. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
448 lines
17 KiB
Python
448 lines
17 KiB
Python
"""
|
|
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
|
|
|
|
Lever hierarchy (what actually works on this box):
|
|
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
|
|
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
|
|
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
|
|
|
|
Profiles are application-specific:
|
|
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
|
|
- comfy : diffusion is compute bound -> lock core clock high + max power
|
|
- balanced: stock boost, power unlocked only
|
|
|
|
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
|
|
"""
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
from typing import Dict, Any, Optional, List
|
|
|
|
logger = logging.getLogger("overclock_manager")
|
|
|
|
_BASE = os.path.dirname(os.path.abspath(__file__))
|
|
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
|
|
|
|
NVIDIA_SMI = "nvidia-smi"
|
|
NVIDIA_SETTINGS = "nvidia-settings"
|
|
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
|
|
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
|
|
|
|
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
|
|
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
|
|
"ollama": {
|
|
"label": "Ollama — LLM decode (memory-bandwidth bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 11501,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 60,
|
|
},
|
|
"comfy": {
|
|
"label": "ComfyUI — diffusion (core-compute bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 2900,
|
|
"lock_core_max": 3105,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 75,
|
|
},
|
|
"balanced": {
|
|
"label": "Balanced — stock boost, power unlocked",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 0,
|
|
"mem_offset_mhz": 0,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "auto",
|
|
"fan_speed_pct": 0,
|
|
},
|
|
}
|
|
|
|
_OFFSETS_SUPPORTED: Optional[bool] = None
|
|
|
|
|
|
def offsets_supported(recheck: bool = False) -> bool:
|
|
"""Whether nvidia-settings clock offsets actually take effect on this driver.
|
|
|
|
Driver 595.84 accepts GPUGraphicsClockOffset/GPUMemoryTransferRateOffset and silently
|
|
discards them: assigning 0 returns success and the attribute still reads back its old
|
|
value. Profiles carrying core_offset_mhz/mem_offset_mhz were therefore configuring
|
|
nothing. Probed once and cached.
|
|
"""
|
|
global _OFFSETS_SUPPORTED
|
|
if _OFFSETS_SUPPORTED is not None and not recheck:
|
|
return _OFFSETS_SUPPORTED
|
|
|
|
def _read() -> Optional[int]:
|
|
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]")
|
|
for line in (q.get("out") or "").splitlines():
|
|
if "Attribute" in line and "):" in line:
|
|
try:
|
|
return int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
before = _read()
|
|
if before is None:
|
|
_OFFSETS_SUPPORTED = False
|
|
return False
|
|
probe = before + 25
|
|
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={probe}")
|
|
after = _read()
|
|
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={before}")
|
|
_OFFSETS_SUPPORTED = (after is not None and after != before)
|
|
if not _OFFSETS_SUPPORTED:
|
|
logger.warning("Clock offsets are not honoured by this driver "
|
|
f"(set {probe}, read back {after}); profile offset fields are inert.")
|
|
return _OFFSETS_SUPPORTED
|
|
|
|
|
|
ACTIVE_PROFILE = "balanced"
|
|
_LAST_RESULT: Dict[str, Any] = {}
|
|
FAN_MANUAL = False
|
|
|
|
|
|
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
|
|
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
|
|
full = list(cmd)
|
|
if use_sudo:
|
|
full = ["sudo", "-n"] + full
|
|
try:
|
|
proc = subprocess.run(
|
|
full, capture_output=True, text=True, timeout=timeout
|
|
)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except subprocess.TimeoutExpired:
|
|
return {"rc": -1, "out": "", "err": "timeout"}
|
|
except FileNotFoundError as e:
|
|
return {"rc": -1, "out": "", "err": f"not found: {e}"}
|
|
|
|
|
|
def _smi(*args: str) -> Dict[str, Any]:
|
|
return _sh([NVIDIA_SMI, *args], use_sudo=True)
|
|
|
|
|
|
def _nvidia_settings(*args: str) -> Dict[str, Any]:
|
|
"""Run nvidia-settings against the headless X display that owns the GPU."""
|
|
env = os.environ.copy()
|
|
env["DISPLAY"] = HEADLESS_DISPLAY
|
|
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
|
|
try:
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except Exception as e:
|
|
return {"rc": -1, "out": "", "err": str(e)}
|
|
|
|
|
|
def load_profiles() -> Dict[str, Dict[str, Any]]:
|
|
"""Load profiles from disk, falling back to defaults and merging new keys."""
|
|
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
|
|
if os.path.exists(CONFIG_PATH):
|
|
try:
|
|
with open(CONFIG_PATH) as f:
|
|
stored = json.load(f)
|
|
for name, cfg in stored.items():
|
|
if name in profiles:
|
|
profiles[name].update(cfg)
|
|
else:
|
|
profiles[name] = cfg
|
|
except Exception as e:
|
|
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
|
|
return profiles
|
|
|
|
|
|
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
|
|
try:
|
|
with open(CONFIG_PATH, "w") as f:
|
|
json.dump(profiles, f, indent=2)
|
|
return True
|
|
except Exception as e:
|
|
logger.error(f"save_profiles failed: {e}")
|
|
return False
|
|
|
|
|
|
def get_profiles() -> Dict[str, Dict[str, Any]]:
|
|
return load_profiles()
|
|
|
|
|
|
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'"}
|
|
profiles[name].update(cfg)
|
|
ok = save_profiles(profiles)
|
|
return {"success": ok, "profiles": profiles if ok else None}
|
|
|
|
|
|
def _apply_power_limit(watts: int) -> Dict[str, Any]:
|
|
r = _smi("-pl", str(watts))
|
|
ok = r["rc"] == 0
|
|
return {"applied": ok, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
|
|
if core_min == 0 and core_max == 0:
|
|
r = _smi("-rgc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lgc", f"{core_min},{core_max}")
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
|
|
if mem_mhz == 0:
|
|
r = _smi("-rmc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lmc", str(mem_mhz))
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
|
|
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
|
|
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
|
|
if core_mhz == 0 and mem_mhz == 0:
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
|
|
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
|
|
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
|
|
|
|
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
|
|
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
|
|
if r["rc"] != 0:
|
|
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
|
|
|
|
# Read back to confirm the driver actually persisted the offsets.
|
|
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
|
|
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
|
|
applied_core = applied_mem = None
|
|
for line in q["out"].splitlines():
|
|
line = line.strip()
|
|
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_core = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
|
|
supported = (applied_core is not None and applied_core != 0) or \
|
|
(applied_mem is not None and applied_mem != 0)
|
|
return {
|
|
"applied": supported,
|
|
"supported": supported,
|
|
"readback_core": applied_core,
|
|
"readback_mem": applied_mem,
|
|
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
|
|
}
|
|
|
|
|
|
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
"""Apply a named overclock profile to the GPU. Returns a full result report.
|
|
|
|
`overrides` lets the thermal governor and the autotuner apply a modified version of a
|
|
profile (a derated offset, a probe clock) without mutating what is stored on disk.
|
|
"""
|
|
global ACTIVE_PROFILE, _LAST_RESULT
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
|
|
|
|
cfg = dict(profiles[name])
|
|
if overrides:
|
|
cfg.update(overrides)
|
|
fan_mode = cfg.get("fan_mode", "auto")
|
|
fan_speed = int(cfg.get("fan_speed_pct", 0))
|
|
|
|
result = {
|
|
"success": True,
|
|
"profile": name,
|
|
"label": cfg.get("label", name),
|
|
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
|
|
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
|
|
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
|
|
"offsets": (_apply_offsets(int(cfg.get("core_offset_mhz", 0)),
|
|
int(cfg.get("mem_offset_mhz", 0)))
|
|
if offsets_supported() else
|
|
{"applied": False, "supported": False,
|
|
"detail": "skipped: this driver accepts clock offsets and ignores them"}),
|
|
"fan": apply_fan_control(fan_mode, fan_speed),
|
|
}
|
|
result["gpu"] = get_gpu_state()
|
|
result["fan_status"] = get_fan_status()
|
|
result["overrides"] = overrides or {}
|
|
|
|
_STATE_CACHE["value"] = None
|
|
_FAN_CACHE["value"] = None
|
|
ACTIVE_PROFILE = name
|
|
_LAST_RESULT = result
|
|
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
|
|
return result
|
|
|
|
|
|
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
STATE_TTL_S = 2.0
|
|
|
|
|
|
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
|
|
"""Read back live GPU clocks/power/limits via nvidia-smi.
|
|
|
|
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
|
|
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
|
|
"""
|
|
import time as _time
|
|
if not force and _STATE_CACHE["value"] is not None and \
|
|
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
|
|
return _STATE_CACHE["value"]
|
|
state: Dict[str, Any] = {}
|
|
r = _smi(
|
|
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
|
|
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
|
|
"temperature.gpu,power.draw,fan.speed",
|
|
"--format=csv,noheader,nounits",
|
|
)
|
|
if r["rc"] == 0 and r["out"]:
|
|
parts = [p.strip() for p in r["out"].split(",")]
|
|
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
|
|
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
|
|
"temp_c", "power_draw_w", "fan_pct"]
|
|
for i, k in enumerate(keys):
|
|
if i < len(parts):
|
|
try:
|
|
state[k] = float(parts[i])
|
|
except ValueError:
|
|
state[k] = parts[i]
|
|
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
|
|
return state
|
|
|
|
|
|
def is_headless_x_running() -> bool:
|
|
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
|
|
return r["rc"] == 0
|
|
|
|
|
|
def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]:
|
|
global FAN_MANUAL
|
|
if mode == "auto":
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")}
|
|
|
|
speed_pct = max(30, min(100, int(speed_pct)))
|
|
r = _nvidia_settings(
|
|
"-a", "[gpu:0]/GPUFanControlState=1",
|
|
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
|
|
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
|
|
)
|
|
ok = (r["rc"] == 0)
|
|
if ok:
|
|
FAN_MANUAL = True
|
|
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def set_fan_speed(percent: int) -> Dict[str, Any]:
|
|
"""Set manual GPU fan target speed (30-100%)."""
|
|
return apply_fan_control(mode="manual", speed_pct=percent)
|
|
|
|
|
|
def set_fan_auto() -> Dict[str, Any]:
|
|
"""Return GPU fan to automatic control."""
|
|
global FAN_MANUAL
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
_FAN_CACHE["value"] = None
|
|
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def get_fan_status(force: bool = False) -> Dict[str, Any]:
|
|
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
|
|
import time as _time
|
|
if not force and _FAN_CACHE["value"] is not None and \
|
|
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
|
|
return _FAN_CACHE["value"]
|
|
global FAN_MANUAL
|
|
target = None
|
|
manual = FAN_MANUAL
|
|
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
|
|
if r.get("rc") == 0 and r.get("out"):
|
|
for line in r["out"].splitlines():
|
|
if "GPUFanControlState" in line and ":" in line:
|
|
try:
|
|
val = int(line.split("):")[-1].split(".")[0].strip())
|
|
manual = (val == 1)
|
|
except Exception:
|
|
pass
|
|
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
|
|
try:
|
|
target = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
|
|
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
|
|
return result
|
|
|
|
|
|
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
|
|
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
|
|
|
|
This matters because every lever here is sticky. If the service dies while a profile is
|
|
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
|
|
manual PWM was last set. Nothing was undoing that.
|
|
"""
|
|
logger.warning(f"Restoring GPU to safe stock state ({reason})")
|
|
result = {
|
|
"reason": reason,
|
|
"clock_lock": _apply_clock_lock(0, 0),
|
|
"mem_lock": _apply_mem_lock(0),
|
|
"offsets": _apply_offsets(0, 0),
|
|
"fan": set_fan_auto(),
|
|
}
|
|
# Hand the power limit back to the card's own default rather than assuming 370 W.
|
|
state = get_gpu_state()
|
|
default_w = state.get("power_default_w")
|
|
if isinstance(default_w, (int, float)) and default_w > 0:
|
|
result["power_limit"] = _apply_power_limit(int(default_w))
|
|
global ACTIVE_PROFILE
|
|
ACTIVE_PROFILE = "stock"
|
|
return result
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
"""Full overclock status for the dashboard."""
|
|
return {
|
|
"active_profile": ACTIVE_PROFILE,
|
|
"offsets_supported": offsets_supported(),
|
|
"effective_levers": (["power_limit", "clock_lock", "mem_lock", "fan"]
|
|
+ (["offsets"] if offsets_supported() else [])),
|
|
"profiles": load_profiles(),
|
|
"gpu": get_gpu_state(),
|
|
"fan": get_fan_status(),
|
|
"headless_x_running": is_headless_x_running(),
|
|
"headless_display": HEADLESS_DISPLAY,
|
|
"last_result": _LAST_RESULT,
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
logging.basicConfig(level=logging.INFO)
|
|
if len(sys.argv) > 1:
|
|
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
|
|
else:
|
|
print(json.dumps(get_status(), indent=2, default=str))
|