""" Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER. Lever hierarchy (what actually works on this box): 1. Power limit nvidia-smi -pl -> 320W -> 370W max (BIG win, works on open module) 2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module) 3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module) Profiles are application-specific: - ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power - comfy : diffusion is compute bound -> lock core clock high + max power - balanced: stock boost, power unlocked only Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator). """ import json import logging import os import shutil import subprocess from typing import Dict, Any, Optional, List logger = logging.getLogger("overclock_manager") _BASE = os.path.dirname(os.path.abspath(__file__)) CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json") NVIDIA_SMI = "nvidia-smi" NVIDIA_SETTINGS = "nvidia-settings" HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless" # Default profile set. lock_* == 0 means "don't lock" (let boost manage). DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = { "ollama": { "label": "Ollama — LLM decode (memory-bandwidth bound)", "power_limit_w": 370, "core_offset_mhz": 100, "mem_offset_mhz": 500, "lock_core_min": 0, "lock_core_max": 0, "lock_mem_mhz": 11501, "fan_mode": "manual", "fan_speed_pct": 60, }, "comfy": { "label": "ComfyUI — diffusion (core-compute bound)", "power_limit_w": 370, "core_offset_mhz": 100, "mem_offset_mhz": 500, "lock_core_min": 2900, "lock_core_max": 3105, "lock_mem_mhz": 0, "fan_mode": "manual", "fan_speed_pct": 75, }, "balanced": { "label": "Balanced — stock boost, power unlocked", "power_limit_w": 370, "core_offset_mhz": 0, "mem_offset_mhz": 0, "lock_core_min": 0, "lock_core_max": 0, "lock_mem_mhz": 0, "fan_mode": "auto", "fan_speed_pct": 0, }, } _OFFSETS_SUPPORTED: Optional[bool] = None def offsets_supported(recheck: bool = False) -> bool: """Whether nvidia-settings clock offsets actually take effect on this driver. Driver 595.84 accepts GPUGraphicsClockOffset/GPUMemoryTransferRateOffset and silently discards them: assigning 0 returns success and the attribute still reads back its old value. Profiles carrying core_offset_mhz/mem_offset_mhz were therefore configuring nothing. Probed once and cached. """ global _OFFSETS_SUPPORTED if _OFFSETS_SUPPORTED is not None and not recheck: return _OFFSETS_SUPPORTED def _read() -> Optional[int]: q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]") for line in (q.get("out") or "").splitlines(): if "Attribute" in line and "):" in line: try: return int(line.split("):")[-1].split(".")[0].strip()) except Exception: pass return None before = _read() if before is None: _OFFSETS_SUPPORTED = False return False probe = before + 25 _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={probe}") after = _read() _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={before}") _OFFSETS_SUPPORTED = (after is not None and after != before) if not _OFFSETS_SUPPORTED: logger.warning("Clock offsets are not honoured by this driver " f"(set {probe}, read back {after}); profile offset fields are inert.") return _OFFSETS_SUPPORTED ACTIVE_PROFILE = "balanced" _LAST_RESULT: Dict[str, Any] = {} FAN_MANUAL = False def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]: """Run a command; return rc/stdout/stderr. Prefers passwordless sudo.""" full = list(cmd) if use_sudo: full = ["sudo", "-n"] + full try: proc = subprocess.run( full, capture_output=True, text=True, timeout=timeout ) return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()} except subprocess.TimeoutExpired: return {"rc": -1, "out": "", "err": "timeout"} except FileNotFoundError as e: return {"rc": -1, "out": "", "err": f"not found: {e}"} def _smi(*args: str) -> Dict[str, Any]: return _sh([NVIDIA_SMI, *args], use_sudo=True) def _nvidia_settings(*args: str) -> Dict[str, Any]: """Run nvidia-settings against the headless X display that owns the GPU.""" env = os.environ.copy() env["DISPLAY"] = HEADLESS_DISPLAY cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args] try: proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env) return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()} except Exception as e: return {"rc": -1, "out": "", "err": str(e)} def load_profiles() -> Dict[str, Dict[str, Any]]: """Load profiles from disk, falling back to defaults and merging new keys.""" profiles = json.loads(json.dumps(DEFAULT_PROFILES)) if os.path.exists(CONFIG_PATH): try: with open(CONFIG_PATH) as f: stored = json.load(f) for name, cfg in stored.items(): if name in profiles: profiles[name].update(cfg) else: profiles[name] = cfg except Exception as e: logger.warning(f"Could not load {CONFIG_PATH}: {e}") return profiles def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool: try: with open(CONFIG_PATH, "w") as f: json.dump(profiles, f, indent=2) return True except Exception as e: logger.error(f"save_profiles failed: {e}") return False def get_profiles() -> Dict[str, Dict[str, Any]]: return load_profiles() def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]: profiles = load_profiles() if name not in profiles: return {"success": False, "error": f"unknown profile '{name}'"} profiles[name].update(cfg) ok = save_profiles(profiles) return {"success": ok, "profiles": profiles if ok else None} def _apply_power_limit(watts: int) -> Dict[str, Any]: r = _smi("-pl", str(watts)) ok = r["rc"] == 0 return {"applied": ok, "detail": r.get("out") or r.get("err")} def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]: if core_min == 0 and core_max == 0: r = _smi("-rgc") return {"applied": r["rc"] == 0, "detail": "reset"} r = _smi("-lgc", f"{core_min},{core_max}") return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")} def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]: if mem_mhz == 0: r = _smi("-rmc") return {"applied": r["rc"] == 0, "detail": "reset"} r = _smi("-lmc", str(mem_mhz)) return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")} def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]: """Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck. NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open.""" if core_mhz == 0 and mem_mhz == 0: r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0", "-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0") return {"applied": r["rc"] == 0, "detail": "reset", "supported": True} r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}", "-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}") if r["rc"] != 0: return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False} # Read back to confirm the driver actually persisted the offsets. q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]", "-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]") applied_core = applied_mem = None for line in q["out"].splitlines(): line = line.strip() if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line: try: applied_core = int(line.split("):")[-1].split(".")[0].strip()) except Exception: pass if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line: try: applied_mem = int(line.split("):")[-1].split(".")[0].strip()) except Exception: pass supported = (applied_core is not None and applied_core != 0) or \ (applied_mem is not None and applied_mem != 0) return { "applied": supported, "supported": supported, "readback_core": applied_core, "readback_mem": applied_mem, "detail": f"core readback={applied_core}, mem readback={applied_mem}", } def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: """Apply a named overclock profile to the GPU. Returns a full result report. `overrides` lets the thermal governor and the autotuner apply a modified version of a profile (a derated offset, a probe clock) without mutating what is stored on disk. """ global ACTIVE_PROFILE, _LAST_RESULT profiles = load_profiles() if name not in profiles: return {"success": False, "error": f"unknown profile '{name}'", "profile": name} cfg = dict(profiles[name]) if overrides: cfg.update(overrides) fan_mode = cfg.get("fan_mode", "auto") fan_speed = int(cfg.get("fan_speed_pct", 0)) result = { "success": True, "profile": name, "label": cfg.get("label", name), "power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))), "clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))), "mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))), "offsets": (_apply_offsets(int(cfg.get("core_offset_mhz", 0)), int(cfg.get("mem_offset_mhz", 0))) if offsets_supported() else {"applied": False, "supported": False, "detail": "skipped: this driver accepts clock offsets and ignores them"}), "fan": apply_fan_control(fan_mode, fan_speed), } result["gpu"] = get_gpu_state() result["fan_status"] = get_fan_status() result["overrides"] = overrides or {} _STATE_CACHE["value"] = None _FAN_CACHE["value"] = None ACTIVE_PROFILE = name _LAST_RESULT = result logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}") return result _STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None} _FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None} STATE_TTL_S = 2.0 def get_gpu_state(force: bool = False) -> Dict[str, Any]: """Read back live GPU clocks/power/limits via nvidia-smi. Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the status endpoint every few seconds. NVML already covers the live 1 Hz telemetry. """ import time as _time if not force and _STATE_CACHE["value"] is not None and \ (_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S: return _STATE_CACHE["value"] state: Dict[str, Any] = {} r = _smi( "--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit," "clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem," "temperature.gpu,power.draw,fan.speed", "--format=csv,noheader,nounits", ) if r["rc"] == 0 and r["out"]: parts = [p.strip() for p in r["out"].split(",")] keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w", "clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz", "temp_c", "power_draw_w", "fan_pct"] for i, k in enumerate(keys): if i < len(parts): try: state[k] = float(parts[i]) except ValueError: state[k] = parts[i] _STATE_CACHE.update({"ts": __import__("time").time(), "value": state}) return state def is_headless_x_running() -> bool: r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False) return r["rc"] == 0 def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]: global FAN_MANUAL if mode == "auto": r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0") ok = r["rc"] == 0 if ok: FAN_MANUAL = False return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")} speed_pct = max(30, min(100, int(speed_pct))) r = _nvidia_settings( "-a", "[gpu:0]/GPUFanControlState=1", "-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}", "-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}" ) ok = (r["rc"] == 0) if ok: FAN_MANUAL = True return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")} def set_fan_speed(percent: int) -> Dict[str, Any]: """Set manual GPU fan target speed (30-100%).""" return apply_fan_control(mode="manual", speed_pct=percent) def set_fan_auto() -> Dict[str, Any]: """Return GPU fan to automatic control.""" global FAN_MANUAL r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0") ok = r["rc"] == 0 if ok: FAN_MANUAL = False _FAN_CACHE["value"] = None return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")} def get_fan_status(force: bool = False) -> Dict[str, Any]: """Read current fan control mode + target speed (cached; forks nvidia-settings).""" import time as _time if not force and _FAN_CACHE["value"] is not None and \ (_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S: return _FAN_CACHE["value"] global FAN_MANUAL target = None manual = FAN_MANUAL r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed") if r.get("rc") == 0 and r.get("out"): for line in r["out"].splitlines(): if "GPUFanControlState" in line and ":" in line: try: val = int(line.split("):")[-1].split(".")[0].strip()) manual = (val == 1) except Exception: pass elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line: try: target = int(line.split("):")[-1].split(".")[0].strip()) except Exception: pass result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target} _FAN_CACHE.update({"ts": __import__("time").time(), "value": result}) return result def restore_safe(reason: str = "shutdown") -> Dict[str, Any]: """Return the card to stock: no clock locks, no offsets, default power, automatic fans. This matters because every lever here is sticky. If the service dies while a profile is applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever manual PWM was last set. Nothing was undoing that. """ logger.warning(f"Restoring GPU to safe stock state ({reason})") result = { "reason": reason, "clock_lock": _apply_clock_lock(0, 0), "mem_lock": _apply_mem_lock(0), "offsets": _apply_offsets(0, 0), "fan": set_fan_auto(), } # Hand the power limit back to the card's own default rather than assuming 370 W. state = get_gpu_state() default_w = state.get("power_default_w") if isinstance(default_w, (int, float)) and default_w > 0: result["power_limit"] = _apply_power_limit(int(default_w)) global ACTIVE_PROFILE ACTIVE_PROFILE = "stock" return result def get_status() -> Dict[str, Any]: """Full overclock status for the dashboard.""" return { "active_profile": ACTIVE_PROFILE, "offsets_supported": offsets_supported(), "effective_levers": (["power_limit", "clock_lock", "mem_lock", "fan"] + (["offsets"] if offsets_supported() else [])), "profiles": load_profiles(), "gpu": get_gpu_state(), "fan": get_fan_status(), "headless_x_running": is_headless_x_running(), "headless_display": HEADLESS_DISPLAY, "last_result": _LAST_RESULT, } if __name__ == "__main__": import sys logging.basicConfig(level=logging.INFO) if len(sys.argv) > 1: print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str)) else: print(json.dumps(get_status(), indent=2, default=str))