The reconciler only compared the power limit, so a profile whose fan setting never applied stayed wrong indefinitely. The headless X server that owns the GPU can be starting when this unit does; in-process retries cover a short delay, but if X arrives later nothing noticed that the fan mode had never been set. profile_drift() now compares fan mode too, gated on fan control having worked at least once so the check does not fire forever on a machine without it. README documents the self-check endpoint, the ollama/comfy/desktop/unmanaged bucketing and why unreclaimable VRAM is reported separately, the SSE trimming, and a tests section listing the measured constants the suite pins. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
568 lines
22 KiB
Python
568 lines
22 KiB
Python
"""
|
|
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
|
|
|
|
Lever hierarchy (what actually works on this box):
|
|
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
|
|
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
|
|
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
|
|
|
|
Profiles are application-specific:
|
|
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
|
|
- comfy : diffusion is compute bound -> lock core clock high + max power
|
|
- balanced: stock boost, power unlocked only
|
|
|
|
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
|
|
"""
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
import time
|
|
from typing import Dict, Any, Optional, List
|
|
|
|
logger = logging.getLogger("overclock_manager")
|
|
|
|
_BASE = os.path.dirname(os.path.abspath(__file__))
|
|
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
|
|
|
|
NVIDIA_SMI = "nvidia-smi"
|
|
NVIDIA_SETTINGS = "nvidia-settings"
|
|
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
|
|
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
|
|
|
|
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
|
|
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
|
|
"ollama": {
|
|
"label": "Ollama — LLM decode (memory-bandwidth bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 11501,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 60,
|
|
},
|
|
"comfy": {
|
|
"label": "ComfyUI — diffusion (core-compute bound)",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 100,
|
|
"mem_offset_mhz": 500,
|
|
"lock_core_min": 2900,
|
|
"lock_core_max": 3105,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "manual",
|
|
"fan_speed_pct": 75,
|
|
},
|
|
"balanced": {
|
|
"label": "Balanced — stock boost, power unlocked",
|
|
"power_limit_w": 370,
|
|
"core_offset_mhz": 0,
|
|
"mem_offset_mhz": 0,
|
|
"lock_core_min": 0,
|
|
"lock_core_max": 0,
|
|
"lock_mem_mhz": 0,
|
|
"fan_mode": "auto",
|
|
"fan_speed_pct": 0,
|
|
},
|
|
}
|
|
|
|
_OFFSETS_SUPPORTED: Optional[bool] = None
|
|
|
|
|
|
def offsets_supported(recheck: bool = False) -> bool:
|
|
"""Whether nvidia-settings clock offsets actually take effect on this driver.
|
|
|
|
Driver 595.84 accepts GPUGraphicsClockOffset/GPUMemoryTransferRateOffset and silently
|
|
discards them: assigning 0 returns success and the attribute still reads back its old
|
|
value. Profiles carrying core_offset_mhz/mem_offset_mhz were therefore configuring
|
|
nothing. Probed once and cached.
|
|
"""
|
|
global _OFFSETS_SUPPORTED
|
|
if _OFFSETS_SUPPORTED is not None and not recheck:
|
|
return _OFFSETS_SUPPORTED
|
|
|
|
def _read() -> Optional[int]:
|
|
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]")
|
|
for line in (q.get("out") or "").splitlines():
|
|
if "Attribute" in line and "):" in line:
|
|
try:
|
|
return int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
before = _read()
|
|
if before is None:
|
|
_OFFSETS_SUPPORTED = False
|
|
return False
|
|
probe = before + 25
|
|
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={probe}")
|
|
after = _read()
|
|
_nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={before}")
|
|
_OFFSETS_SUPPORTED = (after is not None and after != before)
|
|
if not _OFFSETS_SUPPORTED:
|
|
logger.warning("Clock offsets are not honoured by this driver "
|
|
f"(set {probe}, read back {after}); profile offset fields are inert.")
|
|
return _OFFSETS_SUPPORTED
|
|
|
|
|
|
ACTIVE_PROFILE = "balanced"
|
|
_LAST_RESULT: Dict[str, Any] = {}
|
|
FAN_MANUAL = False
|
|
|
|
|
|
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
|
|
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
|
|
full = list(cmd)
|
|
if use_sudo:
|
|
full = ["sudo", "-n"] + full
|
|
try:
|
|
proc = subprocess.run(
|
|
full, capture_output=True, text=True, timeout=timeout
|
|
)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except subprocess.TimeoutExpired:
|
|
return {"rc": -1, "out": "", "err": "timeout"}
|
|
except FileNotFoundError as e:
|
|
return {"rc": -1, "out": "", "err": f"not found: {e}"}
|
|
|
|
|
|
def _smi(*args: str) -> Dict[str, Any]:
|
|
return _sh([NVIDIA_SMI, *args], use_sudo=True)
|
|
|
|
|
|
def _nvidia_settings(*args: str) -> Dict[str, Any]:
|
|
"""Run nvidia-settings against the headless X display that owns the GPU."""
|
|
env = os.environ.copy()
|
|
env["DISPLAY"] = HEADLESS_DISPLAY
|
|
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
|
|
try:
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
|
|
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
|
|
except Exception as e:
|
|
return {"rc": -1, "out": "", "err": str(e)}
|
|
|
|
|
|
def load_profiles() -> Dict[str, Dict[str, Any]]:
|
|
"""Load profiles from disk, falling back to defaults and merging new keys."""
|
|
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
|
|
if os.path.exists(CONFIG_PATH):
|
|
try:
|
|
with open(CONFIG_PATH) as f:
|
|
stored = json.load(f)
|
|
for name, cfg in stored.items():
|
|
if name in profiles:
|
|
profiles[name].update(cfg)
|
|
else:
|
|
profiles[name] = cfg
|
|
except Exception as e:
|
|
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
|
|
return profiles
|
|
|
|
|
|
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
|
|
try:
|
|
with open(CONFIG_PATH, "w") as f:
|
|
json.dump(profiles, f, indent=2)
|
|
return True
|
|
except Exception as e:
|
|
logger.error(f"save_profiles failed: {e}")
|
|
return False
|
|
|
|
|
|
def get_profiles() -> Dict[str, Dict[str, Any]]:
|
|
return load_profiles()
|
|
|
|
|
|
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'"}
|
|
profiles[name].update(cfg)
|
|
ok = save_profiles(profiles)
|
|
return {"success": ok, "profiles": profiles if ok else None}
|
|
|
|
|
|
def _apply_power_limit(watts: int) -> Dict[str, Any]:
|
|
r = _smi("-pl", str(watts))
|
|
ok = r["rc"] == 0
|
|
return {"applied": ok, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
|
|
if core_min == 0 and core_max == 0:
|
|
r = _smi("-rgc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lgc", f"{core_min},{core_max}")
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
|
|
if mem_mhz == 0:
|
|
r = _smi("-rmc")
|
|
return {"applied": r["rc"] == 0, "detail": "reset"}
|
|
r = _smi("-lmc", str(mem_mhz))
|
|
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
|
|
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
|
|
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
|
|
if core_mhz == 0 and mem_mhz == 0:
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
|
|
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
|
|
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
|
|
|
|
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
|
|
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
|
|
if r["rc"] != 0:
|
|
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
|
|
|
|
# Read back to confirm the driver actually persisted the offsets.
|
|
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
|
|
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
|
|
applied_core = applied_mem = None
|
|
for line in q["out"].splitlines():
|
|
line = line.strip()
|
|
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_core = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
|
|
try:
|
|
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
|
|
supported = (applied_core is not None and applied_core != 0) or \
|
|
(applied_mem is not None and applied_mem != 0)
|
|
return {
|
|
"applied": supported,
|
|
"supported": supported,
|
|
"readback_core": applied_core,
|
|
"readback_mem": applied_mem,
|
|
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
|
|
}
|
|
|
|
|
|
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
"""Apply a named overclock profile to the GPU. Returns a full result report.
|
|
|
|
`overrides` lets the thermal governor and the autotuner apply a modified version of a
|
|
profile (a derated offset, a probe clock) without mutating what is stored on disk.
|
|
"""
|
|
global ACTIVE_PROFILE, _LAST_RESULT
|
|
profiles = load_profiles()
|
|
if name not in profiles:
|
|
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
|
|
|
|
cfg = dict(profiles[name])
|
|
if overrides:
|
|
cfg.update(overrides)
|
|
fan_mode = cfg.get("fan_mode", "auto")
|
|
fan_speed = int(cfg.get("fan_speed_pct", 0))
|
|
|
|
result = {
|
|
"success": True,
|
|
"profile": name,
|
|
"label": cfg.get("label", name),
|
|
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
|
|
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
|
|
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
|
|
"offsets": (_apply_offsets(int(cfg.get("core_offset_mhz", 0)),
|
|
int(cfg.get("mem_offset_mhz", 0)))
|
|
if offsets_supported() else
|
|
{"applied": False, "supported": False,
|
|
"detail": "skipped: this driver accepts clock offsets and ignores them"}),
|
|
"fan": apply_fan_control(fan_mode, fan_speed),
|
|
}
|
|
# Invalidate BEFORE reading back. These caches exist so the dashboard's polling does
|
|
# not fork sudo every few seconds, but reading through them here reported the
|
|
# pre-change value: a profile that had just moved the card 370W -> 320W came back
|
|
# claiming 370W, so the API contradicted nvidia-smi.
|
|
_STATE_CACHE["value"] = None
|
|
_FAN_CACHE["value"] = None
|
|
result["gpu"] = get_gpu_state(force=True)
|
|
result["fan_status"] = get_fan_status(force=True)
|
|
result["overrides"] = overrides or {}
|
|
|
|
# Say plainly whether the card ended up where the profile asked it to.
|
|
intended_w = int(cfg.get("power_limit_w", 0))
|
|
actual_w = result["gpu"].get("power_limit_w")
|
|
result["verified"] = {
|
|
"power_limit_intended_w": intended_w,
|
|
"power_limit_actual_w": actual_w,
|
|
"power_limit_ok": (actual_w is None or intended_w == 0
|
|
or abs(float(actual_w) - intended_w) < 1.0),
|
|
"fan_mode_intended": fan_mode,
|
|
"fan_mode_actual": result["fan_status"].get("mode"),
|
|
"fan_ok": result["fan"].get("applied", False),
|
|
}
|
|
if not result["verified"]["power_limit_ok"]:
|
|
logger.warning(f"Profile '{name}' asked for {intended_w}W but the card reports "
|
|
f"{actual_w}W")
|
|
if not result["verified"]["fan_ok"]:
|
|
logger.warning(f"Profile '{name}' could not set fans: "
|
|
f"{result['fan'].get('detail')}")
|
|
|
|
global _APPLIED_ONCE, _FAN_AVAILABLE
|
|
if result["verified"]["power_limit_ok"]:
|
|
_APPLIED_ONCE = True
|
|
if result["verified"]["fan_ok"]:
|
|
_FAN_AVAILABLE = True
|
|
ACTIVE_PROFILE = name
|
|
_LAST_RESULT = result
|
|
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
|
|
return result
|
|
|
|
|
|
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
|
STATE_TTL_S = 2.0
|
|
|
|
|
|
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
|
|
"""Read back live GPU clocks/power/limits via nvidia-smi.
|
|
|
|
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
|
|
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
|
|
"""
|
|
import time as _time
|
|
if not force and _STATE_CACHE["value"] is not None and \
|
|
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
|
|
return _STATE_CACHE["value"]
|
|
state: Dict[str, Any] = {}
|
|
r = _smi(
|
|
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
|
|
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
|
|
"temperature.gpu,power.draw,fan.speed",
|
|
"--format=csv,noheader,nounits",
|
|
)
|
|
if r["rc"] == 0 and r["out"]:
|
|
parts = [p.strip() for p in r["out"].split(",")]
|
|
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
|
|
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
|
|
"temp_c", "power_draw_w", "fan_pct"]
|
|
for i, k in enumerate(keys):
|
|
if i < len(parts):
|
|
try:
|
|
state[k] = float(parts[i])
|
|
except ValueError:
|
|
state[k] = parts[i]
|
|
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
|
|
return state
|
|
|
|
|
|
def is_headless_x_running() -> bool:
|
|
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
|
|
return r["rc"] == 0
|
|
|
|
|
|
FAN_RETRY_ATTEMPTS = 5
|
|
FAN_RETRY_DELAY_S = 2.0
|
|
|
|
|
|
def _fan_target_missing(result: Dict[str, Any]) -> bool:
|
|
"""True when nvidia-settings could not see the GPU at all.
|
|
|
|
On boot this service can start before the headless X server on :8 that owns the GPU
|
|
is accepting connections, and the fan assignment fails with 'Error resolving target
|
|
specification'. Nothing retried, so the fans were simply never configured for the
|
|
whole session and the failure was only visible deep in a log line.
|
|
"""
|
|
text = ((result.get("err") or "") + (result.get("out") or "")).lower()
|
|
return ("error resolving target" in text or "no targets match" in text
|
|
or "cannot open display" in text)
|
|
|
|
|
|
def apply_fan_control(mode: str, speed_pct: int, _attempt: int = 0) -> Dict[str, Any]:
|
|
global FAN_MANUAL
|
|
if mode == "auto":
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if not ok and _fan_target_missing(r) and _attempt < FAN_RETRY_ATTEMPTS:
|
|
logger.info(f"Fan control target not ready (attempt {_attempt + 1}/"
|
|
f"{FAN_RETRY_ATTEMPTS}); X on {HEADLESS_DISPLAY} may still be "
|
|
f"starting — retrying in {FAN_RETRY_DELAY_S}s")
|
|
time.sleep(FAN_RETRY_DELAY_S)
|
|
return apply_fan_control(mode, speed_pct, _attempt + 1)
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
_FAN_CACHE["value"] = None
|
|
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err"),
|
|
"attempts": _attempt + 1}
|
|
|
|
speed_pct = max(30, min(100, int(speed_pct)))
|
|
r = _nvidia_settings(
|
|
"-a", "[gpu:0]/GPUFanControlState=1",
|
|
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
|
|
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
|
|
)
|
|
ok = (r["rc"] == 0)
|
|
if ok:
|
|
FAN_MANUAL = True
|
|
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def set_fan_speed(percent: int) -> Dict[str, Any]:
|
|
"""Set manual GPU fan target speed (30-100%)."""
|
|
return apply_fan_control(mode="manual", speed_pct=percent)
|
|
|
|
|
|
def set_fan_auto() -> Dict[str, Any]:
|
|
"""Return GPU fan to automatic control."""
|
|
global FAN_MANUAL
|
|
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
|
|
ok = r["rc"] == 0
|
|
if ok:
|
|
FAN_MANUAL = False
|
|
_FAN_CACHE["value"] = None
|
|
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
|
|
|
|
|
|
def get_fan_status(force: bool = False) -> Dict[str, Any]:
|
|
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
|
|
import time as _time
|
|
if not force and _FAN_CACHE["value"] is not None and \
|
|
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
|
|
return _FAN_CACHE["value"]
|
|
global FAN_MANUAL
|
|
target = None
|
|
manual = FAN_MANUAL
|
|
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
|
|
if r.get("rc") == 0 and r.get("out"):
|
|
for line in r["out"].splitlines():
|
|
if "GPUFanControlState" in line and ":" in line:
|
|
try:
|
|
val = int(line.split("):")[-1].split(".")[0].strip())
|
|
manual = (val == 1)
|
|
except Exception:
|
|
pass
|
|
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
|
|
try:
|
|
target = int(line.split("):")[-1].split(".")[0].strip())
|
|
except Exception:
|
|
pass
|
|
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
|
|
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
|
|
return result
|
|
|
|
|
|
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
|
|
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
|
|
|
|
This matters because every lever here is sticky. If the service dies while a profile is
|
|
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
|
|
manual PWM was last set. Nothing was undoing that.
|
|
"""
|
|
logger.warning(f"Restoring GPU to safe stock state ({reason})")
|
|
result = {
|
|
"reason": reason,
|
|
"clock_lock": _apply_clock_lock(0, 0),
|
|
"mem_lock": _apply_mem_lock(0),
|
|
"offsets": _apply_offsets(0, 0),
|
|
"fan": set_fan_auto(),
|
|
}
|
|
# Hand the power limit back to the card's own default rather than assuming 370 W.
|
|
state = get_gpu_state()
|
|
default_w = state.get("power_default_w")
|
|
if isinstance(default_w, (int, float)) and default_w > 0:
|
|
result["power_limit"] = _apply_power_limit(int(default_w))
|
|
global ACTIVE_PROFILE
|
|
ACTIVE_PROFILE = "stock"
|
|
return result
|
|
|
|
|
|
_APPLIED_ONCE = False
|
|
# Set once fan control has worked at least once, so drift checks do not fire forever on
|
|
# a machine that simply has no fan control available.
|
|
_FAN_AVAILABLE = False
|
|
|
|
|
|
def profile_drift() -> Dict[str, Any]:
|
|
"""Compare what the active profile asks for against what the card actually reports.
|
|
|
|
ACTIVE_PROFILE defaults to "balanced" at import, which is indistinguishable from
|
|
"balanced was successfully applied" -- so a startup apply that failed left the app
|
|
confidently reporting a profile it had never put on the hardware. This makes the
|
|
difference visible instead.
|
|
"""
|
|
profiles = load_profiles()
|
|
cfg = profiles.get(ACTIVE_PROFILE, {})
|
|
state = get_gpu_state()
|
|
intended = int(cfg.get("power_limit_w", 0) or 0)
|
|
actual = state.get("power_limit_w")
|
|
power_drift = bool(intended and actual is not None
|
|
and abs(float(actual) - intended) >= 1.0)
|
|
|
|
# Fan mode is checked too. The headless X server that owns the GPU can still be
|
|
# starting when this unit does, and the fan assignment then fails; the in-process
|
|
# retries cover a short delay, but if X arrives later nothing else would ever notice
|
|
# that the profile's fan setting was never applied.
|
|
fan_intended = cfg.get("fan_mode", "auto")
|
|
fan_actual = None
|
|
fan_drift = False
|
|
if _FAN_AVAILABLE:
|
|
fan_actual = get_fan_status().get("mode")
|
|
fan_drift = bool(fan_actual and fan_actual != fan_intended)
|
|
|
|
drifted = power_drift or fan_drift or not _APPLIED_ONCE
|
|
reasons = []
|
|
if not _APPLIED_ONCE:
|
|
reasons.append("no profile has been successfully applied since startup")
|
|
if power_drift:
|
|
reasons.append(f"card reports {actual}W, profile asks {intended}W")
|
|
if fan_drift:
|
|
reasons.append(f"fans are {fan_actual}, profile asks {fan_intended}")
|
|
return {
|
|
"profile": ACTIVE_PROFILE,
|
|
"applied_since_start": _APPLIED_ONCE,
|
|
"power_limit_intended_w": intended,
|
|
"power_limit_actual_w": actual,
|
|
"fan_mode_intended": fan_intended,
|
|
"fan_mode_actual": fan_actual,
|
|
"fan_available": _FAN_AVAILABLE,
|
|
"drifted": drifted,
|
|
"reason": "; ".join(reasons) or None,
|
|
}
|
|
|
|
|
|
def reconcile_profile() -> Dict[str, Any]:
|
|
"""Re-apply the active profile if the hardware has drifted away from it."""
|
|
drift = profile_drift()
|
|
if not drift["drifted"]:
|
|
return {"reconciled": False, "drift": drift}
|
|
logger.warning(f"Overclock drift detected — re-applying '{ACTIVE_PROFILE}': "
|
|
f"{drift['reason']}")
|
|
res = apply_profile(ACTIVE_PROFILE)
|
|
return {"reconciled": True, "drift": drift, "result": res.get("verified")}
|
|
|
|
|
|
def get_status() -> Dict[str, Any]:
|
|
"""Full overclock status for the dashboard."""
|
|
return {
|
|
"active_profile": ACTIVE_PROFILE,
|
|
"drift": profile_drift(),
|
|
"offsets_supported": offsets_supported(),
|
|
"effective_levers": (["power_limit", "clock_lock", "mem_lock", "fan"]
|
|
+ (["offsets"] if offsets_supported() else [])),
|
|
"profiles": load_profiles(),
|
|
"gpu": get_gpu_state(),
|
|
"fan": get_fan_status(),
|
|
"headless_x_running": is_headless_x_running(),
|
|
"headless_display": HEADLESS_DISPLAY,
|
|
"last_result": _LAST_RESULT,
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
logging.basicConfig(level=logging.INFO)
|
|
if len(sys.argv) > 1:
|
|
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
|
|
else:
|
|
print(json.dumps(get_status(), indent=2, default=str))
|