Report the GPU state that is real, and reconcile when it drifts

The API claimed the card was at 320W while nvidia-smi reported 370W. Three
separate defects, all introduced by me in this branch.

The readback was stale. get_gpu_state()/get_fan_status() gained a 2s cache so the
dashboard's polling would stop forking sudo every few seconds, but apply_profile
read back through that cache and its invalidation ran afterwards. A profile that
had just moved the card 370W -> 320W therefore returned a payload whose detail
string said "set to 320.00 W from 370.00 W" next to a power_limit_w of 370.0.
Caches are now cleared before the readback, which is forced.

Fan control could fail for an entire session. On boot this unit can start before
the headless X server on :8 that owns the GPU accepts connections, and the fan
assignment fails with "Error resolving target specification 'gpu:0'". Nothing
retried and nothing surfaced it, so the fans were left unconfigured with the
failure visible only inside one log line. apply_fan_control now recognises that
specific error and retries up to 5 times.

Nothing verified the result. ACTIVE_PROFILE defaults to "balanced" at import,
which is indistinguishable from "balanced was successfully applied" -- so a failed
startup apply left the app confidently reporting a profile it had never put on the
hardware. apply_profile now returns a `verified` block comparing intent against
readback and logs a warning on mismatch; profile_drift() exposes the comparison
plus whether any profile has actually been applied since startup; and the 1Hz
sampler calls reconcile_profile() once a minute to re-apply on drift.

Verified by setting 370W externally behind the service's back: the drift was
reported immediately and corrected automatically 40s later.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-02 19:51:23 -07:00
parent 868d82794d
commit f5917a0464
2 changed files with 109 additions and 6 deletions

View File

@@ -18,6 +18,7 @@ import logging
import os
import shutil
import subprocess
import time
from typing import Dict, Any, Optional, List
logger = logging.getLogger("overclock_manager")
@@ -278,12 +279,38 @@ def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict
"detail": "skipped: this driver accepts clock offsets and ignores them"}),
"fan": apply_fan_control(fan_mode, fan_speed),
}
result["gpu"] = get_gpu_state()
result["fan_status"] = get_fan_status()
result["overrides"] = overrides or {}
# Invalidate BEFORE reading back. These caches exist so the dashboard's polling does
# not fork sudo every few seconds, but reading through them here reported the
# pre-change value: a profile that had just moved the card 370W -> 320W came back
# claiming 370W, so the API contradicted nvidia-smi.
_STATE_CACHE["value"] = None
_FAN_CACHE["value"] = None
result["gpu"] = get_gpu_state(force=True)
result["fan_status"] = get_fan_status(force=True)
result["overrides"] = overrides or {}
# Say plainly whether the card ended up where the profile asked it to.
intended_w = int(cfg.get("power_limit_w", 0))
actual_w = result["gpu"].get("power_limit_w")
result["verified"] = {
"power_limit_intended_w": intended_w,
"power_limit_actual_w": actual_w,
"power_limit_ok": (actual_w is None or intended_w == 0
or abs(float(actual_w) - intended_w) < 1.0),
"fan_mode_intended": fan_mode,
"fan_mode_actual": result["fan_status"].get("mode"),
"fan_ok": result["fan"].get("applied", False),
}
if not result["verified"]["power_limit_ok"]:
logger.warning(f"Profile '{name}' asked for {intended_w}W but the card reports "
f"{actual_w}W")
if not result["verified"]["fan_ok"]:
logger.warning(f"Profile '{name}' could not set fans: "
f"{result['fan'].get('detail')}")
global _APPLIED_ONCE
if result["verified"]["power_limit_ok"]:
_APPLIED_ONCE = True
ACTIVE_PROFILE = name
_LAST_RESULT = result
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
@@ -332,14 +359,39 @@ def is_headless_x_running() -> bool:
return r["rc"] == 0
def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]:
FAN_RETRY_ATTEMPTS = 5
FAN_RETRY_DELAY_S = 2.0
def _fan_target_missing(result: Dict[str, Any]) -> bool:
"""True when nvidia-settings could not see the GPU at all.
On boot this service can start before the headless X server on :8 that owns the GPU
is accepting connections, and the fan assignment fails with 'Error resolving target
specification'. Nothing retried, so the fans were simply never configured for the
whole session and the failure was only visible deep in a log line.
"""
text = ((result.get("err") or "") + (result.get("out") or "")).lower()
return ("error resolving target" in text or "no targets match" in text
or "cannot open display" in text)
def apply_fan_control(mode: str, speed_pct: int, _attempt: int = 0) -> Dict[str, Any]:
global FAN_MANUAL
if mode == "auto":
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if not ok and _fan_target_missing(r) and _attempt < FAN_RETRY_ATTEMPTS:
logger.info(f"Fan control target not ready (attempt {_attempt + 1}/"
f"{FAN_RETRY_ATTEMPTS}); X on {HEADLESS_DISPLAY} may still be "
f"starting — retrying in {FAN_RETRY_DELAY_S}s")
time.sleep(FAN_RETRY_DELAY_S)
return apply_fan_control(mode, speed_pct, _attempt + 1)
if ok:
FAN_MANUAL = False
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")}
_FAN_CACHE["value"] = None
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err"),
"attempts": _attempt + 1}
speed_pct = max(30, min(100, int(speed_pct)))
r = _nvidia_settings(
@@ -422,10 +474,52 @@ def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
return result
_APPLIED_ONCE = False
def profile_drift() -> Dict[str, Any]:
"""Compare what the active profile asks for against what the card actually reports.
ACTIVE_PROFILE defaults to "balanced" at import, which is indistinguishable from
"balanced was successfully applied" -- so a startup apply that failed left the app
confidently reporting a profile it had never put on the hardware. This makes the
difference visible instead.
"""
profiles = load_profiles()
cfg = profiles.get(ACTIVE_PROFILE, {})
state = get_gpu_state()
intended = int(cfg.get("power_limit_w", 0) or 0)
actual = state.get("power_limit_w")
drifted = bool(intended and actual is not None
and abs(float(actual) - intended) >= 1.0)
return {
"profile": ACTIVE_PROFILE,
"applied_since_start": _APPLIED_ONCE,
"power_limit_intended_w": intended,
"power_limit_actual_w": actual,
"drifted": drifted or not _APPLIED_ONCE,
"reason": ("no profile has been successfully applied since startup"
if not _APPLIED_ONCE else
f"card reports {actual}W, profile asks {intended}W" if drifted else None),
}
def reconcile_profile() -> Dict[str, Any]:
"""Re-apply the active profile if the hardware has drifted away from it."""
drift = profile_drift()
if not drift["drifted"]:
return {"reconciled": False, "drift": drift}
logger.warning(f"Overclock drift detected — re-applying '{ACTIVE_PROFILE}': "
f"{drift['reason']}")
res = apply_profile(ACTIVE_PROFILE)
return {"reconciled": True, "drift": drift, "result": res.get("verified")}
def get_status() -> Dict[str, Any]:
"""Full overclock status for the dashboard."""
return {
"active_profile": ACTIVE_PROFILE,
"drift": profile_drift(),
"offsets_supported": offsets_supported(),
"effective_levers": (["power_limit", "clock_lock", "mem_lock", "fan"]
+ (["offsets"] if offsets_supported() else [])),