Files
gpu-program-swapper/thermal_governor.py
drjones 5431144b2e Add barrier-confirmed yielding, measured residency, persistence and closed-loop tuning
Nine changes, in rough order of how much they affect real behaviour:

1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload;
   measured here, the HTTP call returns in 63ms while the driver takes a further
   77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up
   allocating into VRAM that is still occupied. instant_free_ollama_vram() polls
   NVML until the allocation is actually gone and reports request/confirm split.

2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full
   checkpoint reload on each workflow iteration. It is held for 30s of genuinely
   empty queue, with an immediate purge when Ollama actually asks for the memory.

3. Cache-hit classification uses achieved bandwidth (size / load duration) rather
   than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read
   at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit.

4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB
   resident on a box with 46GB of page cache: the kernel only permits page-cache
   introspection on files you own, and the Ollama blobs are owned by uid ollama,
   for which mincore answers "all resident" instead of failing. Uses cachestat(2)
   where permitted and a randomised read-rate probe elsewhere, labelling which was
   used. Fixed-offset probing was self-fulfilling, so windows are random and cold
   ones are returned with FADV_DONTNEED.

5. Warming is budgeted and ranked by recency/frequency instead of reading every
   file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first.

6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a
   50-entry in-memory deque, so /api/analytics/profiles can finally answer whether
   an overclock profile actually delivers more tok/s.

7. Thermal governor walks the overclock back on sustained heat or hardware
   throttling, with hysteresis, fed from the existing sampler.

8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid
   errors and degenerate output, and restores the profile in a finally block.

9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost.
   Nothing previously undid a locked clock or a manually pinned fan.

Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than
every client re-running the whole snapshot; wall-clock timestamps in place of the
event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-28 08:57:35 -07:00

176 lines
7.1 KiB
Python

"""Thermal governor: walks the overclock back when the card says it is unhappy.
Nothing in HyperSwap used to de-escalate. A profile pinned the fans to a manual PWM and
locked the clocks, and those settings stayed exactly where they were put no matter what
the card reported. This module watches the telemetry the sampler already collects (so it
costs no extra NVML calls) and derates the active profile when the GPU is hot or
throttling, then restores it once things settle.
Hysteresis is deliberate: escalation needs HOT_SAMPLES consecutive bad samples and
recovery needs COOL_SAMPLES consecutive good ones, so a single spike during a diffusion
step does not cause profile thrash.
"""
import logging
import threading
import time
from typing import Any, Dict, List, Optional
import overclock_manager
logger = logging.getLogger("thermal_governor")
# Derate ladder: each step scales the profile's clock offsets and raises the fan floor.
DERATE_LADDER = [
{"level": 0, "offset_scale": 1.00, "fan_floor": 0, "label": "full"},
{"level": 1, "offset_scale": 0.60, "fan_floor": 80, "label": "derated -40%"},
{"level": 2, "offset_scale": 0.25, "fan_floor": 90, "label": "derated -75%"},
{"level": 3, "offset_scale": 0.00, "fan_floor": 100, "label": "stock clocks, fans max"},
]
TEMP_ESCALATE_C = 83.0
TEMP_RECOVER_C = 72.0
HOT_SAMPLES = 5 # ~5 s at 1 Hz before we act
COOL_SAMPLES = 30 # ~30 s of calm before we give the clocks back
REAPPLY_COOLDOWN_S = 20.0
# Throttle reasons that mean the hardware is protecting itself, not just hitting a cap.
HARD_THROTTLES = {"hw_thermal_slowdown", "sw_thermal_slowdown", "hw_slowdown",
"hw_power_brake_slowdown"}
class ThermalGovernor:
def __init__(self) -> None:
self.enabled = True
self.level = 0
self.hot_streak = 0
self.cool_streak = 0
self.last_change = 0.0
self.last_reason = "cold start"
self.history: List[Dict[str, Any]] = []
self._lock = threading.Lock()
self._base_profile: Optional[str] = None
# ---------------------------------------------------------------- observation
def observe(self, gpu: Dict[str, Any], active_profile: Optional[str] = None) -> None:
"""Feed one telemetry sample. Cheap and non-blocking; actuation runs in a thread."""
if not self.enabled or not gpu.get("available"):
return
temp = gpu.get("temperature_c") or 0
reasons = set(gpu.get("throttle_reasons") or [])
hard = bool(reasons & HARD_THROTTLES)
hot = temp >= TEMP_ESCALATE_C or hard
cool = temp <= TEMP_RECOVER_C and not hard
with self._lock:
if hot:
self.hot_streak += 1
self.cool_streak = 0
elif cool:
self.cool_streak += 1
self.hot_streak = 0
else:
self.hot_streak = 0
self.cool_streak = 0
now = time.time()
if now - self.last_change < REAPPLY_COOLDOWN_S:
return
if self.hot_streak >= HOT_SAMPLES and self.level < len(DERATE_LADDER) - 1:
why = (f"{temp:.0f}°C" if temp >= TEMP_ESCALATE_C else "") + \
(f" throttling: {','.join(sorted(reasons & HARD_THROTTLES))}" if hard else "")
self._step(self.level + 1, why.strip(), active_profile, temp)
elif self.cool_streak >= COOL_SAMPLES and self.level > 0:
self._step(self.level - 1, f"stable at {temp:.0f}°C", active_profile, temp)
def _step(self, new_level: int, reason: str, active_profile: Optional[str],
temp: float) -> None:
old = self.level
self.level = new_level
self.hot_streak = self.cool_streak = 0
self.last_change = time.time()
self.last_reason = reason
profile = active_profile or overclock_manager.ACTIVE_PROFILE
self._base_profile = profile
entry = {
"ts": time.time(), "from_level": old, "to_level": new_level,
"label": DERATE_LADDER[new_level]["label"], "reason": reason,
"temp_c": temp, "profile": profile,
}
self.history.insert(0, entry)
del self.history[50:]
direction = "escalating" if new_level > old else "recovering"
logger.warning(f"Thermal governor {direction} to level {new_level} "
f"({DERATE_LADDER[new_level]['label']}) — {reason}")
threading.Thread(target=self._actuate, args=(profile,), daemon=True).start()
def _actuate(self, profile: str) -> None:
try:
overclock_manager.apply_profile(profile, overrides=self.overrides_for(profile))
except Exception as e:
logger.error(f"Governor failed to apply derate: {e}")
# ---------------------------------------------------------------- overrides
def overrides_for(self, profile: str) -> Dict[str, Any]:
"""Translate the current derate level into profile overrides."""
step = DERATE_LADDER[self.level]
if self.level == 0:
return {}
cfg = overclock_manager.load_profiles().get(profile, {})
scale = step["offset_scale"]
overrides: Dict[str, Any] = {
"core_offset_mhz": int(cfg.get("core_offset_mhz", 0) * scale),
"mem_offset_mhz": int(cfg.get("mem_offset_mhz", 0) * scale),
}
if step["level"] >= 2:
# Stop pinning the core clock high when the card is already backing off.
overrides["lock_core_min"] = 0
overrides["lock_core_max"] = 0
floor = step["fan_floor"]
if floor:
overrides["fan_mode"] = "manual"
overrides["fan_speed_pct"] = max(int(cfg.get("fan_speed_pct", 0)), floor)
return overrides
# ---------------------------------------------------------------- control
def reset(self) -> Dict[str, Any]:
with self._lock:
self.level = 0
self.hot_streak = self.cool_streak = 0
self.last_change = time.time()
self.last_reason = "manual reset"
profile = self._base_profile or overclock_manager.ACTIVE_PROFILE
if profile and profile != "stock":
overclock_manager.apply_profile(profile)
return self.get_status()
def set_enabled(self, enabled: bool) -> Dict[str, Any]:
self.enabled = bool(enabled)
if not enabled and self.level > 0:
self.reset()
return self.get_status()
def get_status(self) -> Dict[str, Any]:
step = DERATE_LADDER[self.level]
return {
"enabled": self.enabled,
"level": self.level,
"label": step["label"],
"offset_scale": step["offset_scale"],
"fan_floor": step["fan_floor"],
"last_reason": self.last_reason,
"hot_streak": self.hot_streak,
"cool_streak": self.cool_streak,
"escalate_at_c": TEMP_ESCALATE_C,
"recover_below_c": TEMP_RECOVER_C,
"seconds_since_change": round(time.time() - self.last_change, 1) if self.last_change else None,
"history": self.history[:10],
}
governor = ThermalGovernor()