Nine changes, in rough order of how much they affect real behaviour: 1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload; measured here, the HTTP call returns in 63ms while the driver takes a further 77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up allocating into VRAM that is still occupied. instant_free_ollama_vram() polls NVML until the allocation is actually gone and reports request/confirm split. 2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full checkpoint reload on each workflow iteration. It is held for 30s of genuinely empty queue, with an immediate purge when Ollama actually asks for the memory. 3. Cache-hit classification uses achieved bandwidth (size / load duration) rather than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit. 4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB resident on a box with 46GB of page cache: the kernel only permits page-cache introspection on files you own, and the Ollama blobs are owned by uid ollama, for which mincore answers "all resident" instead of failing. Uses cachestat(2) where permitted and a randomised read-rate probe elsewhere, labelling which was used. Fixed-offset probing was self-fulfilling, so windows are random and cold ones are returned with FADV_DONTNEED. 5. Warming is budgeted and ranked by recency/frequency instead of reading every file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first. 6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a 50-entry in-memory deque, so /api/analytics/profiles can finally answer whether an overclock profile actually delivers more tok/s. 7. Thermal governor walks the overclock back on sustained heat or hardware throttling, with hysteresis, fed from the existing sampler. 8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid errors and degenerate output, and restores the profile in a finally block. 9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost. Nothing previously undid a locked clock or a manually pinned fan. Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than every client re-running the whole snapshot; wall-clock timestamps in place of the event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
176 lines
7.1 KiB
Python
176 lines
7.1 KiB
Python
"""Thermal governor: walks the overclock back when the card says it is unhappy.
|
|
|
|
Nothing in HyperSwap used to de-escalate. A profile pinned the fans to a manual PWM and
|
|
locked the clocks, and those settings stayed exactly where they were put no matter what
|
|
the card reported. This module watches the telemetry the sampler already collects (so it
|
|
costs no extra NVML calls) and derates the active profile when the GPU is hot or
|
|
throttling, then restores it once things settle.
|
|
|
|
Hysteresis is deliberate: escalation needs HOT_SAMPLES consecutive bad samples and
|
|
recovery needs COOL_SAMPLES consecutive good ones, so a single spike during a diffusion
|
|
step does not cause profile thrash.
|
|
"""
|
|
import logging
|
|
import threading
|
|
import time
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
import overclock_manager
|
|
|
|
logger = logging.getLogger("thermal_governor")
|
|
|
|
# Derate ladder: each step scales the profile's clock offsets and raises the fan floor.
|
|
DERATE_LADDER = [
|
|
{"level": 0, "offset_scale": 1.00, "fan_floor": 0, "label": "full"},
|
|
{"level": 1, "offset_scale": 0.60, "fan_floor": 80, "label": "derated -40%"},
|
|
{"level": 2, "offset_scale": 0.25, "fan_floor": 90, "label": "derated -75%"},
|
|
{"level": 3, "offset_scale": 0.00, "fan_floor": 100, "label": "stock clocks, fans max"},
|
|
]
|
|
|
|
TEMP_ESCALATE_C = 83.0
|
|
TEMP_RECOVER_C = 72.0
|
|
HOT_SAMPLES = 5 # ~5 s at 1 Hz before we act
|
|
COOL_SAMPLES = 30 # ~30 s of calm before we give the clocks back
|
|
REAPPLY_COOLDOWN_S = 20.0
|
|
|
|
# Throttle reasons that mean the hardware is protecting itself, not just hitting a cap.
|
|
HARD_THROTTLES = {"hw_thermal_slowdown", "sw_thermal_slowdown", "hw_slowdown",
|
|
"hw_power_brake_slowdown"}
|
|
|
|
|
|
class ThermalGovernor:
|
|
def __init__(self) -> None:
|
|
self.enabled = True
|
|
self.level = 0
|
|
self.hot_streak = 0
|
|
self.cool_streak = 0
|
|
self.last_change = 0.0
|
|
self.last_reason = "cold start"
|
|
self.history: List[Dict[str, Any]] = []
|
|
self._lock = threading.Lock()
|
|
self._base_profile: Optional[str] = None
|
|
|
|
# ---------------------------------------------------------------- observation
|
|
|
|
def observe(self, gpu: Dict[str, Any], active_profile: Optional[str] = None) -> None:
|
|
"""Feed one telemetry sample. Cheap and non-blocking; actuation runs in a thread."""
|
|
if not self.enabled or not gpu.get("available"):
|
|
return
|
|
temp = gpu.get("temperature_c") or 0
|
|
reasons = set(gpu.get("throttle_reasons") or [])
|
|
hard = bool(reasons & HARD_THROTTLES)
|
|
|
|
hot = temp >= TEMP_ESCALATE_C or hard
|
|
cool = temp <= TEMP_RECOVER_C and not hard
|
|
|
|
with self._lock:
|
|
if hot:
|
|
self.hot_streak += 1
|
|
self.cool_streak = 0
|
|
elif cool:
|
|
self.cool_streak += 1
|
|
self.hot_streak = 0
|
|
else:
|
|
self.hot_streak = 0
|
|
self.cool_streak = 0
|
|
|
|
now = time.time()
|
|
if now - self.last_change < REAPPLY_COOLDOWN_S:
|
|
return
|
|
|
|
if self.hot_streak >= HOT_SAMPLES and self.level < len(DERATE_LADDER) - 1:
|
|
why = (f"{temp:.0f}°C" if temp >= TEMP_ESCALATE_C else "") + \
|
|
(f" throttling: {','.join(sorted(reasons & HARD_THROTTLES))}" if hard else "")
|
|
self._step(self.level + 1, why.strip(), active_profile, temp)
|
|
elif self.cool_streak >= COOL_SAMPLES and self.level > 0:
|
|
self._step(self.level - 1, f"stable at {temp:.0f}°C", active_profile, temp)
|
|
|
|
def _step(self, new_level: int, reason: str, active_profile: Optional[str],
|
|
temp: float) -> None:
|
|
old = self.level
|
|
self.level = new_level
|
|
self.hot_streak = self.cool_streak = 0
|
|
self.last_change = time.time()
|
|
self.last_reason = reason
|
|
profile = active_profile or overclock_manager.ACTIVE_PROFILE
|
|
self._base_profile = profile
|
|
entry = {
|
|
"ts": time.time(), "from_level": old, "to_level": new_level,
|
|
"label": DERATE_LADDER[new_level]["label"], "reason": reason,
|
|
"temp_c": temp, "profile": profile,
|
|
}
|
|
self.history.insert(0, entry)
|
|
del self.history[50:]
|
|
direction = "escalating" if new_level > old else "recovering"
|
|
logger.warning(f"Thermal governor {direction} to level {new_level} "
|
|
f"({DERATE_LADDER[new_level]['label']}) — {reason}")
|
|
threading.Thread(target=self._actuate, args=(profile,), daemon=True).start()
|
|
|
|
def _actuate(self, profile: str) -> None:
|
|
try:
|
|
overclock_manager.apply_profile(profile, overrides=self.overrides_for(profile))
|
|
except Exception as e:
|
|
logger.error(f"Governor failed to apply derate: {e}")
|
|
|
|
# ---------------------------------------------------------------- overrides
|
|
|
|
def overrides_for(self, profile: str) -> Dict[str, Any]:
|
|
"""Translate the current derate level into profile overrides."""
|
|
step = DERATE_LADDER[self.level]
|
|
if self.level == 0:
|
|
return {}
|
|
cfg = overclock_manager.load_profiles().get(profile, {})
|
|
scale = step["offset_scale"]
|
|
overrides: Dict[str, Any] = {
|
|
"core_offset_mhz": int(cfg.get("core_offset_mhz", 0) * scale),
|
|
"mem_offset_mhz": int(cfg.get("mem_offset_mhz", 0) * scale),
|
|
}
|
|
if step["level"] >= 2:
|
|
# Stop pinning the core clock high when the card is already backing off.
|
|
overrides["lock_core_min"] = 0
|
|
overrides["lock_core_max"] = 0
|
|
floor = step["fan_floor"]
|
|
if floor:
|
|
overrides["fan_mode"] = "manual"
|
|
overrides["fan_speed_pct"] = max(int(cfg.get("fan_speed_pct", 0)), floor)
|
|
return overrides
|
|
|
|
# ---------------------------------------------------------------- control
|
|
|
|
def reset(self) -> Dict[str, Any]:
|
|
with self._lock:
|
|
self.level = 0
|
|
self.hot_streak = self.cool_streak = 0
|
|
self.last_change = time.time()
|
|
self.last_reason = "manual reset"
|
|
profile = self._base_profile or overclock_manager.ACTIVE_PROFILE
|
|
if profile and profile != "stock":
|
|
overclock_manager.apply_profile(profile)
|
|
return self.get_status()
|
|
|
|
def set_enabled(self, enabled: bool) -> Dict[str, Any]:
|
|
self.enabled = bool(enabled)
|
|
if not enabled and self.level > 0:
|
|
self.reset()
|
|
return self.get_status()
|
|
|
|
def get_status(self) -> Dict[str, Any]:
|
|
step = DERATE_LADDER[self.level]
|
|
return {
|
|
"enabled": self.enabled,
|
|
"level": self.level,
|
|
"label": step["label"],
|
|
"offset_scale": step["offset_scale"],
|
|
"fan_floor": step["fan_floor"],
|
|
"last_reason": self.last_reason,
|
|
"hot_streak": self.hot_streak,
|
|
"cool_streak": self.cool_streak,
|
|
"escalate_at_c": TEMP_ESCALATE_C,
|
|
"recover_below_c": TEMP_RECOVER_C,
|
|
"seconds_since_change": round(time.time() - self.last_change, 1) if self.last_change else None,
|
|
"history": self.history[:10],
|
|
}
|
|
|
|
|
|
governor = ThermalGovernor()
|