Add barrier-confirmed yielding, measured residency, persistence and closed-loop tuning
Nine changes, in rough order of how much they affect real behaviour: 1. VRAM yield is now a barrier. Posting keep_alive:0 only asks Ollama to unload; measured here, the HTTP call returns in 63ms while the driver takes a further 77ms to release 14.9GB. Returning inside that window is how ComfyUI ends up allocating into VRAM that is still occupied. instant_free_ollama_vram() polls NVML until the allocation is actually gone and reports request/confirm split. 2. ComfyUI VRAM is no longer purged 1.5s after every prompt, which forced a full checkpoint reload on each workflow iteration. It is held for 30s of genuinely empty queue, with an immediate purge when Ollama actually asks for the memory. 3. Cache-hit classification uses achieved bandwidth (size / load duration) rather than a fixed `load_duration < 2500ms`. That constant called a 12.9GB model read at 2.9GB/s a cold load, and a 0.5GB model read from NVMe a cache hit. 4. Page-cache residency is measured, not assumed. mincore(2) reported 128GB resident on a box with 46GB of page cache: the kernel only permits page-cache introspection on files you own, and the Ollama blobs are owned by uid ollama, for which mincore answers "all resident" instead of failing. Uses cachestat(2) where permitted and a randomised read-rate probe elsewhere, labelling which was used. Fixed-offset probing was self-fulfilling, so windows are random and cold ones are returned with FADV_DONTNEED. 5. Warming is budgeted and ranked by recency/frequency instead of reading every file top-to-bottom, which on 64GB of RAM just evicts whatever was warmed first. 6. Telemetry and events persist to SQLite (~0.38 MB/hour) instead of living in a 50-entry in-memory deque, so /api/analytics/profiles can finally answer whether an overclock profile actually delivers more tok/s. 7. Thermal governor walks the overclock back on sustained heat or hardware throttling, with hysteresis, fed from the existing sampler. 8. Autotune sweeps a clock offset, benchmarks decode at each step, watches for Xid errors and degenerate output, and restores the profile in a finally block. 9. Stock clocks/power/fans are restored on shutdown and via systemd ExecStopPost. Nothing previously undid a locked clock or a manually pinned fan. Also: one shared 1Hz telemetry sampler fanned out to SSE subscribers rather than every client re-running the whole snapshot; wall-clock timestamps in place of the event loop's monotonic clock; cached nvidia-smi shell-outs; quieter httpx logging. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -207,14 +207,20 @@ def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def apply_profile(name: str) -> Dict[str, Any]:
|
||||
"""Apply a named overclock profile to the GPU. Returns a full result report."""
|
||||
def apply_profile(name: str, overrides: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
||||
"""Apply a named overclock profile to the GPU. Returns a full result report.
|
||||
|
||||
`overrides` lets the thermal governor and the autotuner apply a modified version of a
|
||||
profile (a derated offset, a probe clock) without mutating what is stored on disk.
|
||||
"""
|
||||
global ACTIVE_PROFILE, _LAST_RESULT
|
||||
profiles = load_profiles()
|
||||
if name not in profiles:
|
||||
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
|
||||
|
||||
cfg = profiles[name]
|
||||
cfg = dict(profiles[name])
|
||||
if overrides:
|
||||
cfg.update(overrides)
|
||||
fan_mode = cfg.get("fan_mode", "auto")
|
||||
fan_speed = int(cfg.get("fan_speed_pct", 0))
|
||||
|
||||
@@ -230,15 +236,31 @@ def apply_profile(name: str) -> Dict[str, Any]:
|
||||
}
|
||||
result["gpu"] = get_gpu_state()
|
||||
result["fan_status"] = get_fan_status()
|
||||
result["overrides"] = overrides or {}
|
||||
|
||||
_STATE_CACHE["value"] = None
|
||||
_FAN_CACHE["value"] = None
|
||||
ACTIVE_PROFILE = name
|
||||
_LAST_RESULT = result
|
||||
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
|
||||
return result
|
||||
|
||||
|
||||
def get_gpu_state() -> Dict[str, Any]:
|
||||
"""Read back live GPU clocks/power/limits via nvidia-smi."""
|
||||
_STATE_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
||||
_FAN_CACHE: Dict[str, Any] = {"ts": 0.0, "value": None}
|
||||
STATE_TTL_S = 2.0
|
||||
|
||||
|
||||
def get_gpu_state(force: bool = False) -> Dict[str, Any]:
|
||||
"""Read back live GPU clocks/power/limits via nvidia-smi.
|
||||
|
||||
Cached for STATE_TTL_S: this forks `sudo nvidia-smi`, and the dashboard polls the
|
||||
status endpoint every few seconds. NVML already covers the live 1 Hz telemetry.
|
||||
"""
|
||||
import time as _time
|
||||
if not force and _STATE_CACHE["value"] is not None and \
|
||||
(_time.time() - _STATE_CACHE["ts"]) < STATE_TTL_S:
|
||||
return _STATE_CACHE["value"]
|
||||
state: Dict[str, Any] = {}
|
||||
r = _smi(
|
||||
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
|
||||
@@ -257,6 +279,7 @@ def get_gpu_state() -> Dict[str, Any]:
|
||||
state[k] = float(parts[i])
|
||||
except ValueError:
|
||||
state[k] = parts[i]
|
||||
_STATE_CACHE.update({"ts": __import__("time").time(), "value": state})
|
||||
return state
|
||||
|
||||
|
||||
@@ -298,11 +321,16 @@ def set_fan_auto() -> Dict[str, Any]:
|
||||
ok = r["rc"] == 0
|
||||
if ok:
|
||||
FAN_MANUAL = False
|
||||
_FAN_CACHE["value"] = None
|
||||
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
|
||||
|
||||
|
||||
def get_fan_status() -> Dict[str, Any]:
|
||||
"""Read current fan control mode + target speed."""
|
||||
def get_fan_status(force: bool = False) -> Dict[str, Any]:
|
||||
"""Read current fan control mode + target speed (cached; forks nvidia-settings)."""
|
||||
import time as _time
|
||||
if not force and _FAN_CACHE["value"] is not None and \
|
||||
(_time.time() - _FAN_CACHE["ts"]) < STATE_TTL_S:
|
||||
return _FAN_CACHE["value"]
|
||||
global FAN_MANUAL
|
||||
target = None
|
||||
manual = FAN_MANUAL
|
||||
@@ -320,7 +348,34 @@ def get_fan_status() -> Dict[str, Any]:
|
||||
target = int(line.split("):")[-1].split(".")[0].strip())
|
||||
except Exception:
|
||||
pass
|
||||
return {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
|
||||
result = {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
|
||||
_FAN_CACHE.update({"ts": __import__("time").time(), "value": result})
|
||||
return result
|
||||
|
||||
|
||||
def restore_safe(reason: str = "shutdown") -> Dict[str, Any]:
|
||||
"""Return the card to stock: no clock locks, no offsets, default power, automatic fans.
|
||||
|
||||
This matters because every lever here is sticky. If the service dies while a profile is
|
||||
applied, the GPU keeps the locked clocks and, worse, keeps the fans pinned at whatever
|
||||
manual PWM was last set. Nothing was undoing that.
|
||||
"""
|
||||
logger.warning(f"Restoring GPU to safe stock state ({reason})")
|
||||
result = {
|
||||
"reason": reason,
|
||||
"clock_lock": _apply_clock_lock(0, 0),
|
||||
"mem_lock": _apply_mem_lock(0),
|
||||
"offsets": _apply_offsets(0, 0),
|
||||
"fan": set_fan_auto(),
|
||||
}
|
||||
# Hand the power limit back to the card's own default rather than assuming 370 W.
|
||||
state = get_gpu_state()
|
||||
default_w = state.get("power_default_w")
|
||||
if isinstance(default_w, (int, float)) and default_w > 0:
|
||||
result["power_limit"] = _apply_power_limit(int(default_w))
|
||||
global ACTIVE_PROFILE
|
||||
ACTIVE_PROFILE = "stock"
|
||||
return result
|
||||
|
||||
|
||||
def get_status() -> Dict[str, Any]:
|
||||
|
||||
Reference in New Issue
Block a user