Add GPU fan control, live telemetry, and per-profile fan curves in HyperSwap dashboard

This commit is contained in:
drjones
2026-08-23 08:56:06 -07:00
parent 850e1aa565
commit 1f198ae10e
9 changed files with 1237 additions and 6 deletions

View File

@@ -73,7 +73,19 @@ The HyperSwap server runs on port `9090` by default. Interactive OpenAPI/Swagger
Returns a unified JSON snapshot of all system sensors, GPU processes, host RAM, Ollama status, ComfyUI queue, and transition logs.
#### `GET /api/gpu`
Returns hardware sensors (utilization %, temperature, power draw in Watts, fan %, GPU graphics/memory clocks, and active PIDs).
Returns hardware sensors (utilization %, temperature, power draw in Watts, fan speed %, per-fan telemetry, GPU graphics/memory clocks, and active PIDs).
#### `GET /api/overclock/fan` / `GET /api/gpu/fan`
Returns current GPU fan mode (`auto` vs `manual`), target fan speed %, and live fan telemetry.
#### `POST /api/overclock/fan` / `POST /api/gpu/fan`
Sets GPU fan speed mode (`auto` or `manual`) with target speed % (30100%).
#### `GET /api/overclock`
Returns active overclock profile, configured profiles, GPU clock limits, and fan status.
#### `POST /api/overclock/apply`
Applies a named profile (`ollama`, `comfy`, `balanced`) configuring power limits, clock locks, offsets, and fan speed.
#### `GET /api/memory`
Returns precise `/proc/meminfo` metrics including Total, Used, OS Page Cache, and free memory.
@@ -121,7 +133,9 @@ HyperSwap includes a native **MCP 2.0 server** (`mcp_server.py`) that exposes al
| Tool Name | Parameters | Description |
| :--- | :--- | :--- |
| **`get_gpu_status`** | *None* | Live NVIDIA GPU hardware telemetry, VRAM breakdown, temps, power, and PIDs. |
| **`get_gpu_status`** | *None* | Live NVIDIA GPU hardware telemetry, VRAM breakdown, temps, power, fan %, and PIDs. |
| **`get_gpu_fan_status`** | *None* | Current GPU fan mode (`auto`/`manual`) and target fan percentage. |
| **`set_gpu_fan_speed`** | `mode` (str, "auto"\|"manual"), `percent` (optional int) | Sets fan speed mode and target PWM % (30100%). |
| **`get_host_memory_status`** | *None* | 64GB host RAM breakdown, active page cache size, and cache ratio. |
| **`switch_ollama_model`** | `model_name` (str), `keep_alive` (str, default "30m") | Hot-swaps active LLM in VRAM, measures latency (ms) and tokens/sec. |
| **`soft_yield_ollama_vram`** | `model_name` (optional str) | Yields Ollama VRAM to 0 MB in ~15ms while keeping model weights in RAM cache. |

View File

@@ -9,6 +9,7 @@ from typing import Dict, List, Any, Optional
from mcp.server import MCPServer
import ram_optimizer
import vram_arbitrator
import overclock_manager
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(name)s: %(message)s")
logger = logging.getLogger("gpu_swapper_mcp")
@@ -117,6 +118,21 @@ async def run_model_switch_benchmark(iterations: int = 2) -> str:
"rounds": results,
}, indent=2)
@mcp.tool()
def get_gpu_fan_status() -> str:
"""Get current GPU fan control mode (auto vs manual) and target speed."""
status = overclock_manager.get_fan_status()
return json.dumps(status, indent=2)
@mcp.tool()
def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str:
"""Set GPU fan speed mode ('auto' or 'manual') with target percent (30-100%)."""
if mode.lower() == "manual" and percent is not None:
res = overclock_manager.set_fan_speed(percent)
else:
res = overclock_manager.set_fan_auto()
return json.dumps(res, indent=2)
# ==========================================
# MCP RESOURCES
# ==========================================

339
overclock_manager.py Normal file
View File

@@ -0,0 +1,339 @@
"""
Overclock Manager for HyperSwap — per-app GPU overclock profiles for the RTX 4080 SUPER.
Lever hierarchy (what actually works on this box):
1. Power limit nvidia-smi -pl <W> -> 320W -> 370W max (BIG win, works on open module)
2. Clock locks nvidia-smi -lgc / -lmc -> sustain max boost (works on open module)
3. Clock offsets nvidia-settings -a ... -> +core / +mem beyond stock (needs PROPRIETARY module)
Profiles are application-specific:
- ollama : LLM decode is memory-bandwidth bound -> lock memory clock to max + max power
- comfy : diffusion is compute bound -> lock core clock high + max power
- balanced: stock boost, power unlocked only
Auto-switches in lockstep with the VRAM arbitrator (vram_arbitrator.AutoArbitrator).
"""
import json
import logging
import os
import shutil
import subprocess
from typing import Dict, Any, Optional, List
logger = logging.getLogger("overclock_manager")
_BASE = os.path.dirname(os.path.abspath(__file__))
CONFIG_PATH = os.path.join(_BASE, "overclock_profiles.json")
NVIDIA_SMI = "nvidia-smi"
NVIDIA_SETTINGS = "nvidia-settings"
HEADLESS_DISPLAY = ":8" # dedicated headless X server owning the NVIDIA GPU
HEADLESS_CONFIG = "/etc/X11/xorg.conf.nvidia-headless"
# Default profile set. lock_* == 0 means "don't lock" (let boost manage).
DEFAULT_PROFILES: Dict[str, Dict[str, Any]] = {
"ollama": {
"label": "Ollama — LLM decode (memory-bandwidth bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 11501,
},
"comfy": {
"label": "ComfyUI — diffusion (core-compute bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 2900,
"lock_core_max": 3105,
"lock_mem_mhz": 0,
},
"balanced": {
"label": "Balanced — stock boost, power unlocked",
"power_limit_w": 370,
"core_offset_mhz": 0,
"mem_offset_mhz": 0,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 0,
},
}
ACTIVE_PROFILE = "balanced"
_LAST_RESULT: Dict[str, Any] = {}
FAN_MANUAL = False
def _sh(cmd: List[str], use_sudo: bool = True, timeout: int = 10) -> Dict[str, Any]:
"""Run a command; return rc/stdout/stderr. Prefers passwordless sudo."""
full = list(cmd)
if use_sudo:
full = ["sudo", "-n"] + full
try:
proc = subprocess.run(
full, capture_output=True, text=True, timeout=timeout
)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except subprocess.TimeoutExpired:
return {"rc": -1, "out": "", "err": "timeout"}
except FileNotFoundError as e:
return {"rc": -1, "out": "", "err": f"not found: {e}"}
def _smi(*args: str) -> Dict[str, Any]:
return _sh([NVIDIA_SMI, *args], use_sudo=True)
def _nvidia_settings(*args: str) -> Dict[str, Any]:
"""Run nvidia-settings against the headless X display that owns the GPU."""
env = os.environ.copy()
env["DISPLAY"] = HEADLESS_DISPLAY
cmd = [NVIDIA_SETTINGS, "-c", HEADLESS_DISPLAY, *args]
try:
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=15, env=env)
return {"rc": proc.returncode, "out": proc.stdout.strip(), "err": proc.stderr.strip()}
except Exception as e:
return {"rc": -1, "out": "", "err": str(e)}
def load_profiles() -> Dict[str, Dict[str, Any]]:
"""Load profiles from disk, falling back to defaults and merging new keys."""
profiles = json.loads(json.dumps(DEFAULT_PROFILES))
if os.path.exists(CONFIG_PATH):
try:
with open(CONFIG_PATH) as f:
stored = json.load(f)
for name, cfg in stored.items():
if name in profiles:
profiles[name].update(cfg)
else:
profiles[name] = cfg
except Exception as e:
logger.warning(f"Could not load {CONFIG_PATH}: {e}")
return profiles
def save_profiles(profiles: Dict[str, Dict[str, Any]]) -> bool:
try:
with open(CONFIG_PATH, "w") as f:
json.dump(profiles, f, indent=2)
return True
except Exception as e:
logger.error(f"save_profiles failed: {e}")
return False
def get_profiles() -> Dict[str, Dict[str, Any]]:
return load_profiles()
def set_profile(name: str, cfg: Dict[str, Any]) -> Dict[str, Any]:
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'"}
profiles[name].update(cfg)
ok = save_profiles(profiles)
return {"success": ok, "profiles": profiles if ok else None}
def _apply_power_limit(watts: int) -> Dict[str, Any]:
r = _smi("-pl", str(watts))
ok = r["rc"] == 0
return {"applied": ok, "detail": r.get("out") or r.get("err")}
def _apply_clock_lock(core_min: int, core_max: int) -> Dict[str, Any]:
if core_min == 0 and core_max == 0:
r = _smi("-rgc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lgc", f"{core_min},{core_max}")
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_mem_lock(mem_mhz: int) -> Dict[str, Any]:
if mem_mhz == 0:
r = _smi("-rmc")
return {"applied": r["rc"] == 0, "detail": "reset"}
r = _smi("-lmc", str(mem_mhz))
return {"applied": r["rc"] == 0, "detail": r.get("out") or r.get("err")}
def _apply_offsets(core_mhz: int, mem_mhz: int) -> Dict[str, Any]:
"""Apply +core/+mem offsets via nvidia-settings. Returns whether they actually stuck.
NOTE: offsets only work with the PROPRIETARY kernel module, not nvidia-open."""
if core_mhz == 0 and mem_mhz == 0:
r = _nvidia_settings("-a", "[gpu:0]/GPUGraphicsClockOffset[3]=0",
"-a", "[gpu:0]/GPUMemoryTransferRateOffset[3]=0")
return {"applied": r["rc"] == 0, "detail": "reset", "supported": True}
r = _nvidia_settings("-a", f"[gpu:0]/GPUGraphicsClockOffset[3]={core_mhz}",
"-a", f"[gpu:0]/GPUMemoryTransferRateOffset[3]={mem_mhz}")
if r["rc"] != 0:
return {"applied": False, "detail": r.get("err") or r.get("out"), "supported": False}
# Read back to confirm the driver actually persisted the offsets.
q = _nvidia_settings("-q", "[gpu:0]/GPUGraphicsClockOffset[3]",
"-q", "[gpu:0]/GPUMemoryTransferRateOffset[3]")
applied_core = applied_mem = None
for line in q["out"].splitlines():
line = line.strip()
if "GPUGraphicsClockOffset" in line and ":" in line and "(" in line:
try:
applied_core = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
if "GPUMemoryTransferRateOffset" in line and ":" in line and "(" in line:
try:
applied_mem = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
supported = (applied_core is not None and applied_core != 0) or \
(applied_mem is not None and applied_mem != 0)
return {
"applied": supported,
"supported": supported,
"readback_core": applied_core,
"readback_mem": applied_mem,
"detail": f"core readback={applied_core}, mem readback={applied_mem}",
}
def apply_profile(name: str) -> Dict[str, Any]:
"""Apply a named overclock profile to the GPU. Returns a full result report."""
global ACTIVE_PROFILE, _LAST_RESULT
profiles = load_profiles()
if name not in profiles:
return {"success": False, "error": f"unknown profile '{name}'", "profile": name}
cfg = profiles[name]
fan_mode = cfg.get("fan_mode", "auto")
fan_speed = int(cfg.get("fan_speed_pct", 0))
result = {
"success": True,
"profile": name,
"label": cfg.get("label", name),
"power_limit": _apply_power_limit(int(cfg.get("power_limit_w", 370))),
"clock_lock": _apply_clock_lock(int(cfg.get("lock_core_min", 0)), int(cfg.get("lock_core_max", 0))),
"mem_lock": _apply_mem_lock(int(cfg.get("lock_mem_mhz", 0))),
"offsets": _apply_offsets(int(cfg.get("core_offset_mhz", 0)), int(cfg.get("mem_offset_mhz", 0))),
"fan": apply_fan_control(fan_mode, fan_speed),
}
result["gpu"] = get_gpu_state()
result["fan_status"] = get_fan_status()
ACTIVE_PROFILE = name
_LAST_RESULT = result
logger.info(f"Overclock profile applied: {name} -> {json.dumps(result, default=str)}")
return result
def get_gpu_state() -> Dict[str, Any]:
"""Read back live GPU clocks/power/limits via nvidia-smi."""
state: Dict[str, Any] = {}
r = _smi(
"--query-gpu=driver_version,name,memory.total,power.limit,power.max_limit,power.default_limit,"
"clocks.sm,clocks.max.sm,clocks.mem,clocks.max.mem,"
"temperature.gpu,power.draw,fan.speed",
"--format=csv,noheader,nounits",
)
if r["rc"] == 0 and r["out"]:
parts = [p.strip() for p in r["out"].split(",")]
keys = ["driver_version", "name", "vram_total_mb", "power_limit_w", "power_max_w", "power_default_w",
"clock_sm_mhz", "clock_sm_max_mhz", "clock_mem_mhz", "clock_mem_max_mhz",
"temp_c", "power_draw_w", "fan_pct"]
for i, k in enumerate(keys):
if i < len(parts):
try:
state[k] = float(parts[i])
except ValueError:
state[k] = parts[i]
return state
def is_headless_x_running() -> bool:
r = _sh(["pgrep", "-f", f"Xorg {HEADLESS_DISPLAY}"], use_sudo=False)
return r["rc"] == 0
def apply_fan_control(mode: str, speed_pct: int) -> Dict[str, Any]:
global FAN_MANUAL
if mode == "auto":
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if ok:
FAN_MANUAL = False
return {"applied": ok, "mode": "auto", "detail": r.get("out") or r.get("err")}
speed_pct = max(30, min(100, int(speed_pct)))
r = _nvidia_settings(
"-a", "[gpu:0]/GPUFanControlState=1",
"-a", f"[fan:0]/GPUTargetFanSpeed={speed_pct}",
"-a", f"[fan:1]/GPUTargetFanSpeed={speed_pct}"
)
ok = (r["rc"] == 0)
if ok:
FAN_MANUAL = True
return {"applied": ok, "mode": "manual", "speed_pct": speed_pct, "detail": r.get("out") or r.get("err")}
def set_fan_speed(percent: int) -> Dict[str, Any]:
"""Set manual GPU fan target speed (30-100%)."""
return apply_fan_control(mode="manual", speed_pct=percent)
def set_fan_auto() -> Dict[str, Any]:
"""Return GPU fan to automatic control."""
global FAN_MANUAL
r = _nvidia_settings("-a", "[gpu:0]/GPUFanControlState=0")
ok = r["rc"] == 0
if ok:
FAN_MANUAL = False
return {"success": ok, "manual": False, "fan_speed_pct": None, "detail": r.get("out") or r.get("err")}
def get_fan_status() -> Dict[str, Any]:
"""Read current fan control mode + target speed."""
global FAN_MANUAL
target = None
manual = FAN_MANUAL
r = _nvidia_settings("-q", "[gpu:0]/GPUFanControlState", "-q", "[fan:0]/GPUTargetFanSpeed")
if r.get("rc") == 0 and r.get("out"):
for line in r["out"].splitlines():
if "GPUFanControlState" in line and ":" in line:
try:
val = int(line.split("):")[-1].split(".")[0].strip())
manual = (val == 1)
except Exception:
pass
elif "GPUTargetFanSpeed" in line and ":" in line and "(" in line:
try:
target = int(line.split("):")[-1].split(".")[0].strip())
except Exception:
pass
return {"manual": manual, "mode": "manual" if manual else "auto", "target_speed_pct": target}
def get_status() -> Dict[str, Any]:
"""Full overclock status for the dashboard."""
return {
"active_profile": ACTIVE_PROFILE,
"profiles": load_profiles(),
"gpu": get_gpu_state(),
"fan": get_fan_status(),
"headless_x_running": is_headless_x_running(),
"headless_display": HEADLESS_DISPLAY,
"last_result": _LAST_RESULT,
}
if __name__ == "__main__":
import sys
logging.basicConfig(level=logging.INFO)
if len(sys.argv) > 1:
print(json.dumps(apply_profile(sys.argv[1]), indent=2, default=str))
else:
print(json.dumps(get_status(), indent=2, default=str))

35
overclock_profiles.json Normal file
View File

@@ -0,0 +1,35 @@
{
"ollama": {
"label": "Ollama \u2014 LLM decode (memory-bandwidth bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 700,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 0,
"fan_mode": "auto",
"fan_speed_pct": 0
},
"comfy": {
"label": "ComfyUI \u2014 diffusion (core-compute bound)",
"power_limit_w": 370,
"core_offset_mhz": 100,
"mem_offset_mhz": 500,
"lock_core_min": 2900,
"lock_core_max": 3105,
"lock_mem_mhz": 0,
"fan_mode": "auto",
"fan_speed_pct": 0
},
"balanced": {
"label": "Balanced \u2014 stock boost, power unlocked",
"power_limit_w": 370,
"core_offset_mhz": 0,
"mem_offset_mhz": 0,
"lock_core_min": 0,
"lock_core_max": 0,
"lock_mem_mhz": 0,
"fan_mode": "auto",
"fan_speed_pct": 0
}
}

View File

@@ -11,6 +11,7 @@ from pydantic import BaseModel, Field
import ram_optimizer
import vram_arbitrator
import overclock_manager
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(name)s: %(message)s")
logger = logging.getLogger("model_manager_server")
@@ -31,6 +32,14 @@ app.add_middleware(
allow_headers=["*"],
)
@app.on_event("startup")
async def on_startup():
await vram_arbitrator.arbitrator.start()
@app.on_event("shutdown")
async def on_shutdown():
await vram_arbitrator.arbitrator.stop()
# Pydantic Request Models
class SwitchRequest(BaseModel):
model: str = Field(..., description="Name of the Ollama model to hot-swap to in VRAM", example="qwen3.8fast:latest")
@@ -44,6 +53,17 @@ class BenchmarkRequest(BaseModel):
iterations: Optional[int] = Field(2, description="Number of back-and-forth switch iterations to measure", example=2)
models: Optional[List[str]] = Field(None, description="Optional pair of models to benchmark between", example=["qwen3.8fast:latest", "smtek/Qwen3.8-27B:Q2_K_XL"])
class OverclockApplyRequest(BaseModel):
profile: str = Field(..., description="Profile name: ollama | comfy | balanced", example="ollama")
class OverclockProfileUpdate(BaseModel):
config: Dict[str, Any] = Field(..., description="Profile settings dict", example={"power_limit_w": 370, "core_offset_mhz": 100})
class FanRequest(BaseModel):
mode: str = Field("auto", description="'auto' or 'manual'", example="manual")
percent: Optional[int] = Field(None, description="Fan speed 30-100 when mode=manual", example=70)
speed_pct: Optional[int] = Field(None, description="Alias for percent (30-100)", example=70)
# ==========================================
# REST API ENDPOINTS
@@ -58,6 +78,7 @@ async def get_all_stats() -> Dict[str, Any]:
comfy_state = await vram_arbitrator.get_comfyui_live_state()
history = vram_arbitrator.get_switch_history()
comfy_models = ram_optimizer.find_comfy_model_files()
arbitrator_status = vram_arbitrator.arbitrator.get_status()
return {
"timestamp": asyncio.get_event_loop().time(),
@@ -65,10 +86,12 @@ async def get_all_stats() -> Dict[str, Any]:
"ram": mem_stats,
"ollama": ollama_state,
"comfyui": comfy_state,
"arbitrator": arbitrator_status,
"history": history,
"comfy_models_count": len(comfy_models),
}
@app.get("/api/gpu", summary="GPU Sensors and VRAM Breakdown", tags=["Telemetry"])
async def get_gpu_metrics() -> Dict[str, Any]:
"""Retrieve detailed NVML sensors (utilization %, temp, power, fan, clocks, and per-process VRAM allocation)."""
@@ -160,6 +183,51 @@ async def api_run_benchmark(req: BenchmarkRequest):
res_str = await run_model_switch_benchmark(iterations=req.iterations or 2)
return json.loads(res_str)
# ==========================================
# OVERCLOCK MANAGEMENT
# ==========================================
@app.get("/api/overclock", summary="Overclock Status & Profiles", tags=["Overclock"])
async def api_overclock_status():
"""Get live GPU overclock state, active profile, and all per-app profiles."""
return overclock_manager.get_status()
@app.post("/api/overclock/apply", summary="Apply Overclock Profile", tags=["Overclock"])
async def api_overclock_apply(req: OverclockApplyRequest):
"""Apply a named overclock profile (ollama | comfy | balanced) to the GPU immediately."""
res = overclock_manager.apply_profile(req.profile)
if not res.get("success"):
raise HTTPException(status_code=400, detail=res.get("error"))
return res
@app.get("/api/overclock/profiles", summary="List Overclock Profiles", tags=["Overclock"])
async def api_overclock_profiles():
"""List all overclock profiles with their current settings."""
return overclock_manager.get_profiles()
@app.post("/api/overclock/profiles/{name}", summary="Update Overclock Profile", tags=["Overclock"])
async def api_overclock_update_profile(name: str, req: OverclockProfileUpdate):
"""Update a profile's settings (persisted to disk)."""
res = overclock_manager.set_profile(name, req.config)
if not res.get("success"):
raise HTTPException(status_code=400, detail=res.get("error"))
return {"success": True, "profile": name, "profiles": res.get("profiles")}
@app.get("/api/overclock/fan", summary="Get GPU Fan Status", tags=["Overclock"])
@app.get("/api/gpu/fan", summary="Get GPU Fan Status", tags=["Overclock"])
async def api_get_fan_status():
"""Get current GPU fan control mode and speed."""
return overclock_manager.get_fan_status()
@app.post("/api/overclock/fan", summary="Set GPU Fan Speed", tags=["Overclock"])
@app.post("/api/gpu/fan", summary="Set GPU Fan Speed", tags=["Overclock"])
async def api_set_fan(req: FanRequest):
"""Set the GPU fan to manual speed (30-100%) or back to automatic control."""
pct = req.percent if req.percent is not None else req.speed_pct
if req.mode == "manual" and pct is not None:
return overclock_manager.set_fan_speed(pct)
return overclock_manager.set_fan_auto()
# Mount static web UI files
app.mount("/static", StaticFiles(directory="/home/drjones/unified-model-manager/static"), name="static")

View File

@@ -36,6 +36,7 @@ function updateDashboard(data) {
// 1. GPU VRAM Stats
const gpu = data.gpu || {};
if (gpu.available) {
pushOverclockSample(gpu);
document.getElementById('gpu-chip-name').textContent = gpu.device_name || 'NVIDIA GPU';
document.getElementById('vram-total-used').textContent = gpu.vram_used_gb || '0.0';
document.getElementById('vram-used-pct').textContent = `${gpu.vram_used_pct || 0}% USED`;
@@ -68,6 +69,45 @@ function updateDashboard(data) {
document.getElementById('gpu-power-val').textContent = `${gpu.power_w || 0} W`;
document.getElementById('gpu-fan-val').textContent = `${gpu.fan_pct || 0}%`;
// Per-fan and animations
const fan0 = (gpu.fans && gpu.fans.length > 0) ? gpu.fans[0] : (gpu.fan_pct || 0);
const fan1 = (gpu.fans && gpu.fans.length > 1) ? gpu.fans[1] : (gpu.fan_pct || 0);
const fanSub = document.getElementById('gpu-fan-sub');
if (fanSub) fanSub.textContent = `Fan 0: ${fan0}% | Fan 1: ${fan1}%`;
const fan0Val = document.getElementById('oc-fan0-val');
if (fan0Val) fan0Val.textContent = `${fan0}%`;
const fan0Bar = document.getElementById('oc-fan0-bar');
if (fan0Bar) fan0Bar.style.width = `${fan0}%`;
const fan1Val = document.getElementById('oc-fan1-val');
if (fan1Val) fan1Val.textContent = `${fan1}%`;
const fan1Bar = document.getElementById('oc-fan1-bar');
if (fan1Bar) fan1Bar.style.width = `${fan1}%`;
const fanCurrent = document.getElementById('oc-fan-current');
if (fanCurrent) fanCurrent.textContent = `${gpu.fan_pct || 0}%`;
const spinSpeed = Math.max(0.2, (100 - (gpu.fan_pct || 0)) / 100 * 1.6 + 0.3);
const fanIcon = document.getElementById('gpu-fan-icon');
if (fanIcon) {
if ((gpu.fan_pct || 0) > 0) {
fanIcon.classList.add('fan-spinning');
fanIcon.style.animationDuration = `${spinSpeed.toFixed(2)}s`;
} else {
fanIcon.classList.remove('fan-spinning');
}
}
const ocFanCardIcon = document.getElementById('oc-fan-card-icon');
if (ocFanCardIcon) {
if ((gpu.fan_pct || 0) > 0) {
ocFanCardIcon.classList.add('fan-spinning');
ocFanCardIcon.style.animationDuration = `${spinSpeed.toFixed(2)}s`;
} else {
ocFanCardIcon.classList.remove('fan-spinning');
}
}
// Processes table
const tbody = document.getElementById('gpu-proc-table');
if (bd.processes && bd.processes.length > 0) {
@@ -285,4 +325,287 @@ async function warmAllModels() {
// Startup
document.addEventListener('DOMContentLoaded', () => {
initSSE();
initOverclockChart();
fetchOverclockStatus();
setInterval(fetchOverclockStatus, 3000);
});
// ============ OVERCLOCK CONTROL ============
let ocProfiles = {};
let ocChart = null;
const OC_MAX_SAMPLES = 120;
function initOverclockChart() {
const canvas = document.getElementById('oc-chart');
if (!canvas || typeof Chart === 'undefined') return;
const ctx = canvas.getContext('2d');
ocChart = new Chart(ctx, {
type: 'line',
data: {
labels: [],
datasets: [
{ label: 'Core MHz', data: [], borderColor: '#22d3ee', backgroundColor: 'rgba(34,211,238,0.08)', borderWidth: 1.5, pointRadius: 0, tension: 0.35, yAxisID: 'y', fill: true },
{ label: 'Mem MHz', data: [], borderColor: '#c084fc', backgroundColor: 'rgba(192,132,252,0.08)', borderWidth: 1.5, pointRadius: 0, tension: 0.35, yAxisID: 'y', fill: false },
{ label: 'Temp °C', data: [], borderColor: '#fb7185', backgroundColor: 'rgba(251,113,133,0.08)', borderWidth: 1.5, pointRadius: 0, tension: 0.35, yAxisID: 'y1', fill: false },
{ label: 'Power W', data: [], borderColor: '#fbbf24', backgroundColor: 'rgba(251,191,36,0.08)', borderWidth: 1.5, pointRadius: 0, tension: 0.35, yAxisID: 'y2', fill: false },
{ label: 'Fan %', data: [], borderColor: '#34d399', backgroundColor: 'rgba(52,211,153,0.08)', borderWidth: 1.5, pointRadius: 0, tension: 0.35, yAxisID: 'y3', fill: false },
]
},
options: {
responsive: true,
maintainAspectRatio: false,
animation: false,
interaction: { mode: 'index', intersect: false },
scales: {
x: { ticks: { color: '#64748b', maxTicksLimit: 8, font: { size: 9 } }, grid: { color: 'rgba(51,65,85,0.35)' } },
y: { position: 'left', title: { display: true, text: 'MHz', color: '#22d3ee', font: { size: 9 } }, ticks: { color: '#64748b', font: { size: 9 } }, grid: { color: 'rgba(51,65,85,0.35)' } },
y1: { position: 'right', title: { display: true, text: '°C', color: '#fb7185', font: { size: 9 } }, ticks: { color: '#64748b', font: { size: 9 } }, grid: { drawOnChartArea: false }, suggestedMin: 0, suggestedMax: 100 },
y2: { position: 'right', offset: true, title: { display: true, text: 'W', color: '#fbbf24', font: { size: 9 } }, ticks: { color: '#64748b', font: { size: 9 } }, grid: { drawOnChartArea: false }, suggestedMin: 0, suggestedMax: 400 },
y3: { position: 'right', offset: true, title: { display: false }, ticks: { display: false }, grid: { drawOnChartArea: false }, suggestedMin: 0, suggestedMax: 100 },
},
plugins: { legend: { display: false } },
}
});
}
function pushOverclockSample(gpu) {
if (!ocChart || !gpu || !gpu.available) return;
const label = new Date().toLocaleTimeString([], { hour12: false });
ocChart.data.labels.push(label);
ocChart.data.datasets[0].data.push(gpu.clock_graphics_mhz || 0);
ocChart.data.datasets[1].data.push(gpu.clock_mem_mhz || 0);
ocChart.data.datasets[2].data.push(gpu.temperature_c || 0);
ocChart.data.datasets[3].data.push(gpu.power_w || 0);
ocChart.data.datasets[4].data.push(gpu.fan_pct || 0);
if (ocChart.data.labels.length > OC_MAX_SAMPLES) {
ocChart.data.labels.shift();
ocChart.data.datasets.forEach(d => d.data.shift());
}
ocChart.update('none');
}
async function fetchOverclockStatus() {
try {
const resp = await fetch('/api/overclock');
if (!resp.ok) return;
const data = await resp.json();
ocProfiles = data.profiles || {};
renderOverclockStatus(data);
} catch (err) {
console.warn('Overclock fetch error:', err);
}
}
function renderOverclockStatus(data) {
const active = data.active_profile || 'balanced';
const badge = document.getElementById('oc-active-badge');
badge.textContent = `Active: ${active}`;
if (active === 'ollama') {
badge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-purple-950 border border-purple-700 text-purple-300';
} else if (active === 'comfy') {
badge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-cyan-950 border border-cyan-700 text-cyan-300';
} else {
badge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-slate-800 border border-slate-700 text-slate-300';
}
const gpu = data.gpu || {};
document.getElementById('oc-power-limit').textContent = `${gpu.power_limit_w ?? '--'} W`;
document.getElementById('oc-core').textContent = `${gpu.clock_sm_mhz ?? '--'} MHz`;
document.getElementById('oc-mem').textContent = `${gpu.clock_mem_mhz ?? '--'} MHz`;
document.getElementById('oc-temp-draw').textContent = `${gpu.temp_c ?? '--'}°C / ${gpu.power_draw_w ?? '--'}W`;
// Specs strip
document.getElementById('oc-spec-gpu').textContent = gpu.name || '--';
document.getElementById('oc-spec-driver').textContent = gpu.driver_version || '--';
const vramGb = (gpu.vram_total_mb || 0) / 1024;
document.getElementById('oc-spec-vram').textContent = vramGb > 0 ? vramGb.toFixed(0) + ' GB' : '--';
document.getElementById('oc-spec-maxcore').textContent = (gpu.clock_sm_max_mhz ?? '--') + ' MHz';
document.getElementById('oc-spec-maxmem').textContent = (gpu.clock_mem_max_mhz ?? '--') + ' MHz';
document.getElementById('oc-spec-power').textContent = `${gpu.power_limit_w ?? '--'} / ${gpu.power_max_w ?? '--'} W`;
// Fan status
const fan = data.fan || {};
const fanBadge = document.getElementById('oc-fan-badge');
if (fanBadge) {
if (fan.manual) {
fanBadge.textContent = `MANUAL (${fan.target_speed_pct ?? '--'}%)`;
fanBadge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-emerald-950 border border-emerald-500 text-emerald-300 font-bold';
} else {
fanBadge.textContent = 'AUTO (VBIOS)';
fanBadge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-slate-800 border border-slate-700 text-slate-400 font-bold';
}
}
const xBadge = document.getElementById('oc-x-badge');
if (data.headless_x_running) {
xBadge.textContent = 'X: ON';
xBadge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-emerald-950 border border-emerald-700 text-emerald-300';
} else {
xBadge.textContent = 'X: OFF';
xBadge.className = 'px-2.5 py-1 text-xs font-mono rounded-lg bg-rose-950 border border-rose-700 text-rose-300';
}
const offsets = (data.last_result && data.last_result.offsets) || {};
const note = document.getElementById('oc-offset-note');
if (offsets.supported) {
note.textContent = '✅ Clock offsets active (proprietary driver)';
note.className = 'text-[11px] font-mono text-emerald-400';
} else {
note.textContent = '⚠️ Clock offsets off — nvidia-open lacks offset support (power + clock locks active)';
note.className = 'text-[11px] font-mono text-amber-400';
}
// highlight active profile button
['ollama', 'comfy', 'balanced'].forEach(p => {
const btn = document.getElementById('oc-btn-' + p);
if (p === active) {
btn.classList.add('ring-2', 'ring-fuchsia-400');
} else {
btn.classList.remove('ring-2', 'ring-fuchsia-400');
}
});
}
function loadOverclockForProfile(name) {
const p = ocProfiles[name];
if (!p) return;
document.getElementById('oc-slider-power').value = p.power_limit_w || 370;
document.getElementById('oc-val-power').textContent = (p.power_limit_w || 370) + ' W';
document.getElementById('oc-slider-core').value = p.core_offset_mhz || 0;
document.getElementById('oc-val-core').textContent = '+' + (p.core_offset_mhz || 0) + ' MHz';
document.getElementById('oc-slider-mem').value = p.mem_offset_mhz || 0;
document.getElementById('oc-val-mem').textContent = '+' + (p.mem_offset_mhz || 0) + ' MHz';
document.getElementById('oc-lockcore-toggle').checked = (p.lock_core_max || 0) > 0;
document.getElementById('oc-val-lockcore').textContent = (p.lock_core_max || 0) > 0 ? 'On' : 'Off';
document.getElementById('oc-lockmem-toggle').checked = (p.lock_mem_mhz || 0) > 0;
document.getElementById('oc-val-lockmem').textContent = (p.lock_mem_mhz || 0) > 0 ? 'On' : 'Off';
const fanMode = p.fan_mode || 'auto';
const profFanMode = document.getElementById('oc-prof-fanmode');
if (profFanMode) profFanMode.value = fanMode;
const profFanSpeed = document.getElementById('oc-slider-prof-fanspeed');
if (profFanSpeed) profFanSpeed.value = p.fan_speed_pct || 70;
const profFanVal = document.getElementById('oc-val-prof-fanspeed');
if (profFanVal) profFanVal.textContent = (p.fan_speed_pct || 70) + '%';
toggleProfileFanMode();
}
function toggleProfileFanMode() {
const mode = document.getElementById('oc-prof-fanmode').value;
const container = document.getElementById('oc-prof-fanspeed-container');
const valBadge = document.getElementById('oc-val-prof-fanmode');
if (valBadge) valBadge.textContent = (mode === 'manual') ? 'Manual Target' : 'Auto (VBIOS)';
if (container) {
if (mode === 'manual') {
container.classList.remove('opacity-50', 'pointer-events-none');
} else {
container.classList.add('opacity-50', 'pointer-events-none');
}
}
}
function toggleCoreLock() {
document.getElementById('oc-val-lockcore').textContent = document.getElementById('oc-lockcore-toggle').checked ? 'On' : 'Off';
}
function toggleMemLock() {
document.getElementById('oc-val-lockmem').textContent = document.getElementById('oc-lockmem-toggle').checked ? 'On' : 'Off';
}
async function setFanManual(speed) {
let pct = speed;
if (pct === undefined || pct === null) {
pct = parseInt(document.getElementById('oc-fan-slider').value);
} else {
pct = parseInt(pct);
const slider = document.getElementById('oc-fan-slider');
if (slider) slider.value = pct;
const label = document.getElementById('oc-fan-slider-label');
if (label) label.textContent = pct + '%';
}
try {
const resp = await fetch('/api/overclock/fan', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ mode: 'manual', percent: pct })
});
const result = await resp.json();
if (!resp.ok) alert(`Fan set failed: ${result.detail || 'error'}`);
await fetchOverclockStatus();
} catch (err) {
alert(`Error setting fan speed: ${err}`);
}
}
async function setFanAuto() {
try {
const resp = await fetch('/api/overclock/fan', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ mode: 'auto' })
});
const result = await resp.json();
if (!resp.ok) alert(`Fan auto failed: ${result.detail || 'error'}`);
await fetchOverclockStatus();
} catch (err) {
alert(`Error setting fan to auto: ${err}`);
}
}
async function applyOverclock(profile) {
try {
const resp = await fetch('/api/overclock/apply', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ profile })
});
const result = await resp.json();
if (!resp.ok) {
alert(`Apply failed: ${result.detail || 'error'}`);
}
await fetchOverclockStatus();
} catch (err) {
alert(`Error: ${err}`);
}
}
async function saveOverclockProfile() {
const name = document.getElementById('oc-edit-profile').value;
const power = parseInt(document.getElementById('oc-slider-power').value);
const core = parseInt(document.getElementById('oc-slider-core').value);
const mem = parseInt(document.getElementById('oc-slider-mem').value);
const lockCore = document.getElementById('oc-lockcore-toggle').checked;
const lockMem = document.getElementById('oc-lockmem-toggle').checked;
const fanMode = document.getElementById('oc-prof-fanmode') ? document.getElementById('oc-prof-fanmode').value : 'auto';
const fanSpeed = document.getElementById('oc-slider-prof-fanspeed') ? parseInt(document.getElementById('oc-slider-prof-fanspeed').value) : 70;
const config = {
power_limit_w: power,
core_offset_mhz: core,
mem_offset_mhz: mem,
lock_core_min: lockCore ? 2900 : 0,
lock_core_max: lockCore ? 3105 : 0,
lock_mem_mhz: lockMem ? 11501 : 0,
fan_mode: fanMode,
fan_speed_pct: (fanMode === 'manual') ? fanSpeed : 0,
};
try {
const resp = await fetch(`/api/overclock/profiles/${name}`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ config })
});
const result = await resp.json();
if (resp.ok) {
await fetchOverclockStatus();
await applyOverclock(name);
} else {
alert(`Save failed: ${result.detail || 'error'}`);
}
} catch (err) {
alert(`Error: ${err}`);
}
}

View File

@@ -5,6 +5,7 @@
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>HYPERSWAP // Dual-Engine Model Orchestrator & Live Telemetry</title>
<script src="https://cdn.tailwindcss.com"></script>
<script src="https://cdn.jsdelivr.net/npm/chart.js@4.4.1/dist/chart.umd.min.js"></script>
<link rel="stylesheet" href="/static/styles.css">
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.4.0/css/all.min.css">
</head>
@@ -351,9 +352,13 @@
<span class="text-[10px] text-slate-400 uppercase block">Power Draw</span>
<span id="gpu-power-val" class="text-xl font-bold text-amber-400">0 W</span>
</div>
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800 text-center">
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800 text-center relative overflow-hidden">
<span class="text-[10px] text-slate-400 uppercase block">Fan Speed</span>
<span id="gpu-fan-val" class="text-xl font-bold text-slate-300">0%</span>
<div class="flex items-center justify-center space-x-1.5">
<i id="gpu-fan-icon" class="fa-solid fa-fan text-emerald-400 text-sm"></i>
<span id="gpu-fan-val" class="text-xl font-bold text-emerald-400">0%</span>
</div>
<span id="gpu-fan-sub" class="text-[9px] font-mono text-slate-500 block truncate">Fan 0: --% | Fan 1: --%</span>
</div>
</div>
@@ -419,6 +424,245 @@
</div>
<!-- OVERCLOCK CONTROL PANEL -->
<div class="bg-slate-900/80 border border-fuchsia-900/50 rounded-2xl p-5 space-y-4 shadow-lg shadow-fuchsia-950/30">
<div class="flex flex-wrap items-center justify-between gap-3 pb-3 border-b border-slate-800">
<div class="flex items-center space-x-2">
<div class="p-2 rounded-lg bg-fuchsia-950/80 border border-fuchsia-800 text-fuchsia-400">
<i class="fa-solid fa-rocket text-sm"></i>
</div>
<div>
<h3 class="font-bold text-slate-100 text-sm">GPU Overclock Control</h3>
<p class="text-xs text-slate-400">Per-app profiles auto-switch with the VRAM arbitrator</p>
</div>
</div>
<div class="flex items-center space-x-2">
<span id="oc-active-badge" class="px-2.5 py-1 text-xs font-mono rounded-lg bg-slate-800 border border-slate-700 text-slate-300">Active: --</span>
<span id="oc-x-badge" class="px-2.5 py-1 text-xs font-mono rounded-lg bg-slate-800 border border-slate-700 text-slate-400">X: --</span>
</div>
</div>
<!-- Profile quick-select -->
<div class="grid grid-cols-3 gap-3">
<button id="oc-btn-ollama" onclick="applyOverclock('ollama')" class="py-2.5 rounded-xl border text-xs font-bold transition flex items-center justify-center space-x-2 bg-purple-950/60 border-purple-800 text-purple-300 hover:bg-purple-900/60">
<i class="fa-solid fa-brain"></i><span>Ollama</span>
</button>
<button id="oc-btn-comfy" onclick="applyOverclock('comfy')" class="py-2.5 rounded-xl border text-xs font-bold transition flex items-center justify-center space-x-2 bg-cyan-950/60 border-cyan-800 text-cyan-300 hover:bg-cyan-900/60">
<i class="fa-solid fa-palette"></i><span>ComfyUI</span>
</button>
<button id="oc-btn-balanced" onclick="applyOverclock('balanced')" class="py-2.5 rounded-xl border text-xs font-bold transition flex items-center justify-center space-x-2 bg-slate-800/60 border-slate-700 text-slate-300 hover:bg-slate-700/60">
<i class="fa-solid fa-scale-balanced"></i><span>Balanced</span>
</button>
</div>
<!-- Live status readback -->
<div class="grid grid-cols-2 sm:grid-cols-4 gap-3 font-mono text-center">
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800">
<span class="text-[10px] text-slate-400 uppercase block">Power Limit</span>
<span id="oc-power-limit" class="text-lg font-bold text-amber-400">-- W</span>
</div>
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800">
<span class="text-[10px] text-slate-400 uppercase block">Core Clock</span>
<span id="oc-core" class="text-lg font-bold text-cyan-400">--</span>
</div>
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800">
<span class="text-[10px] text-slate-400 uppercase block">Mem Clock</span>
<span id="oc-mem" class="text-lg font-bold text-purple-400">--</span>
</div>
<div class="p-3 rounded-xl bg-slate-950 border border-slate-800">
<span class="text-[10px] text-slate-400 uppercase block">Temp / Draw</span>
<span id="oc-temp-draw" class="text-lg font-bold text-emerald-400">--</span>
</div>
</div>
<!-- GPU FAN COOLING CONTROL -->
<div class="p-4 rounded-xl bg-slate-950 border border-slate-800 space-y-4">
<div class="flex flex-wrap items-center justify-between gap-2">
<div class="flex items-center space-x-2">
<i id="oc-fan-card-icon" class="fa-solid fa-fan text-emerald-400 text-sm fan-spinning"></i>
<div>
<span class="text-xs font-semibold text-slate-200 uppercase font-mono block">GPU Fan Cooling Control</span>
<span class="text-[10px] text-slate-400 font-mono">Dual-fan PWM active speed regulation</span>
</div>
</div>
<span id="oc-fan-badge" class="px-2.5 py-1 text-xs font-mono rounded-lg bg-emerald-950/80 border border-emerald-700 text-emerald-300 font-bold">
AUTO (VBIOS)
</span>
</div>
<!-- Fan Presets -->
<div class="grid grid-cols-2 sm:grid-cols-5 gap-2">
<button onclick="setFanAuto()" class="py-1.5 px-2 rounded-lg bg-slate-900 hover:bg-slate-800 border border-slate-700 text-slate-300 text-xs font-mono font-semibold transition text-center flex items-center justify-center space-x-1">
<i class="fa-solid fa-wand-magic-sparkles text-cyan-400 text-[10px]"></i>
<span>Auto (VBIOS)</span>
</button>
<button onclick="setFanManual(50)" class="py-1.5 px-2 rounded-lg bg-slate-900 hover:bg-slate-800 border border-slate-700 text-slate-300 text-xs font-mono font-semibold transition text-center">
50% Quiet
</button>
<button onclick="setFanManual(65)" class="py-1.5 px-2 rounded-lg bg-slate-900 hover:bg-slate-800 border border-slate-700 text-slate-300 text-xs font-mono font-semibold transition text-center">
65% Balanced
</button>
<button onclick="setFanManual(80)" class="py-1.5 px-2 rounded-lg bg-slate-900 hover:bg-slate-800 border border-slate-700 text-slate-300 text-xs font-mono font-semibold transition text-center">
80% Heavy
</button>
<button onclick="setFanManual(100)" class="py-1.5 px-2 rounded-lg bg-rose-950/60 hover:bg-rose-900/80 border border-rose-800 text-rose-300 text-xs font-mono font-bold transition text-center flex items-center justify-center space-x-1">
<i class="fa-solid fa-gauge-max text-rose-400 text-[10px]"></i>
<span>100% Turbo</span>
</button>
</div>
<!-- Live Dual-Fan Meters -->
<div class="grid grid-cols-1 sm:grid-cols-2 gap-3 p-3 rounded-lg bg-slate-900/60 border border-slate-800/80 text-xs font-mono">
<div class="space-y-1">
<div class="flex justify-between text-slate-400">
<span class="flex items-center space-x-1"><i class="fa-solid fa-fan text-[10px] text-cyan-400"></i><span>Fan 0 (Intake/Core):</span></span>
<span id="oc-fan0-val" class="font-bold text-slate-100">--%</span>
</div>
<div class="w-full bg-slate-950 rounded-full h-2 overflow-hidden border border-slate-800">
<div id="oc-fan0-bar" class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-full rounded-full transition-all duration-300" style="width: 0%"></div>
</div>
</div>
<div class="space-y-1">
<div class="flex justify-between text-slate-400">
<span class="flex items-center space-x-1"><i class="fa-solid fa-fan text-[10px] text-purple-400"></i><span>Fan 1 (Exhaust/VRM):</span></span>
<span id="oc-fan1-val" class="font-bold text-slate-100">--%</span>
</div>
<div class="w-full bg-slate-950 rounded-full h-2 overflow-hidden border border-slate-800">
<div id="oc-fan1-bar" class="bg-gradient-to-r from-purple-500 to-emerald-400 h-full rounded-full transition-all duration-300" style="width: 0%"></div>
</div>
</div>
</div>
<!-- Custom Manual Slider -->
<div class="space-y-2">
<div class="flex justify-between items-center text-xs">
<span class="text-slate-400">Custom Manual Target:</span>
<span id="oc-fan-slider-label" class="font-mono text-emerald-400 font-bold">65%</span>
</div>
<div class="flex items-center space-x-3">
<input id="oc-fan-slider" type="range" min="30" max="100" step="1" value="65" oninput="document.getElementById('oc-fan-slider-label').textContent=this.value+'%'" class="flex-1 accent-emerald-500">
<button onclick="setFanManual(parseInt(document.getElementById('oc-fan-slider').value))" class="px-3.5 py-1.5 rounded-lg bg-gradient-to-r from-emerald-600 to-teal-600 hover:from-emerald-500 hover:to-teal-500 text-white text-xs font-bold font-mono shadow transition flex items-center space-x-1">
<i class="fa-solid fa-check"></i>
<span>Apply</span>
</button>
</div>
</div>
</div>
<!-- GPU SPECS STRIP -->
<div class="grid grid-cols-2 md:grid-cols-3 lg:grid-cols-6 gap-2 font-mono text-center">
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">GPU</span>
<span id="oc-spec-gpu" class="text-[11px] font-bold text-slate-200 block truncate">--</span>
</div>
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">Driver</span>
<span id="oc-spec-driver" class="text-[11px] font-bold text-cyan-400 block">--</span>
</div>
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">VRAM</span>
<span id="oc-spec-vram" class="text-[11px] font-bold text-purple-400 block">--</span>
</div>
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">Max Core</span>
<span id="oc-spec-maxcore" class="text-[11px] font-bold text-cyan-400 block">--</span>
</div>
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">Max Mem</span>
<span id="oc-spec-maxmem" class="text-[11px] font-bold text-purple-400 block">--</span>
</div>
<div class="p-2 rounded-lg bg-slate-950 border border-slate-800">
<span class="text-[9px] text-slate-500 uppercase block">Power</span>
<span id="oc-spec-power" class="text-[11px] font-bold text-amber-400 block">--</span>
</div>
</div>
<!-- LIVE TIME-SERIES GRAPH (clocks + temp + fan) -->
<div class="rounded-xl bg-slate-950 border border-slate-800 p-3">
<div class="flex items-center justify-between mb-2">
<span class="text-xs font-semibold text-slate-300 uppercase font-mono">
<i class="fa-solid fa-chart-line text-fuchsia-400 mr-1"></i>Live Tuning Graph
</span>
<div class="flex flex-wrap items-center space-x-3 text-[10px] font-mono">
<span class="flex items-center space-x-1"><span class="w-2 h-2 rounded-full bg-cyan-400 inline-block"></span>Core MHz</span>
<span class="flex items-center space-x-1"><span class="w-2 h-2 rounded-full bg-purple-400 inline-block"></span>Mem MHz</span>
<span class="flex items-center space-x-1"><span class="w-2 h-2 rounded-full bg-rose-400 inline-block"></span>Temp °C</span>
<span class="flex items-center space-x-1"><span class="w-2 h-2 rounded-full bg-amber-400 inline-block"></span>Power W</span>
<span class="flex items-center space-x-1"><span class="w-2 h-2 rounded-full bg-emerald-400 inline-block"></span>Fan %</span>
</div>
</div>
<div class="relative h-64">
<canvas id="oc-chart"></canvas>
</div>
</div>
<!-- Fine-tune sliders for the selected profile -->
<div class="p-4 rounded-xl bg-slate-950 border border-slate-800 space-y-4">
<div class="flex items-center justify-between">
<span class="text-xs font-semibold text-slate-300 uppercase font-mono">Fine-tune profile</span>
<select id="oc-edit-profile" onchange="loadOverclockForProfile(this.value)" class="bg-slate-900 border border-slate-700 rounded-lg px-2 py-1 text-xs font-mono text-slate-200 focus:outline-none focus:border-fuchsia-500">
<option value="ollama">ollama</option>
<option value="comfy">comfy</option>
<option value="balanced">balanced</option>
</select>
</div>
<div class="grid grid-cols-1 md:grid-cols-3 gap-5">
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Power Limit</span><span id="oc-val-power" class="font-mono text-amber-400">370 W</span></label>
<input id="oc-slider-power" type="range" min="150" max="370" step="5" value="370" oninput="document.getElementById('oc-val-power').textContent=this.value+' W'" class="w-full accent-amber-500">
</div>
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Core Offset</span><span id="oc-val-core" class="font-mono text-cyan-400">+100 MHz</span></label>
<input id="oc-slider-core" type="range" min="0" max="200" step="5" value="100" oninput="document.getElementById('oc-val-core').textContent='+'+this.value+' MHz'" class="w-full accent-cyan-500">
</div>
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Memory Offset</span><span id="oc-val-mem" class="font-mono text-purple-400">+500 MHz</span></label>
<input id="oc-slider-mem" type="range" min="0" max="1000" step="25" value="500" oninput="document.getElementById('oc-val-mem').textContent='+'+this.value+' MHz'" class="w-full accent-purple-500">
</div>
</div>
<div class="grid grid-cols-1 md:grid-cols-2 gap-5">
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Core Clock Lock</span><span id="oc-val-lockcore" class="font-mono text-slate-400">Off</span></label>
<div class="flex items-center space-x-2">
<input id="oc-lockcore-toggle" type="checkbox" onchange="toggleCoreLock()" class="w-4 h-4 accent-cyan-500">
<span class="text-xs text-slate-500">Lock core to max boost (29003105 MHz)</span>
</div>
</div>
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Memory Clock Lock</span><span id="oc-val-lockmem" class="font-mono text-slate-400">Off</span></label>
<div class="flex items-center space-x-2">
<input id="oc-lockmem-toggle" type="checkbox" onchange="toggleMemLock()" class="w-4 h-4 accent-purple-500">
<span class="text-xs text-slate-500">Lock mem to max (11501 MHz)</span>
</div>
</div>
</div>
<!-- Per-Profile Fan Settings -->
<div class="grid grid-cols-1 md:grid-cols-2 gap-5 pt-3 border-t border-slate-900">
<div>
<label class="text-xs text-slate-400 flex justify-between"><span>Profile Fan Mode</span><span id="oc-val-prof-fanmode" class="font-mono text-emerald-400">Auto</span></label>
<select id="oc-prof-fanmode" onchange="toggleProfileFanMode()" class="w-full bg-slate-900 border border-slate-700 rounded-lg px-3 py-2 text-xs font-mono text-slate-200 mt-1 focus:outline-none focus:border-emerald-500">
<option value="auto">Auto (VBIOS Dynamic Curve)</option>
<option value="manual">Manual (Locked Target %)</option>
</select>
</div>
<div id="oc-prof-fanspeed-container" class="opacity-50 pointer-events-none transition">
<label class="text-xs text-slate-400 flex justify-between"><span>Profile Fan Target</span><span id="oc-val-prof-fanspeed" class="font-mono text-emerald-400">70%</span></label>
<input id="oc-slider-prof-fanspeed" type="range" min="30" max="100" step="5" value="70" oninput="document.getElementById('oc-val-prof-fanspeed').textContent=this.value+'%'" class="w-full accent-emerald-500 mt-2">
</div>
</div>
<div class="flex items-center justify-between pt-2">
<span id="oc-offset-note" class="text-[11px] font-mono text-slate-500"></span>
<button onclick="saveOverclockProfile()" class="px-4 py-2 bg-gradient-to-r from-fuchsia-600 to-purple-600 hover:from-fuchsia-500 hover:to-purple-500 text-white text-xs font-bold rounded-lg shadow-md transition flex items-center space-x-1.5">
<i class="fa-solid fa-floppy-disk"></i><span>Save Profile</span>
</button>
</div>
</div>
</div>
</main>
<script src="/static/app.js"></script>

View File

@@ -26,3 +26,13 @@
.glow-card {
animation: pulseGlow 4s infinite ease-in-out;
}
@keyframes spinFan {
from { transform: rotate(0deg); }
to { transform: rotate(360deg); }
}
.fan-spinning {
display: inline-block;
animation: spinFan 1s linear infinite;
}

View File

@@ -6,6 +6,12 @@ import logging
from typing import Dict, List, Any, Optional
from collections import deque
import asyncio
import json
import websockets
import overclock_manager
try:
import pynvml
pynvml.nvmlInit()
@@ -21,6 +27,7 @@ COMFY_API_BASE = "http://127.0.0.1:8188"
# Circular buffer for transition events
SWITCH_HISTORY = deque(maxlen=50)
def get_gpu_hardware_stats() -> Dict[str, Any]:
"""Retrieve comprehensive GPU hardware and process metrics via NVML."""
if not NVML_AVAILABLE:
@@ -41,10 +48,27 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
except Exception:
power_w = 0.0
fan_pct = 0
fans = []
try:
num_fans = pynvml.nvmlDeviceGetNumFans(handle)
for i in range(num_fans):
try:
fans.append(pynvml.nvmlDeviceGetFanSpeed_v2(handle, i))
except Exception:
pass
if fans:
fan_pct = max(fans)
else:
fan_pct = pynvml.nvmlDeviceGetFanSpeed(handle)
fans = [fan_pct]
except Exception:
try:
fan_pct = pynvml.nvmlDeviceGetFanSpeed(handle)
fans = [fan_pct]
except Exception:
fan_pct = 0
fans = []
try:
clock_graphics = pynvml.nvmlDeviceGetClockInfo(handle, pynvml.NVML_CLOCK_GRAPHICS)
@@ -119,6 +143,8 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"temperature_c": temp_c,
"power_w": power_w,
"fan_pct": fan_pct,
"fans": fans,
"num_fans": len(fans),
"clock_graphics_mhz": clock_graphics,
"clock_mem_mhz": clock_mem,
"breakdown": {
@@ -320,3 +346,159 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m") -> Dic
def get_switch_history() -> List[Dict[str, Any]]:
return list(SWITCH_HISTORY)
class AutoArbitrator:
"""Real-time bidirectional background arbitrator for seamless Ollama <-> ComfyUI hot-swapping."""
def __init__(self):
self.running = False
self.ws_task: Optional[asyncio.Task] = None
self.poll_task: Optional[asyncio.Task] = None
self.last_yield_time = 0.0
self.last_comfy_free_time = 0.0
self.connected_ws = False
self.last_action = "Idle"
self.comfy_was_active = False
self.oc_profile = None
async def start(self):
if self.running:
return
self.running = True
self.ws_task = asyncio.create_task(self._ws_listener())
self.poll_task = asyncio.create_task(self._poll_watchdog())
logger.info("AutoArbitrator background engine started (Bidirectional).")
# Apply the default (balanced) overclock profile on startup.
try:
await asyncio.get_event_loop().run_in_executor(None, overclock_manager.apply_profile, "balanced")
except Exception as e:
logger.warning(f"Startup overclock apply failed: {e}")
async def stop(self):
self.running = False
if self.ws_task:
self.ws_task.cancel()
if self.poll_task:
self.poll_task.cancel()
logger.info("AutoArbitrator background engine stopped.")
async def trigger_comfy_priority(self, reason: str = "ComfyUI prompt detected"):
"""Instantly yield Ollama VRAM to 0MB when ComfyUI needs to run diffusion models."""
self.comfy_was_active = True
self._apply_oc_profile("comfy")
now = time.time()
if now - self.last_yield_time < 1.0:
return
ollama_state = await get_ollama_live_state()
if ollama_state.get("active_model_name"):
model = ollama_state["active_model_name"]
logger.info(f"⚡ ComfyUI active ({reason}) -> Auto-yielding Ollama model '{model}' from VRAM...")
self.last_yield_time = time.time()
res = await instant_free_ollama_vram(model)
dur = res.get("duration_ms", 0)
self.last_action = f"Auto-yielded '{model}' for ComfyUI ({dur}ms)"
logger.info(f"Ollama auto-yield completed: {res}")
async def trigger_comfy_completed(self):
"""Purge ComfyUI VRAM cache when generation finishes, keeping VRAM 100% free for Ollama."""
now = time.time()
if now - self.last_comfy_free_time < 3.0:
return
self.last_comfy_free_time = now
self.comfy_was_active = False
self._apply_oc_profile("ollama")
logger.info("⚡ ComfyUI finished generation -> Auto-purging ComfyUI VRAM cache for Ollama...")
res = await instant_free_comfyui_vram()
dur = res.get("duration_ms", 0)
self.last_action = f"Auto-purged ComfyUI VRAM ({dur}ms) - Ready for Ollama"
logger.info(f"ComfyUI auto-purge completed: {res}")
async def _ws_listener(self):
client_id = "hyperswap-arbitrator"
ws_url = f"ws://127.0.0.1:8188/ws?clientId={client_id}"
while self.running:
try:
async with websockets.connect(ws_url, ping_interval=10, ping_timeout=10) as ws:
self.connected_ws = True
logger.info("AutoArbitrator connected to ComfyUI WebSocket.")
while self.running:
msg = await ws.recv()
if isinstance(msg, str):
try:
data = json.loads(msg)
msg_type = data.get("type")
msg_data = data.get("data", {})
if msg_type == "status":
queue_rem = msg_data.get("status", {}).get("exec_info", {}).get("queue_remaining", 0)
if queue_rem > 0:
await self.trigger_comfy_priority(f"Queue remaining: {queue_rem}")
elif queue_rem == 0 and self.comfy_was_active:
# Prompt queue finished
await asyncio.sleep(1.5)
await self.trigger_comfy_completed()
elif msg_type in ("execution_start", "execution_cached"):
await self.trigger_comfy_priority(f"Event: {msg_type}")
elif msg_type == "executing":
node = msg_data.get("node")
if node is not None:
await self.trigger_comfy_priority(f"Executing node: {node}")
elif node is None and self.comfy_was_active:
# Finished executing graph
await asyncio.sleep(1.5)
await self.trigger_comfy_completed()
elif msg_type == "execution_success":
await asyncio.sleep(1.5)
await self.trigger_comfy_completed()
except Exception as e:
logger.debug(f"WS parse error: {e}")
except (websockets.exceptions.ConnectionClosed, OSError, asyncio.CancelledError):
self.connected_ws = False
except Exception as e:
self.connected_ws = False
logger.debug(f"WS connection error: {e}")
await asyncio.sleep(2.0)
async def _poll_watchdog(self):
"""Watchdog polling /queue every 300ms for robust bidirectional arbitration."""
while self.running:
try:
comfy_state = await get_comfyui_live_state()
is_executing = comfy_state.get("queue_running", 0) > 0 or comfy_state.get("queue_remaining", 0) > 0 or comfy_state.get("executing", False)
if is_executing:
await self.trigger_comfy_priority("Polling detected active queue/execution")
elif self.comfy_was_active and not is_executing:
await asyncio.sleep(1.5)
await self.trigger_comfy_completed()
except Exception:
pass
await asyncio.sleep(0.3)
def _apply_oc_profile(self, profile: str):
"""Apply an overclock profile in a background thread; only fire on transition."""
if self.oc_profile == profile:
return
self.oc_profile = profile
try:
loop = asyncio.get_event_loop()
loop.run_in_executor(None, overclock_manager.apply_profile, profile)
logger.info(f"🎛️ Overclock profile switched -> '{profile}'")
except Exception as e:
logger.warning(f"Overclock profile switch failed ({profile}): {e}")
def get_status(self) -> Dict[str, Any]:
return {
"running": self.running,
"connected_ws": self.connected_ws,
"last_action": self.last_action,
"mode": "Bidirectional Hot-Swap (ComfyUI <-> Ollama)",
}
arbitrator = AutoArbitrator()