Add a dependency self-check; cut the SSE payload by 74%
Self-check. Fan control failed for an entire session -- recoverably, and completely invisibly. It appeared once, inside one field of one log line, and nothing ever asked whether fan control worked. health.py now checks everything this service depends on (NVML, passwordless sudo for nvidia-smi, fan control via the headless X server, overclock drift, the telemetry store, residency measurement capability, model directories, the ComfyUI websocket, and both upstream HTTP services) and reports for each one what is broken, what that breaks, and how to fix it. Exposed at GET /api/health, as an MCP tool, and as a dashboard panel that collapses to a badge when healthy and expands to impact-and-fix when not. Current state: 9 ok, 1 degraded (the known cachestat permission limit on Ollama's blobs). A self-check that returns ok while a dependency is broken is worse than none, so the tests drive each check to its failure state -- including the exact "Error resolving target specification 'gpu:0'" string from the original incident -- and assert that a check which raises surfaces as failed rather than taking down the endpoint. SSE payload. The installed-model catalog was 10.6 KB of a 13.1 KB frame, 81% of the stream, re-sent to every subscriber every second despite changing only when a model is pulled or removed: 135 MB/hour across three tabs. It is now sent on a subscriber's first frame and whenever the set changes; the client keeps the last known list. Steady-state frames dropped from 14041 to 3664 bytes, a 74% reduction, and /api/stats still returns the complete snapshot for API consumers. Tests: 182 (was 169). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
188
health.py
Normal file
188
health.py
Normal file
@@ -0,0 +1,188 @@
|
|||||||
|
"""Dependency self-check.
|
||||||
|
|
||||||
|
Motivation: fan control failed for an entire session because the service started before
|
||||||
|
the headless X server that owns the GPU was accepting connections. The failure was real,
|
||||||
|
recoverable and completely invisible -- it appeared once, inside one field of one log
|
||||||
|
line, and nothing ever asked "is fan control actually working?"
|
||||||
|
|
||||||
|
Everything HyperSwap needs is checked here, each with a plain statement of what breaks
|
||||||
|
when it is missing and how to fix it. A degraded dependency should be loud.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
from typing import Any, Dict, List
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
import overclock_manager
|
||||||
|
import ram_optimizer
|
||||||
|
import telemetry_store
|
||||||
|
import vram_arbitrator
|
||||||
|
|
||||||
|
logger = logging.getLogger("health")
|
||||||
|
|
||||||
|
OK, DEGRADED, FAILED = "ok", "degraded", "failed"
|
||||||
|
|
||||||
|
|
||||||
|
def _check(name: str, status: str, detail: str, impact: str = "",
|
||||||
|
fix: str = "") -> Dict[str, Any]:
|
||||||
|
return {"name": name, "status": status, "detail": detail,
|
||||||
|
"impact": impact, "fix": fix}
|
||||||
|
|
||||||
|
|
||||||
|
def _check_nvml() -> Dict[str, Any]:
|
||||||
|
if not vram_arbitrator.NVML_AVAILABLE:
|
||||||
|
return _check("nvml", FAILED, "pynvml did not initialise",
|
||||||
|
"No GPU telemetry, and VRAM yields cannot be confirmed",
|
||||||
|
"Check the NVIDIA driver and that pynvml is installed in the venv")
|
||||||
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
||||||
|
if not stats.get("available"):
|
||||||
|
return _check("nvml", FAILED, stats.get("error", "unavailable"),
|
||||||
|
"No GPU telemetry", "Check the NVIDIA driver")
|
||||||
|
return _check("nvml", OK, f"{stats.get('device_name')}, "
|
||||||
|
f"{stats.get('vram_total_gb')} GB")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_sudo_smi() -> Dict[str, Any]:
|
||||||
|
r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader")
|
||||||
|
if r["rc"] != 0:
|
||||||
|
return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit",
|
||||||
|
"Power limits and clock locks cannot be applied",
|
||||||
|
"Passwordless sudo for /usr/bin/nvidia-smi is required "
|
||||||
|
"(see /etc/sudoers.d/)")
|
||||||
|
return _check("nvidia-smi (sudo)", OK, "passwordless sudo works")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_fan_control() -> Dict[str, Any]:
|
||||||
|
"""The check that would have caught the startup race."""
|
||||||
|
if not overclock_manager.is_headless_x_running():
|
||||||
|
return _check("fan control", FAILED,
|
||||||
|
f"no X server found on {overclock_manager.HEADLESS_DISPLAY}",
|
||||||
|
"Fan speed cannot be read or set; the thermal governor cannot "
|
||||||
|
"raise the fan floor when the card gets hot",
|
||||||
|
f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}")
|
||||||
|
status = overclock_manager.get_fan_status(force=True)
|
||||||
|
if status.get("target_speed_pct") is None and not status.get("manual"):
|
||||||
|
# Auto mode legitimately reports no target; probe the control attribute instead.
|
||||||
|
probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState")
|
||||||
|
if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe):
|
||||||
|
return _check("fan control", FAILED,
|
||||||
|
probe.get("err") or "GPU target not resolvable",
|
||||||
|
"Fan control unavailable; the governor cannot cool the card",
|
||||||
|
"Check Coolbits and that X on "
|
||||||
|
f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU")
|
||||||
|
return _check("fan control", OK, f"mode={status.get('mode')}")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_profile_drift() -> Dict[str, Any]:
|
||||||
|
drift = overclock_manager.profile_drift()
|
||||||
|
if not drift.get("applied_since_start"):
|
||||||
|
return _check("overclock profile", DEGRADED,
|
||||||
|
"no profile has been successfully applied since startup",
|
||||||
|
"The card may not be running the settings this app reports",
|
||||||
|
"Apply a profile, or check the nvidia-smi/fan checks above")
|
||||||
|
if drift.get("drifted"):
|
||||||
|
return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"),
|
||||||
|
"Reported settings do not match the hardware",
|
||||||
|
"The sampler reconciles once a minute; POST /api/overclock/apply "
|
||||||
|
"to force it now")
|
||||||
|
return _check("overclock profile", OK,
|
||||||
|
f"{drift['profile']} @ {drift['power_limit_actual_w']}W")
|
||||||
|
|
||||||
|
|
||||||
|
async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]:
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=3.0) as c:
|
||||||
|
r = await c.get(url)
|
||||||
|
if r.status_code == 200:
|
||||||
|
return _check(name, OK, "reachable")
|
||||||
|
return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix)
|
||||||
|
except Exception as e:
|
||||||
|
return _check(name, FAILED, str(e)[:120], impact, fix)
|
||||||
|
|
||||||
|
|
||||||
|
def _check_comfy_ws() -> Dict[str, Any]:
|
||||||
|
arb = vram_arbitrator.arbitrator
|
||||||
|
if not arb.running:
|
||||||
|
return _check("arbitrator", FAILED, "background engine not running",
|
||||||
|
"No automatic VRAM handoff between Ollama and ComfyUI",
|
||||||
|
"Restart the service")
|
||||||
|
if not arb.connected_ws:
|
||||||
|
return _check("comfyui websocket", DEGRADED, "not connected",
|
||||||
|
"Falling back to 1 Hz polling; handoffs react more slowly",
|
||||||
|
"Check that ComfyUI is running and reachable on :8188")
|
||||||
|
return _check("comfyui websocket", OK, "subscribed")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_store() -> Dict[str, Any]:
|
||||||
|
info = telemetry_store.db_info()
|
||||||
|
if not info.get("exists"):
|
||||||
|
return _check("telemetry store", DEGRADED, "database not created yet",
|
||||||
|
"No persisted history, so profile comparison cannot be computed",
|
||||||
|
"It is created on first write; check the directory is writable")
|
||||||
|
if not os.access(telemetry_store.DB_PATH, os.W_OK):
|
||||||
|
return _check("telemetry store", FAILED, "database not writable",
|
||||||
|
"Telemetry and swap events are being dropped",
|
||||||
|
f"Check permissions on {telemetry_store.DB_PATH}")
|
||||||
|
return _check("telemetry store", OK,
|
||||||
|
f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history")
|
||||||
|
|
||||||
|
|
||||||
|
def _check_residency() -> Dict[str, Any]:
|
||||||
|
cap = ram_optimizer.residency_capability()
|
||||||
|
if cap.get("exact_everywhere"):
|
||||||
|
return _check("residency measurement", OK, "cachestat available for all models")
|
||||||
|
return _check("residency measurement", DEGRADED, cap.get("reason", ""),
|
||||||
|
"Ollama weight residency is estimated by read-rate probe, not measured",
|
||||||
|
cap.get("hint", ""))
|
||||||
|
|
||||||
|
|
||||||
|
def _check_model_dirs() -> Dict[str, Any]:
|
||||||
|
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
|
||||||
|
if not os.path.isdir(comfy_dir):
|
||||||
|
return _check("model directories", DEGRADED,
|
||||||
|
f"ComfyUI model directory not found: {comfy_dir}",
|
||||||
|
"ComfyUI checkpoints cannot be catalogued or pre-warmed",
|
||||||
|
"Set HYPERSWAP_COMFY_MODELS to the right path")
|
||||||
|
catalog = ram_optimizer.get_model_catalog()
|
||||||
|
return _check("model directories", OK,
|
||||||
|
f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files")
|
||||||
|
|
||||||
|
|
||||||
|
async def run_health_checks() -> Dict[str, Any]:
|
||||||
|
"""Run every dependency check. Never raises."""
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
|
||||||
|
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
|
||||||
|
_check_profile_drift, _check_store, _check_residency,
|
||||||
|
_check_model_dirs, _check_comfy_ws]
|
||||||
|
results: List[Dict[str, Any]] = []
|
||||||
|
for fn in sync_checks:
|
||||||
|
try:
|
||||||
|
results.append(await loop.run_in_executor(None, fn))
|
||||||
|
except Exception as e:
|
||||||
|
results.append(_check(fn.__name__, FAILED, f"check raised: {e}"))
|
||||||
|
|
||||||
|
results.extend(await asyncio.gather(
|
||||||
|
_check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags",
|
||||||
|
"No LLM orchestration", "Start the ollama service"),
|
||||||
|
_check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats",
|
||||||
|
"No diffusion arbitration", "Start ComfyUI on :8188"),
|
||||||
|
))
|
||||||
|
|
||||||
|
failed = [r for r in results if r["status"] == FAILED]
|
||||||
|
degraded = [r for r in results if r["status"] == DEGRADED]
|
||||||
|
overall = FAILED if failed else (DEGRADED if degraded else OK)
|
||||||
|
return {
|
||||||
|
"status": overall,
|
||||||
|
"checked_at": time.time(),
|
||||||
|
"duration_ms": round((time.perf_counter() - t0) * 1000, 1),
|
||||||
|
"summary": (f"{len(results) - len(failed) - len(degraded)} ok, "
|
||||||
|
f"{len(degraded)} degraded, {len(failed)} failed"),
|
||||||
|
"failed": [r["name"] for r in failed],
|
||||||
|
"degraded": [r["name"] for r in degraded],
|
||||||
|
"checks": results,
|
||||||
|
}
|
||||||
@@ -8,6 +8,7 @@ from typing import Dict, List, Any, Optional
|
|||||||
|
|
||||||
from mcp.server import MCPServer
|
from mcp.server import MCPServer
|
||||||
import autotune
|
import autotune
|
||||||
|
import health
|
||||||
import overclock_manager
|
import overclock_manager
|
||||||
import ram_optimizer
|
import ram_optimizer
|
||||||
import telemetry_store
|
import telemetry_store
|
||||||
@@ -136,6 +137,14 @@ def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str:
|
|||||||
res = overclock_manager.set_fan_auto()
|
res = overclock_manager.set_fan_auto()
|
||||||
return json.dumps(res, indent=2)
|
return json.dumps(res, indent=2)
|
||||||
|
|
||||||
|
@mcp.tool()
|
||||||
|
async def check_system_health() -> str:
|
||||||
|
"""Check every dependency HyperSwap needs (NVML, sudo nvidia-smi, fan control via the
|
||||||
|
headless X server, Ollama, ComfyUI, the telemetry store, model directories) and report
|
||||||
|
what is broken, what it breaks, and how to fix it."""
|
||||||
|
return json.dumps(await health.run_health_checks(), indent=2, default=str)
|
||||||
|
|
||||||
|
|
||||||
@mcp.tool()
|
@mcp.tool()
|
||||||
def get_page_cache_residency(include_files: bool = True) -> str:
|
def get_page_cache_residency(include_files: bool = True) -> str:
|
||||||
"""Measure how much of each model on disk is genuinely resident in the Linux page cache.
|
"""Measure how much of each model on disk is genuinely resident in the Linux page cache.
|
||||||
|
|||||||
36
server.py
36
server.py
@@ -16,6 +16,7 @@ from fastapi.middleware.cors import CORSMiddleware
|
|||||||
from pydantic import BaseModel, Field
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
import autotune
|
import autotune
|
||||||
|
import health
|
||||||
import overclock_manager
|
import overclock_manager
|
||||||
import ram_optimizer
|
import ram_optimizer
|
||||||
import telemetry_store
|
import telemetry_store
|
||||||
@@ -59,6 +60,7 @@ class TelemetryBroker:
|
|||||||
self.running = False
|
self.running = False
|
||||||
self.samples = 0
|
self.samples = 0
|
||||||
self.last_sample_ms = 0.0
|
self.last_sample_ms = 0.0
|
||||||
|
self._models_fp = None
|
||||||
# Set on shutdown so open SSE generators finish instead of holding the server up.
|
# Set on shutdown so open SSE generators finish instead of holding the server up.
|
||||||
self.closing = False
|
self.closing = False
|
||||||
|
|
||||||
@@ -89,6 +91,26 @@ class TelemetryBroker:
|
|||||||
with contextlib.suppress(asyncio.CancelledError):
|
with contextlib.suppress(asyncio.CancelledError):
|
||||||
await self.task
|
await self.task
|
||||||
|
|
||||||
|
def _stream_frame(self, snap: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
"""Trim the snapshot for streaming.
|
||||||
|
|
||||||
|
The installed-model catalog is 10.6 KB of a 13.1 KB payload -- 81% -- and it
|
||||||
|
changes only when a model is pulled or removed, yet it was re-sent to every
|
||||||
|
subscriber every second (135 MB/hour across three tabs). It is sent on the first
|
||||||
|
frame and whenever it changes; otherwise the client keeps what it has.
|
||||||
|
|
||||||
|
/api/stats still returns the complete snapshot, so API consumers are unaffected.
|
||||||
|
"""
|
||||||
|
ollama = snap.get("ollama", {})
|
||||||
|
models = ollama.get("installed_models") or []
|
||||||
|
fp = hash(tuple(sorted(m.get("name", "") for m in models)))
|
||||||
|
if fp == self._models_fp:
|
||||||
|
trimmed_ollama = {k: v for k, v in ollama.items() if k != "installed_models"}
|
||||||
|
trimmed_ollama["installed_models_unchanged"] = True
|
||||||
|
return {**snap, "ollama": trimmed_ollama}
|
||||||
|
self._models_fp = fp
|
||||||
|
return snap
|
||||||
|
|
||||||
def subscribe(self) -> asyncio.Queue:
|
def subscribe(self) -> asyncio.Queue:
|
||||||
q: asyncio.Queue = asyncio.Queue(maxsize=2)
|
q: asyncio.Queue = asyncio.Queue(maxsize=2)
|
||||||
self.subscribers.add(q)
|
self.subscribers.add(q)
|
||||||
@@ -122,13 +144,14 @@ class TelemetryBroker:
|
|||||||
throttle_reasons=",".join(snap.get("gpu", {}).get("throttle_reasons") or []),
|
throttle_reasons=",".join(snap.get("gpu", {}).get("throttle_reasons") or []),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
frame = self._stream_frame(snap)
|
||||||
for q in list(self.subscribers):
|
for q in list(self.subscribers):
|
||||||
if q.full():
|
if q.full():
|
||||||
# Slow client: drop the stale frame rather than stalling the sampler.
|
# Slow client: drop the stale frame rather than stalling the sampler.
|
||||||
with contextlib.suppress(asyncio.QueueEmpty):
|
with contextlib.suppress(asyncio.QueueEmpty):
|
||||||
q.get_nowait()
|
q.get_nowait()
|
||||||
with contextlib.suppress(asyncio.QueueFull):
|
with contextlib.suppress(asyncio.QueueFull):
|
||||||
q.put_nowait(snap)
|
q.put_nowait(frame)
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -288,6 +311,16 @@ async def get_all_stats() -> Dict[str, Any]:
|
|||||||
return await broker.get()
|
return await broker.get()
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/api/health", summary="Dependency Self-Check", tags=["Telemetry"])
|
||||||
|
async def api_health():
|
||||||
|
"""Check everything HyperSwap depends on, with impact and remediation for each.
|
||||||
|
|
||||||
|
Returns overall `status` of ok | degraded | failed. Exists because fan control once
|
||||||
|
failed for a whole session -- recoverably, and completely silently.
|
||||||
|
"""
|
||||||
|
return await health.run_health_checks()
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/gpu", summary="GPU Sensors and VRAM Breakdown", tags=["Telemetry"])
|
@app.get("/api/gpu", summary="GPU Sensors and VRAM Breakdown", tags=["Telemetry"])
|
||||||
async def get_gpu_metrics() -> Dict[str, Any]:
|
async def get_gpu_metrics() -> Dict[str, Any]:
|
||||||
"""Detailed NVML sensors (utilization, temp, power, fan, clocks, throttle reasons, per-process VRAM)."""
|
"""Detailed NVML sensors (utilization, temp, power, fan, clocks, throttle reasons, per-process VRAM)."""
|
||||||
@@ -305,6 +338,7 @@ async def sse_telemetry_stream(request: Request):
|
|||||||
q = broker.subscribe()
|
q = broker.subscribe()
|
||||||
try:
|
try:
|
||||||
snap = await broker.get()
|
snap = await broker.get()
|
||||||
|
# Full snapshot first: a new subscriber has no cached catalog yet.
|
||||||
yield f"data: {json.dumps(snap)}\n\n"
|
yield f"data: {json.dumps(snap)}\n\n"
|
||||||
while not broker.closing:
|
while not broker.closing:
|
||||||
if await request.is_disconnected():
|
if await request.is_disconnected():
|
||||||
|
|||||||
@@ -177,9 +177,13 @@ function updateDashboard(data) {
|
|||||||
document.getElementById('ollama-context').textContent = 'Idle';
|
document.getElementById('ollama-context').textContent = 'Idle';
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The catalog is omitted from stream frames when unchanged, so keep the last one.
|
||||||
if (ollama.installed_models && ollama.installed_models.length > 0) {
|
if (ollama.installed_models && ollama.installed_models.length > 0) {
|
||||||
document.getElementById('ollama-total-models').textContent = ollama.installed_models.length;
|
currentInstalledModels = ollama.installed_models;
|
||||||
updateModelSelect(ollama.installed_models, ollama.active_model_name);
|
}
|
||||||
|
if (currentInstalledModels && currentInstalledModels.length > 0) {
|
||||||
|
document.getElementById('ollama-total-models').textContent = currentInstalledModels.length;
|
||||||
|
updateModelSelect(currentInstalledModels, ollama.active_model_name);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
document.getElementById('ollama-status-text').textContent = 'OFFLINE';
|
document.getElementById('ollama-status-text').textContent = 'OFFLINE';
|
||||||
@@ -927,3 +931,48 @@ function renderArbitrator(arb, gpu) {
|
|||||||
}
|
}
|
||||||
el('arb-backoff').textContent = parts.join(' · ');
|
el('arb-backoff').textContent = parts.join(' · ');
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------- health
|
||||||
|
|
||||||
|
async function fetchHealth(verbose = false) {
|
||||||
|
const badge = document.getElementById('health-badge');
|
||||||
|
const body = document.getElementById('health-body');
|
||||||
|
if (!badge) return;
|
||||||
|
try {
|
||||||
|
const d = await (await fetch('/api/health')).json();
|
||||||
|
const style = {
|
||||||
|
ok: 'bg-emerald-950/70 border border-emerald-800 text-emerald-300',
|
||||||
|
degraded: 'bg-amber-950/70 border border-amber-800 text-amber-300',
|
||||||
|
failed: 'bg-rose-950/70 border border-rose-800 text-rose-300',
|
||||||
|
}[d.status];
|
||||||
|
badge.className = `px-2 py-1 text-xs font-bold rounded-lg ${style}`;
|
||||||
|
badge.textContent = `${d.status.toUpperCase()} · ${d.summary}`;
|
||||||
|
|
||||||
|
// Healthy checks collapse to one line; anything wrong is shown in full with the
|
||||||
|
// impact and the fix, because that is the information you actually need.
|
||||||
|
const problems = d.checks.filter(c => c.status !== 'ok');
|
||||||
|
const shown = (verbose || problems.length) ? (verbose ? d.checks : problems) : [];
|
||||||
|
body.innerHTML = shown.map(c => {
|
||||||
|
const colour = c.status === 'ok' ? 'text-emerald-400'
|
||||||
|
: c.status === 'degraded' ? 'text-amber-400' : 'text-rose-400';
|
||||||
|
const mark = c.status === 'ok' ? '✓' : c.status === 'degraded' ? '!' : '✗';
|
||||||
|
let html = `<div><span class="${colour}">${mark} ${c.name}</span>` +
|
||||||
|
`<span class="text-slate-400"> — ${c.detail}</span></div>`;
|
||||||
|
if (c.status !== 'ok') {
|
||||||
|
if (c.impact) html += `<div class="text-slate-500 pl-4">impact: ${c.impact}</div>`;
|
||||||
|
if (c.fix) html += `<div class="text-slate-500 pl-4">fix: ${c.fix}</div>`;
|
||||||
|
}
|
||||||
|
return html;
|
||||||
|
}).join('') || '<div class="text-emerald-500">All dependencies healthy.</div>';
|
||||||
|
} catch (e) {
|
||||||
|
badge.className = 'px-2 py-1 text-xs font-bold rounded-lg bg-rose-950/70 border border-rose-800 text-rose-300';
|
||||||
|
badge.textContent = 'UNREACHABLE';
|
||||||
|
body.innerHTML = `<div class="text-rose-400">${e}</div>`;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
document.addEventListener('DOMContentLoaded', () => {
|
||||||
|
fetchHealth(false);
|
||||||
|
setInterval(() => fetchHealth(false), 60000);
|
||||||
|
});
|
||||||
|
|||||||
@@ -669,6 +669,26 @@
|
|||||||
<div class="grid grid-cols-1 xl:grid-cols-2 gap-5 mt-5">
|
<div class="grid grid-cols-1 xl:grid-cols-2 gap-5 mt-5">
|
||||||
|
|
||||||
|
|
||||||
|
<!-- System health: makes a silently-broken dependency loud -->
|
||||||
|
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
|
||||||
|
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
||||||
|
<div class="flex items-center space-x-2">
|
||||||
|
<div class="p-2 rounded-lg bg-sky-950/80 border border-sky-800 text-sky-400">
|
||||||
|
<i class="fa-solid fa-stethoscope text-sm"></i>
|
||||||
|
</div>
|
||||||
|
<div>
|
||||||
|
<h3 class="font-bold text-slate-100 text-sm">System Health</h3>
|
||||||
|
<p class="text-xs text-slate-400">Every dependency, with impact and how to fix it</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div class="flex items-center space-x-2">
|
||||||
|
<span id="health-badge" class="px-2 py-1 text-xs font-bold rounded-lg bg-slate-800 text-slate-400">checking…</span>
|
||||||
|
<button onclick="fetchHealth(true)" class="px-2.5 py-1 text-xs font-semibold rounded-lg bg-slate-800 border border-slate-700 text-slate-300 hover:bg-slate-700 transition">Recheck</button>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div id="health-body" class="mt-4 space-y-1.5 text-[11px] font-mono"></div>
|
||||||
|
</div>
|
||||||
|
|
||||||
<!-- VRAM Arbitration — the core handoff, previously invisible -->
|
<!-- VRAM Arbitration — the core handoff, previously invisible -->
|
||||||
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
|
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
|
||||||
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
||||||
|
|||||||
134
tests/test_health.py
Normal file
134
tests/test_health.py
Normal file
@@ -0,0 +1,134 @@
|
|||||||
|
"""Tests for the dependency self-check.
|
||||||
|
|
||||||
|
This module exists because fan control failed for an entire session, recoverably and
|
||||||
|
invisibly: the service started before the headless X server that owns the GPU was
|
||||||
|
accepting connections, the assignment failed with "Error resolving target specification",
|
||||||
|
nothing retried, and nothing ever asked whether fan control worked. These tests make sure
|
||||||
|
each check reports the *right* status, since a self-check that returns ok when a
|
||||||
|
dependency is broken is worse than having none.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
import health
|
||||||
|
import overclock_manager
|
||||||
|
|
||||||
|
|
||||||
|
class TestFanControlCheck:
|
||||||
|
"""The check that would have caught the original bug."""
|
||||||
|
|
||||||
|
def test_fails_when_headless_x_is_not_running(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False)
|
||||||
|
res = health._check_fan_control()
|
||||||
|
assert res["status"] == health.FAILED
|
||||||
|
# A bare failure is not enough; it has to say what breaks and how to fix it.
|
||||||
|
assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower()
|
||||||
|
assert res["fix"]
|
||||||
|
|
||||||
|
def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch):
|
||||||
|
# The exact nvidia-settings error seen at startup.
|
||||||
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
||||||
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
||||||
|
lambda force=False: {"manual": False, "mode": "auto",
|
||||||
|
"target_speed_pct": None})
|
||||||
|
monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: {
|
||||||
|
"rc": 1, "out": "",
|
||||||
|
"err": "ERROR: Error resolving target specification 'gpu:0' "
|
||||||
|
"(No targets match target specification)"})
|
||||||
|
res = health._check_fan_control()
|
||||||
|
assert res["status"] == health.FAILED
|
||||||
|
|
||||||
|
def test_ok_when_fan_status_reads_back(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
||||||
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
||||||
|
lambda force=False: {"manual": True, "mode": "manual",
|
||||||
|
"target_speed_pct": 70})
|
||||||
|
assert health._check_fan_control()["status"] == health.OK
|
||||||
|
|
||||||
|
|
||||||
|
class TestProfileDriftCheck:
|
||||||
|
def test_degraded_when_no_profile_applied_since_start(self, monkeypatch):
|
||||||
|
# ACTIVE_PROFILE defaults to "balanced" at import, which used to be
|
||||||
|
# indistinguishable from "balanced was applied successfully".
|
||||||
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
||||||
|
"profile": "balanced", "applied_since_start": False, "drifted": True,
|
||||||
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
||||||
|
"reason": "no profile has been successfully applied since startup"})
|
||||||
|
assert health._check_profile_drift()["status"] == health.DEGRADED
|
||||||
|
|
||||||
|
def test_degraded_when_hardware_disagrees(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
||||||
|
"profile": "balanced", "applied_since_start": True, "drifted": True,
|
||||||
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
||||||
|
"reason": "card reports 370.0W, profile asks 320W"})
|
||||||
|
res = health._check_profile_drift()
|
||||||
|
assert res["status"] == health.DEGRADED
|
||||||
|
assert "370" in res["detail"]
|
||||||
|
|
||||||
|
def test_ok_when_they_agree(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
||||||
|
"profile": "balanced", "applied_since_start": True, "drifted": False,
|
||||||
|
"power_limit_intended_w": 320, "power_limit_actual_w": 320.0,
|
||||||
|
"reason": None})
|
||||||
|
assert health._check_profile_drift()["status"] == health.OK
|
||||||
|
|
||||||
|
|
||||||
|
class TestSudoCheck:
|
||||||
|
def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "_smi",
|
||||||
|
lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"})
|
||||||
|
res = health._check_sudo_smi()
|
||||||
|
assert res["status"] == health.FAILED
|
||||||
|
assert "sudo" in res["fix"].lower()
|
||||||
|
|
||||||
|
def test_ok_when_it_works(self, monkeypatch):
|
||||||
|
monkeypatch.setattr(overclock_manager, "_smi",
|
||||||
|
lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""})
|
||||||
|
assert health._check_sudo_smi()["status"] == health.OK
|
||||||
|
|
||||||
|
|
||||||
|
class TestAggregation:
|
||||||
|
"""Overall status must be driven by the worst individual result."""
|
||||||
|
|
||||||
|
def _fake(self, statuses):
|
||||||
|
return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)]
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("statuses,expected", [
|
||||||
|
([health.OK, health.OK], health.OK),
|
||||||
|
([health.OK, health.DEGRADED], health.DEGRADED),
|
||||||
|
([health.OK, health.FAILED], health.FAILED),
|
||||||
|
([health.DEGRADED, health.FAILED], health.FAILED),
|
||||||
|
])
|
||||||
|
def test_worst_status_wins(self, monkeypatch, statuses, expected):
|
||||||
|
checks = self._fake(statuses)
|
||||||
|
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
|
||||||
|
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
|
||||||
|
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
|
||||||
|
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
|
||||||
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||||
|
|
||||||
|
async def fake_http(name, url, impact, fix):
|
||||||
|
return health._check(name, health.OK, "reachable")
|
||||||
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
||||||
|
|
||||||
|
res = asyncio.run(health.run_health_checks())
|
||||||
|
assert res["status"] == expected
|
||||||
|
|
||||||
|
def test_a_raising_check_does_not_break_the_report(self, monkeypatch):
|
||||||
|
def boom():
|
||||||
|
raise RuntimeError("nvml exploded")
|
||||||
|
monkeypatch.setattr(health, "_check_nvml", boom)
|
||||||
|
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
|
||||||
|
"_check_store", "_check_residency", "_check_model_dirs",
|
||||||
|
"_check_comfy_ws"):
|
||||||
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
||||||
|
|
||||||
|
async def fake_http(name, url, impact, fix):
|
||||||
|
return health._check(name, health.OK, "reachable")
|
||||||
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
||||||
|
|
||||||
|
res = asyncio.run(health.run_health_checks())
|
||||||
|
# A broken check must surface as failed, not take down the endpoint.
|
||||||
|
assert res["status"] == health.FAILED
|
||||||
|
assert any("exploded" in c["detail"] for c in res["checks"])
|
||||||
Reference in New Issue
Block a user