diff --git a/health.py b/health.py new file mode 100644 index 0000000..edf1f91 --- /dev/null +++ b/health.py @@ -0,0 +1,188 @@ +"""Dependency self-check. + +Motivation: fan control failed for an entire session because the service started before +the headless X server that owns the GPU was accepting connections. The failure was real, +recoverable and completely invisible -- it appeared once, inside one field of one log +line, and nothing ever asked "is fan control actually working?" + +Everything HyperSwap needs is checked here, each with a plain statement of what breaks +when it is missing and how to fix it. A degraded dependency should be loud. +""" +import asyncio +import logging +import os +import time +from typing import Any, Dict, List + +import httpx + +import overclock_manager +import ram_optimizer +import telemetry_store +import vram_arbitrator + +logger = logging.getLogger("health") + +OK, DEGRADED, FAILED = "ok", "degraded", "failed" + + +def _check(name: str, status: str, detail: str, impact: str = "", + fix: str = "") -> Dict[str, Any]: + return {"name": name, "status": status, "detail": detail, + "impact": impact, "fix": fix} + + +def _check_nvml() -> Dict[str, Any]: + if not vram_arbitrator.NVML_AVAILABLE: + return _check("nvml", FAILED, "pynvml did not initialise", + "No GPU telemetry, and VRAM yields cannot be confirmed", + "Check the NVIDIA driver and that pynvml is installed in the venv") + stats = vram_arbitrator.get_gpu_hardware_stats() + if not stats.get("available"): + return _check("nvml", FAILED, stats.get("error", "unavailable"), + "No GPU telemetry", "Check the NVIDIA driver") + return _check("nvml", OK, f"{stats.get('device_name')}, " + f"{stats.get('vram_total_gb')} GB") + + +def _check_sudo_smi() -> Dict[str, Any]: + r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader") + if r["rc"] != 0: + return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit", + "Power limits and clock locks cannot be applied", + "Passwordless sudo for /usr/bin/nvidia-smi is required " + "(see /etc/sudoers.d/)") + return _check("nvidia-smi (sudo)", OK, "passwordless sudo works") + + +def _check_fan_control() -> Dict[str, Any]: + """The check that would have caught the startup race.""" + if not overclock_manager.is_headless_x_running(): + return _check("fan control", FAILED, + f"no X server found on {overclock_manager.HEADLESS_DISPLAY}", + "Fan speed cannot be read or set; the thermal governor cannot " + "raise the fan floor when the card gets hot", + f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}") + status = overclock_manager.get_fan_status(force=True) + if status.get("target_speed_pct") is None and not status.get("manual"): + # Auto mode legitimately reports no target; probe the control attribute instead. + probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState") + if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe): + return _check("fan control", FAILED, + probe.get("err") or "GPU target not resolvable", + "Fan control unavailable; the governor cannot cool the card", + "Check Coolbits and that X on " + f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU") + return _check("fan control", OK, f"mode={status.get('mode')}") + + +def _check_profile_drift() -> Dict[str, Any]: + drift = overclock_manager.profile_drift() + if not drift.get("applied_since_start"): + return _check("overclock profile", DEGRADED, + "no profile has been successfully applied since startup", + "The card may not be running the settings this app reports", + "Apply a profile, or check the nvidia-smi/fan checks above") + if drift.get("drifted"): + return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"), + "Reported settings do not match the hardware", + "The sampler reconciles once a minute; POST /api/overclock/apply " + "to force it now") + return _check("overclock profile", OK, + f"{drift['profile']} @ {drift['power_limit_actual_w']}W") + + +async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]: + try: + async with httpx.AsyncClient(timeout=3.0) as c: + r = await c.get(url) + if r.status_code == 200: + return _check(name, OK, "reachable") + return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix) + except Exception as e: + return _check(name, FAILED, str(e)[:120], impact, fix) + + +def _check_comfy_ws() -> Dict[str, Any]: + arb = vram_arbitrator.arbitrator + if not arb.running: + return _check("arbitrator", FAILED, "background engine not running", + "No automatic VRAM handoff between Ollama and ComfyUI", + "Restart the service") + if not arb.connected_ws: + return _check("comfyui websocket", DEGRADED, "not connected", + "Falling back to 1 Hz polling; handoffs react more slowly", + "Check that ComfyUI is running and reachable on :8188") + return _check("comfyui websocket", OK, "subscribed") + + +def _check_store() -> Dict[str, Any]: + info = telemetry_store.db_info() + if not info.get("exists"): + return _check("telemetry store", DEGRADED, "database not created yet", + "No persisted history, so profile comparison cannot be computed", + "It is created on first write; check the directory is writable") + if not os.access(telemetry_store.DB_PATH, os.W_OK): + return _check("telemetry store", FAILED, "database not writable", + "Telemetry and swap events are being dropped", + f"Check permissions on {telemetry_store.DB_PATH}") + return _check("telemetry store", OK, + f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history") + + +def _check_residency() -> Dict[str, Any]: + cap = ram_optimizer.residency_capability() + if cap.get("exact_everywhere"): + return _check("residency measurement", OK, "cachestat available for all models") + return _check("residency measurement", DEGRADED, cap.get("reason", ""), + "Ollama weight residency is estimated by read-rate probe, not measured", + cap.get("hint", "")) + + +def _check_model_dirs() -> Dict[str, Any]: + comfy_dir = ram_optimizer.COMFY_MODELS_DIR + if not os.path.isdir(comfy_dir): + return _check("model directories", DEGRADED, + f"ComfyUI model directory not found: {comfy_dir}", + "ComfyUI checkpoints cannot be catalogued or pre-warmed", + "Set HYPERSWAP_COMFY_MODELS to the right path") + catalog = ram_optimizer.get_model_catalog() + return _check("model directories", OK, + f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files") + + +async def run_health_checks() -> Dict[str, Any]: + """Run every dependency check. Never raises.""" + t0 = time.perf_counter() + loop = asyncio.get_running_loop() + + sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control, + _check_profile_drift, _check_store, _check_residency, + _check_model_dirs, _check_comfy_ws] + results: List[Dict[str, Any]] = [] + for fn in sync_checks: + try: + results.append(await loop.run_in_executor(None, fn)) + except Exception as e: + results.append(_check(fn.__name__, FAILED, f"check raised: {e}")) + + results.extend(await asyncio.gather( + _check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags", + "No LLM orchestration", "Start the ollama service"), + _check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats", + "No diffusion arbitration", "Start ComfyUI on :8188"), + )) + + failed = [r for r in results if r["status"] == FAILED] + degraded = [r for r in results if r["status"] == DEGRADED] + overall = FAILED if failed else (DEGRADED if degraded else OK) + return { + "status": overall, + "checked_at": time.time(), + "duration_ms": round((time.perf_counter() - t0) * 1000, 1), + "summary": (f"{len(results) - len(failed) - len(degraded)} ok, " + f"{len(degraded)} degraded, {len(failed)} failed"), + "failed": [r["name"] for r in failed], + "degraded": [r["name"] for r in degraded], + "checks": results, + } diff --git a/mcp_server.py b/mcp_server.py index 47e037a..e5ac0f6 100644 --- a/mcp_server.py +++ b/mcp_server.py @@ -8,6 +8,7 @@ from typing import Dict, List, Any, Optional from mcp.server import MCPServer import autotune +import health import overclock_manager import ram_optimizer import telemetry_store @@ -136,6 +137,14 @@ def set_gpu_fan_speed(mode: str = "auto", percent: Optional[int] = None) -> str: res = overclock_manager.set_fan_auto() return json.dumps(res, indent=2) +@mcp.tool() +async def check_system_health() -> str: + """Check every dependency HyperSwap needs (NVML, sudo nvidia-smi, fan control via the + headless X server, Ollama, ComfyUI, the telemetry store, model directories) and report + what is broken, what it breaks, and how to fix it.""" + return json.dumps(await health.run_health_checks(), indent=2, default=str) + + @mcp.tool() def get_page_cache_residency(include_files: bool = True) -> str: """Measure how much of each model on disk is genuinely resident in the Linux page cache. diff --git a/server.py b/server.py index 5d66709..3066288 100644 --- a/server.py +++ b/server.py @@ -16,6 +16,7 @@ from fastapi.middleware.cors import CORSMiddleware from pydantic import BaseModel, Field import autotune +import health import overclock_manager import ram_optimizer import telemetry_store @@ -59,6 +60,7 @@ class TelemetryBroker: self.running = False self.samples = 0 self.last_sample_ms = 0.0 + self._models_fp = None # Set on shutdown so open SSE generators finish instead of holding the server up. self.closing = False @@ -89,6 +91,26 @@ class TelemetryBroker: with contextlib.suppress(asyncio.CancelledError): await self.task + def _stream_frame(self, snap: Dict[str, Any]) -> Dict[str, Any]: + """Trim the snapshot for streaming. + + The installed-model catalog is 10.6 KB of a 13.1 KB payload -- 81% -- and it + changes only when a model is pulled or removed, yet it was re-sent to every + subscriber every second (135 MB/hour across three tabs). It is sent on the first + frame and whenever it changes; otherwise the client keeps what it has. + + /api/stats still returns the complete snapshot, so API consumers are unaffected. + """ + ollama = snap.get("ollama", {}) + models = ollama.get("installed_models") or [] + fp = hash(tuple(sorted(m.get("name", "") for m in models))) + if fp == self._models_fp: + trimmed_ollama = {k: v for k, v in ollama.items() if k != "installed_models"} + trimmed_ollama["installed_models_unchanged"] = True + return {**snap, "ollama": trimmed_ollama} + self._models_fp = fp + return snap + def subscribe(self) -> asyncio.Queue: q: asyncio.Queue = asyncio.Queue(maxsize=2) self.subscribers.add(q) @@ -122,13 +144,14 @@ class TelemetryBroker: throttle_reasons=",".join(snap.get("gpu", {}).get("throttle_reasons") or []), ) + frame = self._stream_frame(snap) for q in list(self.subscribers): if q.full(): # Slow client: drop the stale frame rather than stalling the sampler. with contextlib.suppress(asyncio.QueueEmpty): q.get_nowait() with contextlib.suppress(asyncio.QueueFull): - q.put_nowait(snap) + q.put_nowait(frame) except asyncio.CancelledError: raise except Exception as e: @@ -288,6 +311,16 @@ async def get_all_stats() -> Dict[str, Any]: return await broker.get() +@app.get("/api/health", summary="Dependency Self-Check", tags=["Telemetry"]) +async def api_health(): + """Check everything HyperSwap depends on, with impact and remediation for each. + + Returns overall `status` of ok | degraded | failed. Exists because fan control once + failed for a whole session -- recoverably, and completely silently. + """ + return await health.run_health_checks() + + @app.get("/api/gpu", summary="GPU Sensors and VRAM Breakdown", tags=["Telemetry"]) async def get_gpu_metrics() -> Dict[str, Any]: """Detailed NVML sensors (utilization, temp, power, fan, clocks, throttle reasons, per-process VRAM).""" @@ -305,6 +338,7 @@ async def sse_telemetry_stream(request: Request): q = broker.subscribe() try: snap = await broker.get() + # Full snapshot first: a new subscriber has no cached catalog yet. yield f"data: {json.dumps(snap)}\n\n" while not broker.closing: if await request.is_disconnected(): diff --git a/static/app.js b/static/app.js index b516388..020eaa0 100644 --- a/static/app.js +++ b/static/app.js @@ -177,9 +177,13 @@ function updateDashboard(data) { document.getElementById('ollama-context').textContent = 'Idle'; } + // The catalog is omitted from stream frames when unchanged, so keep the last one. if (ollama.installed_models && ollama.installed_models.length > 0) { - document.getElementById('ollama-total-models').textContent = ollama.installed_models.length; - updateModelSelect(ollama.installed_models, ollama.active_model_name); + currentInstalledModels = ollama.installed_models; + } + if (currentInstalledModels && currentInstalledModels.length > 0) { + document.getElementById('ollama-total-models').textContent = currentInstalledModels.length; + updateModelSelect(currentInstalledModels, ollama.active_model_name); } } else { document.getElementById('ollama-status-text').textContent = 'OFFLINE'; @@ -927,3 +931,48 @@ function renderArbitrator(arb, gpu) { } el('arb-backoff').textContent = parts.join(' · '); } + + +// ---------------------------------------------------------------- health + +async function fetchHealth(verbose = false) { + const badge = document.getElementById('health-badge'); + const body = document.getElementById('health-body'); + if (!badge) return; + try { + const d = await (await fetch('/api/health')).json(); + const style = { + ok: 'bg-emerald-950/70 border border-emerald-800 text-emerald-300', + degraded: 'bg-amber-950/70 border border-amber-800 text-amber-300', + failed: 'bg-rose-950/70 border border-rose-800 text-rose-300', + }[d.status]; + badge.className = `px-2 py-1 text-xs font-bold rounded-lg ${style}`; + badge.textContent = `${d.status.toUpperCase()} · ${d.summary}`; + + // Healthy checks collapse to one line; anything wrong is shown in full with the + // impact and the fix, because that is the information you actually need. + const problems = d.checks.filter(c => c.status !== 'ok'); + const shown = (verbose || problems.length) ? (verbose ? d.checks : problems) : []; + body.innerHTML = shown.map(c => { + const colour = c.status === 'ok' ? 'text-emerald-400' + : c.status === 'degraded' ? 'text-amber-400' : 'text-rose-400'; + const mark = c.status === 'ok' ? '✓' : c.status === 'degraded' ? '!' : '✗'; + let html = `
${mark} ${c.name}` + + ` — ${c.detail}
`; + if (c.status !== 'ok') { + if (c.impact) html += `
impact: ${c.impact}
`; + if (c.fix) html += `
fix: ${c.fix}
`; + } + return html; + }).join('') || '
All dependencies healthy.
'; + } catch (e) { + badge.className = 'px-2 py-1 text-xs font-bold rounded-lg bg-rose-950/70 border border-rose-800 text-rose-300'; + badge.textContent = 'UNREACHABLE'; + body.innerHTML = `
${e}
`; + } +} + +document.addEventListener('DOMContentLoaded', () => { + fetchHealth(false); + setInterval(() => fetchHealth(false), 60000); +}); diff --git a/static/index.html b/static/index.html index f27666a..e6b6409 100644 --- a/static/index.html +++ b/static/index.html @@ -669,6 +669,26 @@
+ +
+
+
+
+ +
+
+

System Health

+

Every dependency, with impact and how to fix it

+
+
+
+ checking… + +
+
+
+
+
diff --git a/tests/test_health.py b/tests/test_health.py new file mode 100644 index 0000000..63e6d2e --- /dev/null +++ b/tests/test_health.py @@ -0,0 +1,134 @@ +"""Tests for the dependency self-check. + +This module exists because fan control failed for an entire session, recoverably and +invisibly: the service started before the headless X server that owns the GPU was +accepting connections, the assignment failed with "Error resolving target specification", +nothing retried, and nothing ever asked whether fan control worked. These tests make sure +each check reports the *right* status, since a self-check that returns ok when a +dependency is broken is worse than having none. +""" +import asyncio + +import pytest + +import health +import overclock_manager + + +class TestFanControlCheck: + """The check that would have caught the original bug.""" + + def test_fails_when_headless_x_is_not_running(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False) + res = health._check_fan_control() + assert res["status"] == health.FAILED + # A bare failure is not enough; it has to say what breaks and how to fix it. + assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower() + assert res["fix"] + + def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch): + # The exact nvidia-settings error seen at startup. + monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) + monkeypatch.setattr(overclock_manager, "get_fan_status", + lambda force=False: {"manual": False, "mode": "auto", + "target_speed_pct": None}) + monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: { + "rc": 1, "out": "", + "err": "ERROR: Error resolving target specification 'gpu:0' " + "(No targets match target specification)"}) + res = health._check_fan_control() + assert res["status"] == health.FAILED + + def test_ok_when_fan_status_reads_back(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) + monkeypatch.setattr(overclock_manager, "get_fan_status", + lambda force=False: {"manual": True, "mode": "manual", + "target_speed_pct": 70}) + assert health._check_fan_control()["status"] == health.OK + + +class TestProfileDriftCheck: + def test_degraded_when_no_profile_applied_since_start(self, monkeypatch): + # ACTIVE_PROFILE defaults to "balanced" at import, which used to be + # indistinguishable from "balanced was applied successfully". + monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { + "profile": "balanced", "applied_since_start": False, "drifted": True, + "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, + "reason": "no profile has been successfully applied since startup"}) + assert health._check_profile_drift()["status"] == health.DEGRADED + + def test_degraded_when_hardware_disagrees(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { + "profile": "balanced", "applied_since_start": True, "drifted": True, + "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, + "reason": "card reports 370.0W, profile asks 320W"}) + res = health._check_profile_drift() + assert res["status"] == health.DEGRADED + assert "370" in res["detail"] + + def test_ok_when_they_agree(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { + "profile": "balanced", "applied_since_start": True, "drifted": False, + "power_limit_intended_w": 320, "power_limit_actual_w": 320.0, + "reason": None}) + assert health._check_profile_drift()["status"] == health.OK + + +class TestSudoCheck: + def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "_smi", + lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"}) + res = health._check_sudo_smi() + assert res["status"] == health.FAILED + assert "sudo" in res["fix"].lower() + + def test_ok_when_it_works(self, monkeypatch): + monkeypatch.setattr(overclock_manager, "_smi", + lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""}) + assert health._check_sudo_smi()["status"] == health.OK + + +class TestAggregation: + """Overall status must be driven by the worst individual result.""" + + def _fake(self, statuses): + return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)] + + @pytest.mark.parametrize("statuses,expected", [ + ([health.OK, health.OK], health.OK), + ([health.OK, health.DEGRADED], health.DEGRADED), + ([health.OK, health.FAILED], health.FAILED), + ([health.DEGRADED, health.FAILED], health.FAILED), + ]) + def test_worst_status_wins(self, monkeypatch, statuses, expected): + checks = self._fake(statuses) + monkeypatch.setattr(health, "_check_nvml", lambda: checks[0]) + monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1]) + for fn in ("_check_fan_control", "_check_profile_drift", "_check_store", + "_check_residency", "_check_model_dirs", "_check_comfy_ws"): + monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) + + async def fake_http(name, url, impact, fix): + return health._check(name, health.OK, "reachable") + monkeypatch.setattr(health, "_check_http", fake_http) + + res = asyncio.run(health.run_health_checks()) + assert res["status"] == expected + + def test_a_raising_check_does_not_break_the_report(self, monkeypatch): + def boom(): + raise RuntimeError("nvml exploded") + monkeypatch.setattr(health, "_check_nvml", boom) + for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift", + "_check_store", "_check_residency", "_check_model_dirs", + "_check_comfy_ws"): + monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) + + async def fake_http(name, url, impact, fix): + return health._check(name, health.OK, "reachable") + monkeypatch.setattr(health, "_check_http", fake_http) + + res = asyncio.run(health.run_health_checks()) + # A broken check must surface as failed, not take down the endpoint. + assert res["status"] == health.FAILED + assert any("exploded" in c["detail"] for c in res["checks"])