Files
gpu-program-swapper/health.py
drjones fb104ac9c0 Add a dependency self-check; cut the SSE payload by 74%
Self-check. Fan control failed for an entire session -- recoverably, and completely
invisibly. It appeared once, inside one field of one log line, and nothing ever
asked whether fan control worked. health.py now checks everything this service
depends on (NVML, passwordless sudo for nvidia-smi, fan control via the headless X
server, overclock drift, the telemetry store, residency measurement capability,
model directories, the ComfyUI websocket, and both upstream HTTP services) and
reports for each one what is broken, what that breaks, and how to fix it. Exposed at
GET /api/health, as an MCP tool, and as a dashboard panel that collapses to a badge
when healthy and expands to impact-and-fix when not. Current state: 9 ok, 1 degraded
(the known cachestat permission limit on Ollama's blobs).

A self-check that returns ok while a dependency is broken is worse than none, so the
tests drive each check to its failure state -- including the exact "Error resolving
target specification 'gpu:0'" string from the original incident -- and assert that a
check which raises surfaces as failed rather than taking down the endpoint.

SSE payload. The installed-model catalog was 10.6 KB of a 13.1 KB frame, 81% of the
stream, re-sent to every subscriber every second despite changing only when a model
is pulled or removed: 135 MB/hour across three tabs. It is now sent on a
subscriber's first frame and whenever the set changes; the client keeps the last
known list. Steady-state frames dropped from 14041 to 3664 bytes, a 74% reduction,
and /api/stats still returns the complete snapshot for API consumers.

Tests: 182 (was 169).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 19:15:52 -07:00

189 lines
8.6 KiB
Python

"""Dependency self-check.
Motivation: fan control failed for an entire session because the service started before
the headless X server that owns the GPU was accepting connections. The failure was real,
recoverable and completely invisible -- it appeared once, inside one field of one log
line, and nothing ever asked "is fan control actually working?"
Everything HyperSwap needs is checked here, each with a plain statement of what breaks
when it is missing and how to fix it. A degraded dependency should be loud.
"""
import asyncio
import logging
import os
import time
from typing import Any, Dict, List
import httpx
import overclock_manager
import ram_optimizer
import telemetry_store
import vram_arbitrator
logger = logging.getLogger("health")
OK, DEGRADED, FAILED = "ok", "degraded", "failed"
def _check(name: str, status: str, detail: str, impact: str = "",
fix: str = "") -> Dict[str, Any]:
return {"name": name, "status": status, "detail": detail,
"impact": impact, "fix": fix}
def _check_nvml() -> Dict[str, Any]:
if not vram_arbitrator.NVML_AVAILABLE:
return _check("nvml", FAILED, "pynvml did not initialise",
"No GPU telemetry, and VRAM yields cannot be confirmed",
"Check the NVIDIA driver and that pynvml is installed in the venv")
stats = vram_arbitrator.get_gpu_hardware_stats()
if not stats.get("available"):
return _check("nvml", FAILED, stats.get("error", "unavailable"),
"No GPU telemetry", "Check the NVIDIA driver")
return _check("nvml", OK, f"{stats.get('device_name')}, "
f"{stats.get('vram_total_gb')} GB")
def _check_sudo_smi() -> Dict[str, Any]:
r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader")
if r["rc"] != 0:
return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit",
"Power limits and clock locks cannot be applied",
"Passwordless sudo for /usr/bin/nvidia-smi is required "
"(see /etc/sudoers.d/)")
return _check("nvidia-smi (sudo)", OK, "passwordless sudo works")
def _check_fan_control() -> Dict[str, Any]:
"""The check that would have caught the startup race."""
if not overclock_manager.is_headless_x_running():
return _check("fan control", FAILED,
f"no X server found on {overclock_manager.HEADLESS_DISPLAY}",
"Fan speed cannot be read or set; the thermal governor cannot "
"raise the fan floor when the card gets hot",
f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}")
status = overclock_manager.get_fan_status(force=True)
if status.get("target_speed_pct") is None and not status.get("manual"):
# Auto mode legitimately reports no target; probe the control attribute instead.
probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState")
if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe):
return _check("fan control", FAILED,
probe.get("err") or "GPU target not resolvable",
"Fan control unavailable; the governor cannot cool the card",
"Check Coolbits and that X on "
f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU")
return _check("fan control", OK, f"mode={status.get('mode')}")
def _check_profile_drift() -> Dict[str, Any]:
drift = overclock_manager.profile_drift()
if not drift.get("applied_since_start"):
return _check("overclock profile", DEGRADED,
"no profile has been successfully applied since startup",
"The card may not be running the settings this app reports",
"Apply a profile, or check the nvidia-smi/fan checks above")
if drift.get("drifted"):
return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"),
"Reported settings do not match the hardware",
"The sampler reconciles once a minute; POST /api/overclock/apply "
"to force it now")
return _check("overclock profile", OK,
f"{drift['profile']} @ {drift['power_limit_actual_w']}W")
async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]:
try:
async with httpx.AsyncClient(timeout=3.0) as c:
r = await c.get(url)
if r.status_code == 200:
return _check(name, OK, "reachable")
return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix)
except Exception as e:
return _check(name, FAILED, str(e)[:120], impact, fix)
def _check_comfy_ws() -> Dict[str, Any]:
arb = vram_arbitrator.arbitrator
if not arb.running:
return _check("arbitrator", FAILED, "background engine not running",
"No automatic VRAM handoff between Ollama and ComfyUI",
"Restart the service")
if not arb.connected_ws:
return _check("comfyui websocket", DEGRADED, "not connected",
"Falling back to 1 Hz polling; handoffs react more slowly",
"Check that ComfyUI is running and reachable on :8188")
return _check("comfyui websocket", OK, "subscribed")
def _check_store() -> Dict[str, Any]:
info = telemetry_store.db_info()
if not info.get("exists"):
return _check("telemetry store", DEGRADED, "database not created yet",
"No persisted history, so profile comparison cannot be computed",
"It is created on first write; check the directory is writable")
if not os.access(telemetry_store.DB_PATH, os.W_OK):
return _check("telemetry store", FAILED, "database not writable",
"Telemetry and swap events are being dropped",
f"Check permissions on {telemetry_store.DB_PATH}")
return _check("telemetry store", OK,
f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history")
def _check_residency() -> Dict[str, Any]:
cap = ram_optimizer.residency_capability()
if cap.get("exact_everywhere"):
return _check("residency measurement", OK, "cachestat available for all models")
return _check("residency measurement", DEGRADED, cap.get("reason", ""),
"Ollama weight residency is estimated by read-rate probe, not measured",
cap.get("hint", ""))
def _check_model_dirs() -> Dict[str, Any]:
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
if not os.path.isdir(comfy_dir):
return _check("model directories", DEGRADED,
f"ComfyUI model directory not found: {comfy_dir}",
"ComfyUI checkpoints cannot be catalogued or pre-warmed",
"Set HYPERSWAP_COMFY_MODELS to the right path")
catalog = ram_optimizer.get_model_catalog()
return _check("model directories", OK,
f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files")
async def run_health_checks() -> Dict[str, Any]:
"""Run every dependency check. Never raises."""
t0 = time.perf_counter()
loop = asyncio.get_running_loop()
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
_check_profile_drift, _check_store, _check_residency,
_check_model_dirs, _check_comfy_ws]
results: List[Dict[str, Any]] = []
for fn in sync_checks:
try:
results.append(await loop.run_in_executor(None, fn))
except Exception as e:
results.append(_check(fn.__name__, FAILED, f"check raised: {e}"))
results.extend(await asyncio.gather(
_check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags",
"No LLM orchestration", "Start the ollama service"),
_check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats",
"No diffusion arbitration", "Start ComfyUI on :8188"),
))
failed = [r for r in results if r["status"] == FAILED]
degraded = [r for r in results if r["status"] == DEGRADED]
overall = FAILED if failed else (DEGRADED if degraded else OK)
return {
"status": overall,
"checked_at": time.time(),
"duration_ms": round((time.perf_counter() - t0) * 1000, 1),
"summary": (f"{len(results) - len(failed) - len(degraded)} ok, "
f"{len(degraded)} degraded, {len(failed)} failed"),
"failed": [r["name"] for r in failed],
"degraded": [r["name"] for r in degraded],
"checks": results,
}