Add a dependency self-check; cut the SSE payload by 74%

Self-check. Fan control failed for an entire session -- recoverably, and completely
invisibly. It appeared once, inside one field of one log line, and nothing ever
asked whether fan control worked. health.py now checks everything this service
depends on (NVML, passwordless sudo for nvidia-smi, fan control via the headless X
server, overclock drift, the telemetry store, residency measurement capability,
model directories, the ComfyUI websocket, and both upstream HTTP services) and
reports for each one what is broken, what that breaks, and how to fix it. Exposed at
GET /api/health, as an MCP tool, and as a dashboard panel that collapses to a badge
when healthy and expands to impact-and-fix when not. Current state: 9 ok, 1 degraded
(the known cachestat permission limit on Ollama's blobs).

A self-check that returns ok while a dependency is broken is worse than none, so the
tests drive each check to its failure state -- including the exact "Error resolving
target specification 'gpu:0'" string from the original incident -- and assert that a
check which raises surfaces as failed rather than taking down the endpoint.

SSE payload. The installed-model catalog was 10.6 KB of a 13.1 KB frame, 81% of the
stream, re-sent to every subscriber every second despite changing only when a model
is pulled or removed: 135 MB/hour across three tabs. It is now sent on a
subscriber's first frame and whenever the set changes; the client keeps the last
known list. Steady-state frames dropped from 14041 to 3664 bytes, a 74% reduction,
and /api/stats still returns the complete snapshot for API consumers.

Tests: 182 (was 169).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-05 19:15:52 -07:00
parent 25b601e24a
commit fb104ac9c0
6 changed files with 437 additions and 3 deletions

134
tests/test_health.py Normal file
View File

@@ -0,0 +1,134 @@
"""Tests for the dependency self-check.
This module exists because fan control failed for an entire session, recoverably and
invisibly: the service started before the headless X server that owns the GPU was
accepting connections, the assignment failed with "Error resolving target specification",
nothing retried, and nothing ever asked whether fan control worked. These tests make sure
each check reports the *right* status, since a self-check that returns ok when a
dependency is broken is worse than having none.
"""
import asyncio
import pytest
import health
import overclock_manager
class TestFanControlCheck:
"""The check that would have caught the original bug."""
def test_fails_when_headless_x_is_not_running(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False)
res = health._check_fan_control()
assert res["status"] == health.FAILED
# A bare failure is not enough; it has to say what breaks and how to fix it.
assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower()
assert res["fix"]
def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch):
# The exact nvidia-settings error seen at startup.
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
monkeypatch.setattr(overclock_manager, "get_fan_status",
lambda force=False: {"manual": False, "mode": "auto",
"target_speed_pct": None})
monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: {
"rc": 1, "out": "",
"err": "ERROR: Error resolving target specification 'gpu:0' "
"(No targets match target specification)"})
res = health._check_fan_control()
assert res["status"] == health.FAILED
def test_ok_when_fan_status_reads_back(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
monkeypatch.setattr(overclock_manager, "get_fan_status",
lambda force=False: {"manual": True, "mode": "manual",
"target_speed_pct": 70})
assert health._check_fan_control()["status"] == health.OK
class TestProfileDriftCheck:
def test_degraded_when_no_profile_applied_since_start(self, monkeypatch):
# ACTIVE_PROFILE defaults to "balanced" at import, which used to be
# indistinguishable from "balanced was applied successfully".
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": False, "drifted": True,
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
"reason": "no profile has been successfully applied since startup"})
assert health._check_profile_drift()["status"] == health.DEGRADED
def test_degraded_when_hardware_disagrees(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": True, "drifted": True,
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
"reason": "card reports 370.0W, profile asks 320W"})
res = health._check_profile_drift()
assert res["status"] == health.DEGRADED
assert "370" in res["detail"]
def test_ok_when_they_agree(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": True, "drifted": False,
"power_limit_intended_w": 320, "power_limit_actual_w": 320.0,
"reason": None})
assert health._check_profile_drift()["status"] == health.OK
class TestSudoCheck:
def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "_smi",
lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"})
res = health._check_sudo_smi()
assert res["status"] == health.FAILED
assert "sudo" in res["fix"].lower()
def test_ok_when_it_works(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "_smi",
lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""})
assert health._check_sudo_smi()["status"] == health.OK
class TestAggregation:
"""Overall status must be driven by the worst individual result."""
def _fake(self, statuses):
return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)]
@pytest.mark.parametrize("statuses,expected", [
([health.OK, health.OK], health.OK),
([health.OK, health.DEGRADED], health.DEGRADED),
([health.OK, health.FAILED], health.FAILED),
([health.DEGRADED, health.FAILED], health.FAILED),
])
def test_worst_status_wins(self, monkeypatch, statuses, expected):
checks = self._fake(statuses)
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
return health._check(name, health.OK, "reachable")
monkeypatch.setattr(health, "_check_http", fake_http)
res = asyncio.run(health.run_health_checks())
assert res["status"] == expected
def test_a_raising_check_does_not_break_the_report(self, monkeypatch):
def boom():
raise RuntimeError("nvml exploded")
monkeypatch.setattr(health, "_check_nvml", boom)
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
"_check_store", "_check_residency", "_check_model_dirs",
"_check_comfy_ws"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
return health._check(name, health.OK, "reachable")
monkeypatch.setattr(health, "_check_http", fake_http)
res = asyncio.run(health.run_health_checks())
# A broken check must surface as failed, not take down the endpoint.
assert res["status"] == health.FAILED
assert any("exploded" in c["detail"] for c in res["checks"])