Self-check. Fan control failed for an entire session -- recoverably, and completely invisibly. It appeared once, inside one field of one log line, and nothing ever asked whether fan control worked. health.py now checks everything this service depends on (NVML, passwordless sudo for nvidia-smi, fan control via the headless X server, overclock drift, the telemetry store, residency measurement capability, model directories, the ComfyUI websocket, and both upstream HTTP services) and reports for each one what is broken, what that breaks, and how to fix it. Exposed at GET /api/health, as an MCP tool, and as a dashboard panel that collapses to a badge when healthy and expands to impact-and-fix when not. Current state: 9 ok, 1 degraded (the known cachestat permission limit on Ollama's blobs). A self-check that returns ok while a dependency is broken is worse than none, so the tests drive each check to its failure state -- including the exact "Error resolving target specification 'gpu:0'" string from the original incident -- and assert that a check which raises surfaces as failed rather than taking down the endpoint. SSE payload. The installed-model catalog was 10.6 KB of a 13.1 KB frame, 81% of the stream, re-sent to every subscriber every second despite changing only when a model is pulled or removed: 135 MB/hour across three tabs. It is now sent on a subscriber's first frame and whenever the set changes; the client keeps the last known list. Steady-state frames dropped from 14041 to 3664 bytes, a 74% reduction, and /api/stats still returns the complete snapshot for API consumers. Tests: 182 (was 169). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
135 lines
6.5 KiB
Python
135 lines
6.5 KiB
Python
"""Tests for the dependency self-check.
|
|
|
|
This module exists because fan control failed for an entire session, recoverably and
|
|
invisibly: the service started before the headless X server that owns the GPU was
|
|
accepting connections, the assignment failed with "Error resolving target specification",
|
|
nothing retried, and nothing ever asked whether fan control worked. These tests make sure
|
|
each check reports the *right* status, since a self-check that returns ok when a
|
|
dependency is broken is worse than having none.
|
|
"""
|
|
import asyncio
|
|
|
|
import pytest
|
|
|
|
import health
|
|
import overclock_manager
|
|
|
|
|
|
class TestFanControlCheck:
|
|
"""The check that would have caught the original bug."""
|
|
|
|
def test_fails_when_headless_x_is_not_running(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False)
|
|
res = health._check_fan_control()
|
|
assert res["status"] == health.FAILED
|
|
# A bare failure is not enough; it has to say what breaks and how to fix it.
|
|
assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower()
|
|
assert res["fix"]
|
|
|
|
def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch):
|
|
# The exact nvidia-settings error seen at startup.
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
|
lambda force=False: {"manual": False, "mode": "auto",
|
|
"target_speed_pct": None})
|
|
monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: {
|
|
"rc": 1, "out": "",
|
|
"err": "ERROR: Error resolving target specification 'gpu:0' "
|
|
"(No targets match target specification)"})
|
|
res = health._check_fan_control()
|
|
assert res["status"] == health.FAILED
|
|
|
|
def test_ok_when_fan_status_reads_back(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
|
lambda force=False: {"manual": True, "mode": "manual",
|
|
"target_speed_pct": 70})
|
|
assert health._check_fan_control()["status"] == health.OK
|
|
|
|
|
|
class TestProfileDriftCheck:
|
|
def test_degraded_when_no_profile_applied_since_start(self, monkeypatch):
|
|
# ACTIVE_PROFILE defaults to "balanced" at import, which used to be
|
|
# indistinguishable from "balanced was applied successfully".
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": False, "drifted": True,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
|
"reason": "no profile has been successfully applied since startup"})
|
|
assert health._check_profile_drift()["status"] == health.DEGRADED
|
|
|
|
def test_degraded_when_hardware_disagrees(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": True, "drifted": True,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
|
"reason": "card reports 370.0W, profile asks 320W"})
|
|
res = health._check_profile_drift()
|
|
assert res["status"] == health.DEGRADED
|
|
assert "370" in res["detail"]
|
|
|
|
def test_ok_when_they_agree(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": True, "drifted": False,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 320.0,
|
|
"reason": None})
|
|
assert health._check_profile_drift()["status"] == health.OK
|
|
|
|
|
|
class TestSudoCheck:
|
|
def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "_smi",
|
|
lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"})
|
|
res = health._check_sudo_smi()
|
|
assert res["status"] == health.FAILED
|
|
assert "sudo" in res["fix"].lower()
|
|
|
|
def test_ok_when_it_works(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "_smi",
|
|
lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""})
|
|
assert health._check_sudo_smi()["status"] == health.OK
|
|
|
|
|
|
class TestAggregation:
|
|
"""Overall status must be driven by the worst individual result."""
|
|
|
|
def _fake(self, statuses):
|
|
return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)]
|
|
|
|
@pytest.mark.parametrize("statuses,expected", [
|
|
([health.OK, health.OK], health.OK),
|
|
([health.OK, health.DEGRADED], health.DEGRADED),
|
|
([health.OK, health.FAILED], health.FAILED),
|
|
([health.DEGRADED, health.FAILED], health.FAILED),
|
|
])
|
|
def test_worst_status_wins(self, monkeypatch, statuses, expected):
|
|
checks = self._fake(statuses)
|
|
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
|
|
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
|
|
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
|
|
"_check_residency", "_check_model_dirs", "_check_comfy_ws"):
|
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
|
|
|
async def fake_http(name, url, impact, fix):
|
|
return health._check(name, health.OK, "reachable")
|
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
|
|
|
res = asyncio.run(health.run_health_checks())
|
|
assert res["status"] == expected
|
|
|
|
def test_a_raising_check_does_not_break_the_report(self, monkeypatch):
|
|
def boom():
|
|
raise RuntimeError("nvml exploded")
|
|
monkeypatch.setattr(health, "_check_nvml", boom)
|
|
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
|
|
"_check_store", "_check_residency", "_check_model_dirs",
|
|
"_check_comfy_ws"):
|
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
|
|
|
async def fake_http(name, url, impact, fix):
|
|
return health._check(name, health.OK, "reachable")
|
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
|
|
|
res = asyncio.run(health.run_health_checks())
|
|
# A broken check must surface as failed, not take down the endpoint.
|
|
assert res["status"] == health.FAILED
|
|
assert any("exploded" in c["detail"] for c in res["checks"])
|