"""Tests for the dependency self-check. This module exists because fan control failed for an entire session, recoverably and invisibly: the service started before the headless X server that owns the GPU was accepting connections, the assignment failed with "Error resolving target specification", nothing retried, and nothing ever asked whether fan control worked. These tests make sure each check reports the *right* status, since a self-check that returns ok when a dependency is broken is worse than having none. """ import asyncio import pytest import health import overclock_manager class TestFanControlCheck: """The check that would have caught the original bug.""" def test_fails_when_headless_x_is_not_running(self, monkeypatch): monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False) res = health._check_fan_control() assert res["status"] == health.FAILED # A bare failure is not enough; it has to say what breaks and how to fix it. assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower() assert res["fix"] def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch): # The exact nvidia-settings error seen at startup. monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) monkeypatch.setattr(overclock_manager, "get_fan_status", lambda force=False: {"manual": False, "mode": "auto", "target_speed_pct": None}) monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: { "rc": 1, "out": "", "err": "ERROR: Error resolving target specification 'gpu:0' " "(No targets match target specification)"}) res = health._check_fan_control() assert res["status"] == health.FAILED def test_ok_when_fan_status_reads_back(self, monkeypatch): monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) monkeypatch.setattr(overclock_manager, "get_fan_status", lambda force=False: {"manual": True, "mode": "manual", "target_speed_pct": 70}) assert health._check_fan_control()["status"] == health.OK class TestProfileDriftCheck: def test_degraded_when_no_profile_applied_since_start(self, monkeypatch): # ACTIVE_PROFILE defaults to "balanced" at import, which used to be # indistinguishable from "balanced was applied successfully". monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": False, "drifted": True, "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, "reason": "no profile has been successfully applied since startup"}) assert health._check_profile_drift()["status"] == health.DEGRADED def test_degraded_when_hardware_disagrees(self, monkeypatch): monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": True, "drifted": True, "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, "reason": "card reports 370.0W, profile asks 320W"}) res = health._check_profile_drift() assert res["status"] == health.DEGRADED assert "370" in res["detail"] def test_ok_when_they_agree(self, monkeypatch): monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": True, "drifted": False, "power_limit_intended_w": 320, "power_limit_actual_w": 320.0, "reason": None}) assert health._check_profile_drift()["status"] == health.OK class TestSudoCheck: def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch): monkeypatch.setattr(overclock_manager, "_smi", lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"}) res = health._check_sudo_smi() assert res["status"] == health.FAILED assert "sudo" in res["fix"].lower() def test_ok_when_it_works(self, monkeypatch): monkeypatch.setattr(overclock_manager, "_smi", lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""}) assert health._check_sudo_smi()["status"] == health.OK class TestAggregation: """Overall status must be driven by the worst individual result.""" def _fake(self, statuses): return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)] @pytest.mark.parametrize("statuses,expected", [ ([health.OK, health.OK], health.OK), ([health.OK, health.DEGRADED], health.DEGRADED), ([health.OK, health.FAILED], health.FAILED), ([health.DEGRADED, health.FAILED], health.FAILED), ]) def test_worst_status_wins(self, monkeypatch, statuses, expected): checks = self._fake(statuses) monkeypatch.setattr(health, "_check_nvml", lambda: checks[0]) monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1]) for fn in ("_check_fan_control", "_check_profile_drift", "_check_store", "_check_residency", "_check_model_dirs", "_check_comfy_ws"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): return health._check(name, health.OK, "reachable") monkeypatch.setattr(health, "_check_http", fake_http) res = asyncio.run(health.run_health_checks()) assert res["status"] == expected def test_a_raising_check_does_not_break_the_report(self, monkeypatch): def boom(): raise RuntimeError("nvml exploded") monkeypatch.setattr(health, "_check_nvml", boom) for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift", "_check_store", "_check_residency", "_check_model_dirs", "_check_comfy_ws"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): return health._check(name, health.OK, "reachable") monkeypatch.setattr(health, "_check_http", fake_http) res = asyncio.run(health.run_health_checks()) # A broken check must surface as failed, not take down the endpoint. assert res["status"] == health.FAILED assert any("exploded" in c["detail"] for c in res["checks"])