"""Tests for the dependency self-check. This module exists because fan control failed for an entire session, recoverably and invisibly: the service started before the headless X server that owns the GPU was accepting connections, the assignment failed with "Error resolving target specification", nothing retried, and nothing ever asked whether fan control worked. These tests make sure each check reports the *right* status, since a self-check that returns ok when a dependency is broken is worse than having none. """ import asyncio import pytest import health import overclock_manager class TestFanControlCheck: """The check that would have caught the original bug.""" def test_fails_when_headless_x_is_not_running(self, monkeypatch): monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False) res = health._check_fan_control() assert res["status"] == health.FAILED # A bare failure is not enough; it has to say what breaks and how to fix it. assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower() assert res["fix"] def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch): # The exact nvidia-settings error seen at startup. monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) monkeypatch.setattr(overclock_manager, "get_fan_status", lambda force=False: {"manual": False, "mode": "auto", "target_speed_pct": None}) monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: { "rc": 1, "out": "", "err": "ERROR: Error resolving target specification 'gpu:0' " "(No targets match target specification)"}) res = health._check_fan_control() assert res["status"] == health.FAILED def test_ok_when_fan_status_reads_back(self, monkeypatch): monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True) monkeypatch.setattr(overclock_manager, "get_fan_status", lambda force=False: {"manual": True, "mode": "manual", "target_speed_pct": 70}) assert health._check_fan_control()["status"] == health.OK class TestProfileDriftCheck: def test_degraded_when_no_profile_applied_since_start(self, monkeypatch): # ACTIVE_PROFILE defaults to "balanced" at import, which used to be # indistinguishable from "balanced was applied successfully". monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": False, "drifted": True, "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, "reason": "no profile has been successfully applied since startup"}) assert health._check_profile_drift()["status"] == health.DEGRADED def test_degraded_when_hardware_disagrees(self, monkeypatch): monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": True, "drifted": True, "power_limit_intended_w": 320, "power_limit_actual_w": 370.0, "reason": "card reports 370.0W, profile asks 320W"}) res = health._check_profile_drift() assert res["status"] == health.DEGRADED assert "370" in res["detail"] def test_ok_when_they_agree(self, monkeypatch): monkeypatch.setattr(overclock_manager, "profile_drift", lambda: { "profile": "balanced", "applied_since_start": True, "drifted": False, "power_limit_intended_w": 320, "power_limit_actual_w": 320.0, "reason": None}) assert health._check_profile_drift()["status"] == health.OK class TestSudoCheck: def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch): monkeypatch.setattr(overclock_manager, "_smi", lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"}) res = health._check_sudo_smi() assert res["status"] == health.FAILED assert "sudo" in res["fix"].lower() def test_ok_when_it_works(self, monkeypatch): monkeypatch.setattr(overclock_manager, "_smi", lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""}) assert health._check_sudo_smi()["status"] == health.OK class TestAggregation: """Overall status must be driven by the worst individual result.""" def _fake(self, statuses): return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)] @pytest.mark.parametrize("statuses,expected", [ ([health.OK, health.OK], health.OK), ([health.OK, health.DEGRADED], health.DEGRADED), ([health.OK, health.FAILED], health.FAILED), ([health.DEGRADED, health.FAILED], health.FAILED), ]) def test_worst_status_wins(self, monkeypatch, statuses, expected): checks = self._fake(statuses) monkeypatch.setattr(health, "_check_nvml", lambda: checks[0]) monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1]) for fn in ("_check_fan_control", "_check_profile_drift", "_check_store", "_check_residency", "_check_model_dirs", "_check_comfy_ws", "_check_unmanaged_vram", "_check_comfy_queue"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): return health._check(name, health.OK, "reachable") monkeypatch.setattr(health, "_check_http", fake_http) res = asyncio.run(health.run_health_checks()) assert res["status"] == expected def test_a_raising_check_does_not_break_the_report(self, monkeypatch): def boom(): raise RuntimeError("nvml exploded") monkeypatch.setattr(health, "_check_nvml", boom) for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift", "_check_store", "_check_residency", "_check_model_dirs", "_check_comfy_ws", "_check_unmanaged_vram", "_check_comfy_queue"): monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d")) async def fake_http(name, url, impact, fix): return health._check(name, health.OK, "reachable") monkeypatch.setattr(health, "_check_http", fake_http) res = asyncio.run(health.run_health_checks()) # A broken check must surface as failed, not take down the endpoint. assert res["status"] == health.FAILED assert any("exploded" in c["detail"] for c in res["checks"]) class TestUnmanagedVramCheck: """Turning an unreclaimable-VRAM number into something actionable. The arithmetic here has to be right or the check is worse than useless. A first version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very model the service had just refused with 507 Insufficient Storage. """ def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99, procs=None): return { "available": True, "vram_total_gb": total, "breakdown": { "unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb, "comfyui_gb": comfy_gb, "unmanaged": procs if procs is not None else [{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024, "cmdline": "stt_relay.py"}], }, } def _blobs(self, sizes): return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)] def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch): monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", lambda: self._gpu(unmanaged_gb=0.0, procs=[])) assert health._check_unmanaged_vram()["status"] == health.OK def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch): # 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available. # A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching # the observed 507. monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", lambda: self._gpu()) monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", lambda: self._blobs([12.87])) res = health._check_unmanaged_vram() assert res["status"] == health.DEGRADED assert "1 model(s)" in res["impact"] def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch): # A tiny model fits either way, so the unmanaged process is not what blocks it. monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", lambda: self._gpu()) monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", lambda: self._blobs([2.0])) assert health._check_unmanaged_vram()["status"] == health.OK def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch): # A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB # process is why would send the user after the wrong thing. monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats", lambda: self._gpu()) monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56) monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files", lambda: self._blobs([23.7])) assert health._check_unmanaged_vram()["status"] == health.OK def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch): # Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what # survives a purge. monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}]) assert health._comfy_vram_floor_gb(default=7.0) == 0.24 def test_floor_falls_back_when_history_is_empty(self, monkeypatch): monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: []) assert health._comfy_vram_floor_gb(default=0.56) == 0.56 def test_floor_falls_back_rather_than_raising(self, monkeypatch): def boom(*a, **k): raise RuntimeError("db gone") monkeypatch.setattr(health.telemetry_store, "_rows", boom) assert health._comfy_vram_floor_gb(default=0.5) == 0.5