The health check now reports what unmanaged VRAM actually costs rather than just how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit within 15.42 GB but not the 14.60 GB actually available", naming them. Getting that arithmetic right took a correction. The first version subtracted only the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB -- the same model the service had just refused with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process lives, which a purge does not free, so it is not available either. The floor is taken from the minimum ComfyUI VRAM in recent telemetry rather than its current value, which could be a 7 GB checkpoint mid-generation. The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the 30 s idle window, since a large model takes longer than that to load and the purge was freeing ComfyUI mid-load, so the reclaim path was never reached. Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model too large to fit on the card at all is not blamed on the third-party process. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
217 lines
11 KiB
Python
217 lines
11 KiB
Python
"""Tests for the dependency self-check.
|
|
|
|
This module exists because fan control failed for an entire session, recoverably and
|
|
invisibly: the service started before the headless X server that owns the GPU was
|
|
accepting connections, the assignment failed with "Error resolving target specification",
|
|
nothing retried, and nothing ever asked whether fan control worked. These tests make sure
|
|
each check reports the *right* status, since a self-check that returns ok when a
|
|
dependency is broken is worse than having none.
|
|
"""
|
|
import asyncio
|
|
|
|
import pytest
|
|
|
|
import health
|
|
import overclock_manager
|
|
|
|
|
|
class TestFanControlCheck:
|
|
"""The check that would have caught the original bug."""
|
|
|
|
def test_fails_when_headless_x_is_not_running(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False)
|
|
res = health._check_fan_control()
|
|
assert res["status"] == health.FAILED
|
|
# A bare failure is not enough; it has to say what breaks and how to fix it.
|
|
assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower()
|
|
assert res["fix"]
|
|
|
|
def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch):
|
|
# The exact nvidia-settings error seen at startup.
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
|
lambda force=False: {"manual": False, "mode": "auto",
|
|
"target_speed_pct": None})
|
|
monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: {
|
|
"rc": 1, "out": "",
|
|
"err": "ERROR: Error resolving target specification 'gpu:0' "
|
|
"(No targets match target specification)"})
|
|
res = health._check_fan_control()
|
|
assert res["status"] == health.FAILED
|
|
|
|
def test_ok_when_fan_status_reads_back(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
|
|
monkeypatch.setattr(overclock_manager, "get_fan_status",
|
|
lambda force=False: {"manual": True, "mode": "manual",
|
|
"target_speed_pct": 70})
|
|
assert health._check_fan_control()["status"] == health.OK
|
|
|
|
|
|
class TestProfileDriftCheck:
|
|
def test_degraded_when_no_profile_applied_since_start(self, monkeypatch):
|
|
# ACTIVE_PROFILE defaults to "balanced" at import, which used to be
|
|
# indistinguishable from "balanced was applied successfully".
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": False, "drifted": True,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
|
"reason": "no profile has been successfully applied since startup"})
|
|
assert health._check_profile_drift()["status"] == health.DEGRADED
|
|
|
|
def test_degraded_when_hardware_disagrees(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": True, "drifted": True,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
|
|
"reason": "card reports 370.0W, profile asks 320W"})
|
|
res = health._check_profile_drift()
|
|
assert res["status"] == health.DEGRADED
|
|
assert "370" in res["detail"]
|
|
|
|
def test_ok_when_they_agree(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
|
|
"profile": "balanced", "applied_since_start": True, "drifted": False,
|
|
"power_limit_intended_w": 320, "power_limit_actual_w": 320.0,
|
|
"reason": None})
|
|
assert health._check_profile_drift()["status"] == health.OK
|
|
|
|
|
|
class TestSudoCheck:
|
|
def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "_smi",
|
|
lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"})
|
|
res = health._check_sudo_smi()
|
|
assert res["status"] == health.FAILED
|
|
assert "sudo" in res["fix"].lower()
|
|
|
|
def test_ok_when_it_works(self, monkeypatch):
|
|
monkeypatch.setattr(overclock_manager, "_smi",
|
|
lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""})
|
|
assert health._check_sudo_smi()["status"] == health.OK
|
|
|
|
|
|
class TestAggregation:
|
|
"""Overall status must be driven by the worst individual result."""
|
|
|
|
def _fake(self, statuses):
|
|
return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)]
|
|
|
|
@pytest.mark.parametrize("statuses,expected", [
|
|
([health.OK, health.OK], health.OK),
|
|
([health.OK, health.DEGRADED], health.DEGRADED),
|
|
([health.OK, health.FAILED], health.FAILED),
|
|
([health.DEGRADED, health.FAILED], health.FAILED),
|
|
])
|
|
def test_worst_status_wins(self, monkeypatch, statuses, expected):
|
|
checks = self._fake(statuses)
|
|
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
|
|
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
|
|
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
|
|
"_check_residency", "_check_model_dirs", "_check_comfy_ws",
|
|
"_check_unmanaged_vram"):
|
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
|
|
|
async def fake_http(name, url, impact, fix):
|
|
return health._check(name, health.OK, "reachable")
|
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
|
|
|
res = asyncio.run(health.run_health_checks())
|
|
assert res["status"] == expected
|
|
|
|
def test_a_raising_check_does_not_break_the_report(self, monkeypatch):
|
|
def boom():
|
|
raise RuntimeError("nvml exploded")
|
|
monkeypatch.setattr(health, "_check_nvml", boom)
|
|
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
|
|
"_check_store", "_check_residency", "_check_model_dirs",
|
|
"_check_comfy_ws", "_check_unmanaged_vram"):
|
|
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
|
|
|
|
async def fake_http(name, url, impact, fix):
|
|
return health._check(name, health.OK, "reachable")
|
|
monkeypatch.setattr(health, "_check_http", fake_http)
|
|
|
|
res = asyncio.run(health.run_health_checks())
|
|
# A broken check must surface as failed, not take down the endpoint.
|
|
assert res["status"] == health.FAILED
|
|
assert any("exploded" in c["detail"] for c in res["checks"])
|
|
|
|
|
|
class TestUnmanagedVramCheck:
|
|
"""Turning an unreclaimable-VRAM number into something actionable.
|
|
|
|
The arithmetic here has to be right or the check is worse than useless. A first
|
|
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
|
|
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
|
|
model the service had just refused with 507 Insufficient Storage.
|
|
"""
|
|
|
|
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
|
|
procs=None):
|
|
return {
|
|
"available": True,
|
|
"vram_total_gb": total,
|
|
"breakdown": {
|
|
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
|
|
"comfyui_gb": comfy_gb,
|
|
"unmanaged": procs if procs is not None else
|
|
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
|
|
"cmdline": "stt_relay.py"}],
|
|
},
|
|
}
|
|
|
|
def _blobs(self, sizes):
|
|
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
|
|
|
|
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
|
|
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
|
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
|
|
assert health._check_unmanaged_vram()["status"] == health.OK
|
|
|
|
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
|
|
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
|
|
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
|
|
# the observed 507.
|
|
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
|
lambda: self._gpu())
|
|
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
|
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
|
lambda: self._blobs([12.87]))
|
|
res = health._check_unmanaged_vram()
|
|
assert res["status"] == health.DEGRADED
|
|
assert "1 model(s)" in res["impact"]
|
|
|
|
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
|
|
# A tiny model fits either way, so the unmanaged process is not what blocks it.
|
|
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
|
lambda: self._gpu())
|
|
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
|
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
|
lambda: self._blobs([2.0]))
|
|
assert health._check_unmanaged_vram()["status"] == health.OK
|
|
|
|
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
|
|
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
|
|
# process is why would send the user after the wrong thing.
|
|
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
|
|
lambda: self._gpu())
|
|
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
|
|
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
|
|
lambda: self._blobs([23.7]))
|
|
assert health._check_unmanaged_vram()["status"] == health.OK
|
|
|
|
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
|
|
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
|
|
# survives a purge.
|
|
monkeypatch.setattr(health.telemetry_store, "_rows",
|
|
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
|
|
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
|
|
|
|
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
|
|
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
|
|
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
|
|
|
|
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
|
|
def boom(*a, **k):
|
|
raise RuntimeError("db gone")
|
|
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
|
|
assert health._comfy_vram_floor_gb(default=0.5) == 0.5
|