Files
gpu-program-swapper/tests/test_health.py
drjones aeba1b47fd Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context
The health check now reports what unmanaged VRAM actually costs rather than just
how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit
within 15.42 GB but not the 14.60 GB actually available", naming them.

Getting that arithmetic right took a correction. The first version subtracted only
the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting
against a real ceiling of 14.60 GB -- the same model the service had just refused
with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process
lives, which a purge does not free, so it is not available either. The floor is taken
from the minimum ComfyUI VRAM in recent telemetry rather than its current value,
which could be a 7 GB checkpoint mid-generation.

The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the
30 s idle window, since a large model takes longer than that to load and the purge
was freeing ComfyUI mid-load, so the reclaim path was never reached.

Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model
too large to fit on the card at all is not blamed on the third-party process.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-06 17:44:21 -07:00

217 lines
11 KiB
Python

"""Tests for the dependency self-check.
This module exists because fan control failed for an entire session, recoverably and
invisibly: the service started before the headless X server that owns the GPU was
accepting connections, the assignment failed with "Error resolving target specification",
nothing retried, and nothing ever asked whether fan control worked. These tests make sure
each check reports the *right* status, since a self-check that returns ok when a
dependency is broken is worse than having none.
"""
import asyncio
import pytest
import health
import overclock_manager
class TestFanControlCheck:
"""The check that would have caught the original bug."""
def test_fails_when_headless_x_is_not_running(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: False)
res = health._check_fan_control()
assert res["status"] == health.FAILED
# A bare failure is not enough; it has to say what breaks and how to fix it.
assert "governor" in res["impact"].lower() or "fan" in res["impact"].lower()
assert res["fix"]
def test_fails_when_the_gpu_target_cannot_be_resolved(self, monkeypatch):
# The exact nvidia-settings error seen at startup.
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
monkeypatch.setattr(overclock_manager, "get_fan_status",
lambda force=False: {"manual": False, "mode": "auto",
"target_speed_pct": None})
monkeypatch.setattr(overclock_manager, "_nvidia_settings", lambda *a, **k: {
"rc": 1, "out": "",
"err": "ERROR: Error resolving target specification 'gpu:0' "
"(No targets match target specification)"})
res = health._check_fan_control()
assert res["status"] == health.FAILED
def test_ok_when_fan_status_reads_back(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "is_headless_x_running", lambda: True)
monkeypatch.setattr(overclock_manager, "get_fan_status",
lambda force=False: {"manual": True, "mode": "manual",
"target_speed_pct": 70})
assert health._check_fan_control()["status"] == health.OK
class TestProfileDriftCheck:
def test_degraded_when_no_profile_applied_since_start(self, monkeypatch):
# ACTIVE_PROFILE defaults to "balanced" at import, which used to be
# indistinguishable from "balanced was applied successfully".
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": False, "drifted": True,
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
"reason": "no profile has been successfully applied since startup"})
assert health._check_profile_drift()["status"] == health.DEGRADED
def test_degraded_when_hardware_disagrees(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": True, "drifted": True,
"power_limit_intended_w": 320, "power_limit_actual_w": 370.0,
"reason": "card reports 370.0W, profile asks 320W"})
res = health._check_profile_drift()
assert res["status"] == health.DEGRADED
assert "370" in res["detail"]
def test_ok_when_they_agree(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "profile_drift", lambda: {
"profile": "balanced", "applied_since_start": True, "drifted": False,
"power_limit_intended_w": 320, "power_limit_actual_w": 320.0,
"reason": None})
assert health._check_profile_drift()["status"] == health.OK
class TestSudoCheck:
def test_failed_when_sudo_smi_returns_nonzero(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "_smi",
lambda *a: {"rc": 1, "out": "", "err": "sudo: a password is required"})
res = health._check_sudo_smi()
assert res["status"] == health.FAILED
assert "sudo" in res["fix"].lower()
def test_ok_when_it_works(self, monkeypatch):
monkeypatch.setattr(overclock_manager, "_smi",
lambda *a: {"rc": 0, "out": "NVIDIA GeForce RTX 4080 SUPER", "err": ""})
assert health._check_sudo_smi()["status"] == health.OK
class TestAggregation:
"""Overall status must be driven by the worst individual result."""
def _fake(self, statuses):
return [health._check(f"c{i}", s, "d") for i, s in enumerate(statuses)]
@pytest.mark.parametrize("statuses,expected", [
([health.OK, health.OK], health.OK),
([health.OK, health.DEGRADED], health.DEGRADED),
([health.OK, health.FAILED], health.FAILED),
([health.DEGRADED, health.FAILED], health.FAILED),
])
def test_worst_status_wins(self, monkeypatch, statuses, expected):
checks = self._fake(statuses)
monkeypatch.setattr(health, "_check_nvml", lambda: checks[0])
monkeypatch.setattr(health, "_check_sudo_smi", lambda: checks[1])
for fn in ("_check_fan_control", "_check_profile_drift", "_check_store",
"_check_residency", "_check_model_dirs", "_check_comfy_ws",
"_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
return health._check(name, health.OK, "reachable")
monkeypatch.setattr(health, "_check_http", fake_http)
res = asyncio.run(health.run_health_checks())
assert res["status"] == expected
def test_a_raising_check_does_not_break_the_report(self, monkeypatch):
def boom():
raise RuntimeError("nvml exploded")
monkeypatch.setattr(health, "_check_nvml", boom)
for fn in ("_check_sudo_smi", "_check_fan_control", "_check_profile_drift",
"_check_store", "_check_residency", "_check_model_dirs",
"_check_comfy_ws", "_check_unmanaged_vram"):
monkeypatch.setattr(health, fn, lambda: health._check("x", health.OK, "d"))
async def fake_http(name, url, impact, fix):
return health._check(name, health.OK, "reachable")
monkeypatch.setattr(health, "_check_http", fake_http)
res = asyncio.run(health.run_health_checks())
# A broken check must surface as failed, not take down the endpoint.
assert res["status"] == health.FAILED
assert any("exploded" in c["detail"] for c in res["checks"])
class TestUnmanagedVramCheck:
"""Turning an unreclaimable-VRAM number into something actionable.
The arithmetic here has to be right or the check is worse than useless. A first
version omitted ComfyUI's CUDA context -- which survives a purge -- and so reported
a 14.93 GB model as fitting against a real ceiling of 14.60 GB. That was the very
model the service had just refused with 507 Insufficient Storage.
"""
def _gpu(self, unmanaged_gb=0.82, desktop_gb=0.01, comfy_gb=0.56, total=15.99,
procs=None):
return {
"available": True,
"vram_total_gb": total,
"breakdown": {
"unmanaged_gb": unmanaged_gb, "desktop_gb": desktop_gb,
"comfyui_gb": comfy_gb,
"unmanaged": procs if procs is not None else
[{"pid": 1, "name": "python", "vram_mb": unmanaged_gb * 1024,
"cmdline": "stt_relay.py"}],
},
}
def _blobs(self, sizes):
return [{"model": f"m{i}", "size_gb": s} for i, s in enumerate(sizes)]
def test_ok_when_nothing_holds_unreclaimable_vram(self, monkeypatch):
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu(unmanaged_gb=0.0, procs=[]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_comfy_cuda_context_counts_against_the_ceiling(self, monkeypatch):
# 15.99 - 0.82 unmanaged - 0.01 desktop - 0.56 comfy floor = 14.60 GB available.
# A 12.87 GB blob needs 12.87 * 1.16 = 14.93 GB, so it does not fit -- matching
# the observed 507.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([12.87]))
res = health._check_unmanaged_vram()
assert res["status"] == health.DEGRADED
assert "1 model(s)" in res["impact"]
def test_model_that_fits_even_without_the_unmanaged_process_is_not_flagged(self, monkeypatch):
# A tiny model fits either way, so the unmanaged process is not what blocks it.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([2.0]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_model_too_big_to_ever_fit_is_not_blamed_on_the_process(self, monkeypatch):
# A 23.7 GB model does not fit on a 16 GB card regardless; saying the 842 MB
# process is why would send the user after the wrong thing.
monkeypatch.setattr(health.vram_arbitrator, "get_gpu_hardware_stats",
lambda: self._gpu())
monkeypatch.setattr(health, "_comfy_vram_floor_gb", lambda default=0, days=1: 0.56)
monkeypatch.setattr(health.ram_optimizer, "find_ollama_model_files",
lambda: self._blobs([23.7]))
assert health._check_unmanaged_vram()["status"] == health.OK
def test_floor_uses_the_minimum_observed_not_the_current_value(self, monkeypatch):
# Current VRAM could be a 7 GB checkpoint mid-generation; the floor is what
# survives a purge.
monkeypatch.setattr(health.telemetry_store, "_rows",
lambda *a, **k: [{"floor": int(0.24 * 1024 ** 3)}])
assert health._comfy_vram_floor_gb(default=7.0) == 0.24
def test_floor_falls_back_when_history_is_empty(self, monkeypatch):
monkeypatch.setattr(health.telemetry_store, "_rows", lambda *a, **k: [])
assert health._comfy_vram_floor_gb(default=0.56) == 0.56
def test_floor_falls_back_rather_than_raising(self, monkeypatch):
def boom(*a, **k):
raise RuntimeError("db gone")
monkeypatch.setattr(health.telemetry_store, "_rows", boom)
assert health._comfy_vram_floor_gb(default=0.5) == 0.5