Files
gpu-program-swapper/tests/test_classify_load.py
drjones 868d82794d Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-01 13:40:30 -07:00

124 lines
5.1 KiB
Python

"""Regression tests for vram_arbitrator.classify_load.
classify_load turns (model size, Ollama's reported load_duration) into a cache verdict
by computing achieved bandwidth. The two thresholds it compares against were measured on
this box, not chosen; the tests below pin the measurements themselves so a future edit
to RAM_HIT_GBPS / PARTIAL_HIT_GBPS that breaks the real data points fails loudly.
"""
import math
import pytest
import vram_arbitrator as va
GIB = 1024 ** 3
# Ground truth, measured 2026-08-28 on the same 12.87 GB model loaded twice:
# 3.1% resident -> 34267 ms -> 0.38 GB/s -> cold
# 100% resident -> 4901 ms -> 2.63 GB/s -> RAM hit
MEASURED_MODEL_BYTES = int(12.87 * GIB)
MEASURED_COLD_MS = 34267.0
MEASURED_WARM_MS = 4901.0
def test_measured_cold_load_classifies_as_cold_disk():
"""The measured cold load (12.87 GB, 34267 ms, 0.38 GB/s) must stay 'Cold Disk Load'."""
res = va.classify_load(MEASURED_MODEL_BYTES, MEASURED_COLD_MS)
assert res["cache_status"] == "Cold Disk Load 💾"
assert res["is_ram_hit"] is False
assert res["load_gbps"] == pytest.approx(0.38, abs=0.01)
def test_measured_warm_load_classifies_as_ram_hit():
"""The measured warm load (12.87 GB, 4901 ms, 2.63 GB/s) must stay a RAM cache hit."""
res = va.classify_load(MEASURED_MODEL_BYTES, MEASURED_WARM_MS)
assert res["cache_status"] == "RAM Cache Hit ⚡"
assert res["is_ram_hit"] is True
assert res["load_gbps"] == pytest.approx(2.63, abs=0.01)
def test_ram_hit_threshold_is_physically_achievable():
"""Guard against the reintroduced bug where the hit bar sat above any real warm load.
A prior version set the hit threshold at 5.0 GB/s. Ollama's load_duration covers
host-to-device transfer and model init as well as the file read, so a fully resident
12.87 GB model still only reports ~2.6 GB/s — meaning *no* load could ever be
classified as a hit. Any threshold that a genuinely warm load cannot clear is wrong.
"""
assert va.RAM_HIT_GBPS <= 2.63, (
"RAM_HIT_GBPS is above the fastest warm load ever measured on this box "
"(2.63 GB/s); no load could be classified as a cache hit."
)
assert va.PARTIAL_HIT_GBPS > 0.38, (
"PARTIAL_HIT_GBPS is at or below the measured cold-disk rate; cold loads would "
"be reported as partial cache hits."
)
assert va.PARTIAL_HIT_GBPS < va.RAM_HIT_GBPS
def test_sub_millisecond_load_is_already_in_vram():
"""load_duration_ms <= 1.0 means Ollama never re-read the model at all."""
res = va.classify_load(MEASURED_MODEL_BYTES, 1.0)
assert res["cache_status"] == "Already in VRAM"
assert res["load_gbps"] is None
assert res["is_ram_hit"] is True
def test_zero_duration_is_already_in_vram():
res = va.classify_load(MEASURED_MODEL_BYTES, 0.0)
assert res["cache_status"] == "Already in VRAM"
def test_boundary_exactly_at_ram_hit_threshold_is_a_hit():
"""Exactly RAM_HIT_GBPS (2.0 GB/s) must count as a hit — the comparison is >=."""
size = int(va.RAM_HIT_GBPS * GIB) # 2 GiB read in exactly 1000 ms -> 2.00 GB/s
res = va.classify_load(size, 1000.0)
assert res["load_gbps"] == pytest.approx(2.0)
assert res["cache_status"] == "RAM Cache Hit ⚡"
assert res["is_ram_hit"] is True
def test_just_below_ram_hit_threshold_is_partial():
size = int(1.99 * GIB)
res = va.classify_load(size, 1000.0)
assert res["cache_status"] == "Partial Cache 🌤"
assert res["is_ram_hit"] is False
def test_boundary_exactly_at_partial_threshold_is_partial():
"""Exactly PARTIAL_HIT_GBPS (0.8 GB/s) is still a partial hit, not a cold load."""
size = math.ceil(va.PARTIAL_HIT_GBPS * GIB) # 0.8 GiB is not a whole number of bytes
res = va.classify_load(size, 1000.0)
assert res["load_gbps"] == pytest.approx(0.8)
assert res["cache_status"] == "Partial Cache 🌤"
assert res["is_ram_hit"] is False
def test_just_below_partial_threshold_is_cold():
size = int(0.79 * GIB)
res = va.classify_load(size, 1000.0)
assert res["cache_status"] == "Cold Disk Load 💾"
@pytest.mark.parametrize("duration_ms,expected_hit", [(4901.0, True), (34267.0, False)])
def test_unknown_size_falls_back_to_duration_guess_and_says_so(duration_ms, expected_hit):
"""With no size on record there is no bandwidth to compute, so the result must be
labelled as a guess rather than presented as a measurement."""
res = va.classify_load(0, duration_ms)
assert res["is_ram_hit"] is expected_hit
assert res["load_gbps"] is None
assert "detail" in res and "guess" in res["detail"].lower()
def test_unknown_size_guess_boundary_is_8s():
"""The size-unknown fallback splits at 8000 ms, between the measured 4.9 s warm and
34.3 s cold loads."""
assert va.classify_load(0, 7999.0)["is_ram_hit"] is True
assert va.classify_load(0, 8000.0)["is_ram_hit"] is False
def test_classification_never_raises_on_odd_inputs():
"""This runs on the swap hot path; it must not be able to throw."""
for size, dur in [(0, 0.0), (1, 1.5), (10 ** 12, 2.0), (0, 1.0)]:
assert "cache_status" in va.classify_load(size, dur)