Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
123
tests/test_classify_load.py
Normal file
123
tests/test_classify_load.py
Normal file
@@ -0,0 +1,123 @@
|
||||
"""Regression tests for vram_arbitrator.classify_load.
|
||||
|
||||
classify_load turns (model size, Ollama's reported load_duration) into a cache verdict
|
||||
by computing achieved bandwidth. The two thresholds it compares against were measured on
|
||||
this box, not chosen; the tests below pin the measurements themselves so a future edit
|
||||
to RAM_HIT_GBPS / PARTIAL_HIT_GBPS that breaks the real data points fails loudly.
|
||||
"""
|
||||
import math
|
||||
|
||||
import pytest
|
||||
|
||||
import vram_arbitrator as va
|
||||
|
||||
GIB = 1024 ** 3
|
||||
|
||||
# Ground truth, measured 2026-08-28 on the same 12.87 GB model loaded twice:
|
||||
# 3.1% resident -> 34267 ms -> 0.38 GB/s -> cold
|
||||
# 100% resident -> 4901 ms -> 2.63 GB/s -> RAM hit
|
||||
MEASURED_MODEL_BYTES = int(12.87 * GIB)
|
||||
MEASURED_COLD_MS = 34267.0
|
||||
MEASURED_WARM_MS = 4901.0
|
||||
|
||||
|
||||
def test_measured_cold_load_classifies_as_cold_disk():
|
||||
"""The measured cold load (12.87 GB, 34267 ms, 0.38 GB/s) must stay 'Cold Disk Load'."""
|
||||
res = va.classify_load(MEASURED_MODEL_BYTES, MEASURED_COLD_MS)
|
||||
assert res["cache_status"] == "Cold Disk Load 💾"
|
||||
assert res["is_ram_hit"] is False
|
||||
assert res["load_gbps"] == pytest.approx(0.38, abs=0.01)
|
||||
|
||||
|
||||
def test_measured_warm_load_classifies_as_ram_hit():
|
||||
"""The measured warm load (12.87 GB, 4901 ms, 2.63 GB/s) must stay a RAM cache hit."""
|
||||
res = va.classify_load(MEASURED_MODEL_BYTES, MEASURED_WARM_MS)
|
||||
assert res["cache_status"] == "RAM Cache Hit ⚡"
|
||||
assert res["is_ram_hit"] is True
|
||||
assert res["load_gbps"] == pytest.approx(2.63, abs=0.01)
|
||||
|
||||
|
||||
def test_ram_hit_threshold_is_physically_achievable():
|
||||
"""Guard against the reintroduced bug where the hit bar sat above any real warm load.
|
||||
|
||||
A prior version set the hit threshold at 5.0 GB/s. Ollama's load_duration covers
|
||||
host-to-device transfer and model init as well as the file read, so a fully resident
|
||||
12.87 GB model still only reports ~2.6 GB/s — meaning *no* load could ever be
|
||||
classified as a hit. Any threshold that a genuinely warm load cannot clear is wrong.
|
||||
"""
|
||||
assert va.RAM_HIT_GBPS <= 2.63, (
|
||||
"RAM_HIT_GBPS is above the fastest warm load ever measured on this box "
|
||||
"(2.63 GB/s); no load could be classified as a cache hit."
|
||||
)
|
||||
assert va.PARTIAL_HIT_GBPS > 0.38, (
|
||||
"PARTIAL_HIT_GBPS is at or below the measured cold-disk rate; cold loads would "
|
||||
"be reported as partial cache hits."
|
||||
)
|
||||
assert va.PARTIAL_HIT_GBPS < va.RAM_HIT_GBPS
|
||||
|
||||
|
||||
def test_sub_millisecond_load_is_already_in_vram():
|
||||
"""load_duration_ms <= 1.0 means Ollama never re-read the model at all."""
|
||||
res = va.classify_load(MEASURED_MODEL_BYTES, 1.0)
|
||||
assert res["cache_status"] == "Already in VRAM"
|
||||
assert res["load_gbps"] is None
|
||||
assert res["is_ram_hit"] is True
|
||||
|
||||
|
||||
def test_zero_duration_is_already_in_vram():
|
||||
res = va.classify_load(MEASURED_MODEL_BYTES, 0.0)
|
||||
assert res["cache_status"] == "Already in VRAM"
|
||||
|
||||
|
||||
def test_boundary_exactly_at_ram_hit_threshold_is_a_hit():
|
||||
"""Exactly RAM_HIT_GBPS (2.0 GB/s) must count as a hit — the comparison is >=."""
|
||||
size = int(va.RAM_HIT_GBPS * GIB) # 2 GiB read in exactly 1000 ms -> 2.00 GB/s
|
||||
res = va.classify_load(size, 1000.0)
|
||||
assert res["load_gbps"] == pytest.approx(2.0)
|
||||
assert res["cache_status"] == "RAM Cache Hit ⚡"
|
||||
assert res["is_ram_hit"] is True
|
||||
|
||||
|
||||
def test_just_below_ram_hit_threshold_is_partial():
|
||||
size = int(1.99 * GIB)
|
||||
res = va.classify_load(size, 1000.0)
|
||||
assert res["cache_status"] == "Partial Cache 🌤"
|
||||
assert res["is_ram_hit"] is False
|
||||
|
||||
|
||||
def test_boundary_exactly_at_partial_threshold_is_partial():
|
||||
"""Exactly PARTIAL_HIT_GBPS (0.8 GB/s) is still a partial hit, not a cold load."""
|
||||
size = math.ceil(va.PARTIAL_HIT_GBPS * GIB) # 0.8 GiB is not a whole number of bytes
|
||||
res = va.classify_load(size, 1000.0)
|
||||
assert res["load_gbps"] == pytest.approx(0.8)
|
||||
assert res["cache_status"] == "Partial Cache 🌤"
|
||||
assert res["is_ram_hit"] is False
|
||||
|
||||
|
||||
def test_just_below_partial_threshold_is_cold():
|
||||
size = int(0.79 * GIB)
|
||||
res = va.classify_load(size, 1000.0)
|
||||
assert res["cache_status"] == "Cold Disk Load 💾"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("duration_ms,expected_hit", [(4901.0, True), (34267.0, False)])
|
||||
def test_unknown_size_falls_back_to_duration_guess_and_says_so(duration_ms, expected_hit):
|
||||
"""With no size on record there is no bandwidth to compute, so the result must be
|
||||
labelled as a guess rather than presented as a measurement."""
|
||||
res = va.classify_load(0, duration_ms)
|
||||
assert res["is_ram_hit"] is expected_hit
|
||||
assert res["load_gbps"] is None
|
||||
assert "detail" in res and "guess" in res["detail"].lower()
|
||||
|
||||
|
||||
def test_unknown_size_guess_boundary_is_8s():
|
||||
"""The size-unknown fallback splits at 8000 ms, between the measured 4.9 s warm and
|
||||
34.3 s cold loads."""
|
||||
assert va.classify_load(0, 7999.0)["is_ram_hit"] is True
|
||||
assert va.classify_load(0, 8000.0)["is_ram_hit"] is False
|
||||
|
||||
|
||||
def test_classification_never_raises_on_odd_inputs():
|
||||
"""This runs on the swap hot path; it must not be able to throw."""
|
||||
for size, dur in [(0, 0.0), (1, 1.5), (10 ** 12, 2.0), (0, 1.0)]:
|
||||
assert "cache_status" in va.classify_load(size, dur)
|
||||
Reference in New Issue
Block a user