Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
184
tests/test_autotune_helpers.py
Normal file
184
tests/test_autotune_helpers.py
Normal file
@@ -0,0 +1,184 @@
|
||||
"""autotune's pure/parsable helpers.
|
||||
|
||||
Only the parsing and sampling helpers are exercised. Nothing here runs a sweep, applies a
|
||||
profile, or talks to nvidia-smi: subprocess.run is monkeypatched at the module boundary.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
import autotune
|
||||
|
||||
|
||||
class _FakeProc:
|
||||
def __init__(self, stdout="", returncode=0, stderr=""):
|
||||
self.stdout = stdout
|
||||
self.stderr = stderr
|
||||
self.returncode = returncode
|
||||
|
||||
|
||||
def _fake_smi(monkeypatch, stdout, record=None):
|
||||
def _run(cmd, capture_output=True, text=True, timeout=None, **kw):
|
||||
if record is not None:
|
||||
record.append(list(cmd))
|
||||
return _FakeProc(stdout)
|
||||
monkeypatch.setattr(autotune.subprocess, "run", _run)
|
||||
|
||||
|
||||
# A trimmed but realistically shaped `nvidia-smi --query-supported-clocks=mem,gr` dump:
|
||||
# graphics clocks are enumerated once per memory clock.
|
||||
SUPPORTED_CLOCKS_CSV = """\
|
||||
10501, 2790
|
||||
10501, 2775
|
||||
10501, 2760
|
||||
9501, 2790
|
||||
9501, 2775
|
||||
405, 645
|
||||
405, 630
|
||||
"""
|
||||
|
||||
|
||||
def test_supported_clocks_mem_returns_sorted_unique_memory_clocks(monkeypatch):
|
||||
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV)
|
||||
assert autotune._supported_clocks("mem") == [405, 9501, 10501]
|
||||
|
||||
|
||||
def test_supported_clocks_gr_returns_clocks_of_the_highest_memory_clock(monkeypatch):
|
||||
"""Graphics clocks are enumerated per memory clock. Only the set belonging to the top
|
||||
memory clock is meaningful — that is the state any real workload runs at."""
|
||||
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV)
|
||||
assert autotune._supported_clocks("gr") == [2760, 2775, 2790]
|
||||
|
||||
|
||||
def test_supported_clocks_always_queries_the_mem_gr_pair(monkeypatch):
|
||||
"""Regression: querying a single field returns one column, and reading index 1 from it
|
||||
silently produced an empty list instead of an error. The query must stay a pair."""
|
||||
seen = []
|
||||
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV, record=seen)
|
||||
autotune._supported_clocks("gr")
|
||||
assert any("--query-supported-clocks=mem,gr" in part for part in seen[0])
|
||||
|
||||
|
||||
def test_supported_clocks_ignores_single_column_output(monkeypatch):
|
||||
"""If the driver ever returns one column, no row parses and we return [] — never a
|
||||
list of memory clocks mislabelled as graphics clocks."""
|
||||
_fake_smi(monkeypatch, "10501\n9501\n405\n")
|
||||
assert autotune._supported_clocks("mem") == []
|
||||
assert autotune._supported_clocks("gr") == []
|
||||
|
||||
|
||||
def test_supported_clocks_skips_header_and_garbage_rows(monkeypatch):
|
||||
_fake_smi(monkeypatch, "memory [MHz], graphics [MHz]\n10501, 2790\n\nN/A, N/A\n")
|
||||
assert autotune._supported_clocks("mem") == [10501]
|
||||
|
||||
|
||||
def test_supported_clocks_returns_empty_when_nvidia_smi_fails(monkeypatch):
|
||||
"""No GPU / no driver must degrade to 'no candidates', not an exception on import of
|
||||
a sweep plan."""
|
||||
def _boom(*a, **kw):
|
||||
raise OSError("nvidia-smi not found")
|
||||
monkeypatch.setattr(autotune.subprocess, "run", _boom)
|
||||
assert autotune._supported_clocks("mem") == []
|
||||
|
||||
|
||||
# --------------------------------------------------------------- power limits
|
||||
|
||||
def test_supported_power_limits_parses_min_default_max(monkeypatch):
|
||||
"""RTX 4080 SUPER on this box: 115 W min, 370 W max, 320 W stock default."""
|
||||
_fake_smi(monkeypatch, "115.00, 370.00, 320.00\n")
|
||||
vals = autotune._supported_power_limits(steps=5)
|
||||
assert vals == sorted(set(vals))
|
||||
assert max(vals) == 370
|
||||
# Never sweeps below 60% of max — the card does no useful work down there.
|
||||
assert min(vals) >= int(370 * 0.6)
|
||||
# The stock default is always included as a reference point.
|
||||
assert 320 in vals
|
||||
|
||||
|
||||
def test_supported_power_limits_respects_step_count(monkeypatch):
|
||||
_fake_smi(monkeypatch, "115.00, 370.00, 320.00\n")
|
||||
vals = autotune._supported_power_limits(steps=3)
|
||||
assert len(vals) <= 4 # 3 evenly spaced values, plus the default if it is not one
|
||||
|
||||
|
||||
def test_supported_power_limits_returns_empty_on_query_failure(monkeypatch):
|
||||
_fake_smi(monkeypatch, "N/A, N/A, N/A\n")
|
||||
assert autotune._supported_power_limits() == []
|
||||
|
||||
|
||||
# --------------------------------------------------------------- subsampling
|
||||
|
||||
def test_subsample_returns_input_when_already_short_enough():
|
||||
assert autotune._subsample([1, 2, 3], 5) == [1, 2, 3]
|
||||
|
||||
|
||||
def test_subsample_keeps_both_endpoints():
|
||||
"""The endpoints are the whole point of a sweep: stock and maximum. Dropping either
|
||||
means never measuring the setting the sweep is supposed to recommend."""
|
||||
values = list(range(0, 195)) # the card enumerates ~194 graphics clocks
|
||||
out = autotune._subsample(values, 8)
|
||||
assert out[0] == values[0]
|
||||
assert out[-1] == values[-1]
|
||||
|
||||
|
||||
def test_subsample_never_exceeds_max_steps():
|
||||
values = list(range(0, 195))
|
||||
for max_steps in (2, 3, 5, 8, 13):
|
||||
assert len(autotune._subsample(values, max_steps)) <= max_steps
|
||||
|
||||
|
||||
def test_subsample_output_is_sorted_and_unique():
|
||||
values = list(range(0, 100))
|
||||
out = autotune._subsample(values, 7)
|
||||
assert out == sorted(set(out))
|
||||
|
||||
|
||||
def test_subsample_is_evenly_spread():
|
||||
"""Clustered samples would tell us nothing about the middle of the range."""
|
||||
out = autotune._subsample(list(range(0, 101)), 5)
|
||||
gaps = [b - a for a, b in zip(out, out[1:])]
|
||||
assert max(gaps) - min(gaps) <= 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------- comfy benchmark graph
|
||||
|
||||
def test_comfy_workflow_seed_varies_between_calls():
|
||||
"""Fixed seeds made ComfyUI serve a cached result in ~1 ms without executing the
|
||||
graph, so every benchmark after the first measured nothing at all."""
|
||||
seeds = {autotune._comfy_workflow()["5"]["inputs"]["seed"] for _ in range(25)}
|
||||
assert len(seeds) > 20
|
||||
|
||||
|
||||
def test_comfy_workflow_seed_can_be_pinned_for_reproduction():
|
||||
assert autotune._comfy_workflow(seed=42)["5"]["inputs"]["seed"] == 42
|
||||
|
||||
|
||||
def test_comfy_workflow_shape_is_otherwise_constant():
|
||||
"""Only the seed may vary; a benchmark whose step count or resolution moved between
|
||||
runs would not be comparable."""
|
||||
a = autotune._comfy_workflow()
|
||||
b = autotune._comfy_workflow()
|
||||
for wf in (a, b):
|
||||
assert wf["5"]["inputs"]["steps"] == autotune.COMFY_BENCH_STEPS
|
||||
assert wf["4"]["inputs"]["width"] == autotune.COMFY_BENCH_SIZE
|
||||
assert wf["4"]["inputs"]["height"] == autotune.COMFY_BENCH_SIZE
|
||||
a["5"]["inputs"]["seed"] = b["5"]["inputs"]["seed"] = 0
|
||||
assert a == b
|
||||
|
||||
|
||||
def test_comfy_workflow_uses_the_named_checkpoint():
|
||||
wf = autotune._comfy_workflow(ckpt="some_other.safetensors")
|
||||
assert wf["1"]["inputs"]["ckpt_name"] == "some_other.safetensors"
|
||||
|
||||
|
||||
def test_temp_ceiling_is_below_the_thermal_governors_escalation_point():
|
||||
"""A sweep step must abort on temperature before the governor starts derating under
|
||||
it, otherwise the sweep measures the governor's derate rather than the knob."""
|
||||
import thermal_governor as tg
|
||||
assert autotune.TEMP_CEILING_C <= tg.TEMP_ESCALATE_C + 1.0
|
||||
|
||||
|
||||
@pytest.mark.parametrize("knob", ["mem_offset_mhz", "core_offset_mhz", "lock_mem_mhz",
|
||||
"lock_core_max", "power_limit_w"])
|
||||
def test_every_knob_declares_a_hardware_verification_field(knob):
|
||||
"""Offsets are silently ignored by some drivers (595.84 accepts an assignment and
|
||||
reads back a different value), so each knob must name the field to read back."""
|
||||
assert autotune.KNOBS[knob]["verify"]
|
||||
Reference in New Issue
Block a user