Files
gpu-program-swapper/tests/test_autotune_helpers.py
drjones 868d82794d Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-01 13:40:30 -07:00

185 lines
7.1 KiB
Python

"""autotune's pure/parsable helpers.
Only the parsing and sampling helpers are exercised. Nothing here runs a sweep, applies a
profile, or talks to nvidia-smi: subprocess.run is monkeypatched at the module boundary.
"""
import pytest
import autotune
class _FakeProc:
def __init__(self, stdout="", returncode=0, stderr=""):
self.stdout = stdout
self.stderr = stderr
self.returncode = returncode
def _fake_smi(monkeypatch, stdout, record=None):
def _run(cmd, capture_output=True, text=True, timeout=None, **kw):
if record is not None:
record.append(list(cmd))
return _FakeProc(stdout)
monkeypatch.setattr(autotune.subprocess, "run", _run)
# A trimmed but realistically shaped `nvidia-smi --query-supported-clocks=mem,gr` dump:
# graphics clocks are enumerated once per memory clock.
SUPPORTED_CLOCKS_CSV = """\
10501, 2790
10501, 2775
10501, 2760
9501, 2790
9501, 2775
405, 645
405, 630
"""
def test_supported_clocks_mem_returns_sorted_unique_memory_clocks(monkeypatch):
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV)
assert autotune._supported_clocks("mem") == [405, 9501, 10501]
def test_supported_clocks_gr_returns_clocks_of_the_highest_memory_clock(monkeypatch):
"""Graphics clocks are enumerated per memory clock. Only the set belonging to the top
memory clock is meaningful — that is the state any real workload runs at."""
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV)
assert autotune._supported_clocks("gr") == [2760, 2775, 2790]
def test_supported_clocks_always_queries_the_mem_gr_pair(monkeypatch):
"""Regression: querying a single field returns one column, and reading index 1 from it
silently produced an empty list instead of an error. The query must stay a pair."""
seen = []
_fake_smi(monkeypatch, SUPPORTED_CLOCKS_CSV, record=seen)
autotune._supported_clocks("gr")
assert any("--query-supported-clocks=mem,gr" in part for part in seen[0])
def test_supported_clocks_ignores_single_column_output(monkeypatch):
"""If the driver ever returns one column, no row parses and we return [] — never a
list of memory clocks mislabelled as graphics clocks."""
_fake_smi(monkeypatch, "10501\n9501\n405\n")
assert autotune._supported_clocks("mem") == []
assert autotune._supported_clocks("gr") == []
def test_supported_clocks_skips_header_and_garbage_rows(monkeypatch):
_fake_smi(monkeypatch, "memory [MHz], graphics [MHz]\n10501, 2790\n\nN/A, N/A\n")
assert autotune._supported_clocks("mem") == [10501]
def test_supported_clocks_returns_empty_when_nvidia_smi_fails(monkeypatch):
"""No GPU / no driver must degrade to 'no candidates', not an exception on import of
a sweep plan."""
def _boom(*a, **kw):
raise OSError("nvidia-smi not found")
monkeypatch.setattr(autotune.subprocess, "run", _boom)
assert autotune._supported_clocks("mem") == []
# --------------------------------------------------------------- power limits
def test_supported_power_limits_parses_min_default_max(monkeypatch):
"""RTX 4080 SUPER on this box: 115 W min, 370 W max, 320 W stock default."""
_fake_smi(monkeypatch, "115.00, 370.00, 320.00\n")
vals = autotune._supported_power_limits(steps=5)
assert vals == sorted(set(vals))
assert max(vals) == 370
# Never sweeps below 60% of max — the card does no useful work down there.
assert min(vals) >= int(370 * 0.6)
# The stock default is always included as a reference point.
assert 320 in vals
def test_supported_power_limits_respects_step_count(monkeypatch):
_fake_smi(monkeypatch, "115.00, 370.00, 320.00\n")
vals = autotune._supported_power_limits(steps=3)
assert len(vals) <= 4 # 3 evenly spaced values, plus the default if it is not one
def test_supported_power_limits_returns_empty_on_query_failure(monkeypatch):
_fake_smi(monkeypatch, "N/A, N/A, N/A\n")
assert autotune._supported_power_limits() == []
# --------------------------------------------------------------- subsampling
def test_subsample_returns_input_when_already_short_enough():
assert autotune._subsample([1, 2, 3], 5) == [1, 2, 3]
def test_subsample_keeps_both_endpoints():
"""The endpoints are the whole point of a sweep: stock and maximum. Dropping either
means never measuring the setting the sweep is supposed to recommend."""
values = list(range(0, 195)) # the card enumerates ~194 graphics clocks
out = autotune._subsample(values, 8)
assert out[0] == values[0]
assert out[-1] == values[-1]
def test_subsample_never_exceeds_max_steps():
values = list(range(0, 195))
for max_steps in (2, 3, 5, 8, 13):
assert len(autotune._subsample(values, max_steps)) <= max_steps
def test_subsample_output_is_sorted_and_unique():
values = list(range(0, 100))
out = autotune._subsample(values, 7)
assert out == sorted(set(out))
def test_subsample_is_evenly_spread():
"""Clustered samples would tell us nothing about the middle of the range."""
out = autotune._subsample(list(range(0, 101)), 5)
gaps = [b - a for a, b in zip(out, out[1:])]
assert max(gaps) - min(gaps) <= 1
# --------------------------------------------------------------- comfy benchmark graph
def test_comfy_workflow_seed_varies_between_calls():
"""Fixed seeds made ComfyUI serve a cached result in ~1 ms without executing the
graph, so every benchmark after the first measured nothing at all."""
seeds = {autotune._comfy_workflow()["5"]["inputs"]["seed"] for _ in range(25)}
assert len(seeds) > 20
def test_comfy_workflow_seed_can_be_pinned_for_reproduction():
assert autotune._comfy_workflow(seed=42)["5"]["inputs"]["seed"] == 42
def test_comfy_workflow_shape_is_otherwise_constant():
"""Only the seed may vary; a benchmark whose step count or resolution moved between
runs would not be comparable."""
a = autotune._comfy_workflow()
b = autotune._comfy_workflow()
for wf in (a, b):
assert wf["5"]["inputs"]["steps"] == autotune.COMFY_BENCH_STEPS
assert wf["4"]["inputs"]["width"] == autotune.COMFY_BENCH_SIZE
assert wf["4"]["inputs"]["height"] == autotune.COMFY_BENCH_SIZE
a["5"]["inputs"]["seed"] = b["5"]["inputs"]["seed"] = 0
assert a == b
def test_comfy_workflow_uses_the_named_checkpoint():
wf = autotune._comfy_workflow(ckpt="some_other.safetensors")
assert wf["1"]["inputs"]["ckpt_name"] == "some_other.safetensors"
def test_temp_ceiling_is_below_the_thermal_governors_escalation_point():
"""A sweep step must abort on temperature before the governor starts derating under
it, otherwise the sweep measures the governor's derate rather than the knob."""
import thermal_governor as tg
assert autotune.TEMP_CEILING_C <= tg.TEMP_ESCALATE_C + 1.0
@pytest.mark.parametrize("knob", ["mem_offset_mhz", "core_offset_mhz", "lock_mem_mhz",
"lock_core_max", "power_limit_w"])
def test_every_knob_declares_a_hardware_verification_field(knob):
"""Offsets are silently ignored by some drivers (595.84 accepts an assignment and
reads back a different value), so each knob must name the field to read back."""
assert autotune.KNOBS[knob]["verify"]