Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit

Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-01 13:40:30 -07:00
parent bacaf50713
commit 868d82794d
16 changed files with 1972 additions and 7 deletions

View File

@@ -0,0 +1,116 @@
"""overclock_manager profile storage: load/merge/validate only.
Nothing here applies a profile or shells out. CONFIG_PATH is redirected to tmp_path so the
repo's real overclock_profiles.json is never written, and conftest's autouse fixture blocks
_sh() and every actuation entry point.
"""
import json
import pytest
import overclock_manager as ocm
@pytest.fixture
def config(tmp_path, monkeypatch):
"""Redirect the profile store to a temp file. Returns a writer for its contents."""
path = tmp_path / "overclock_profiles.json"
monkeypatch.setattr(ocm, "CONFIG_PATH", str(path))
def write(data):
path.write_text(json.dumps(data))
return type("Cfg", (), {"path": path, "write": staticmethod(write)})
def test_defaults_are_returned_when_no_config_file_exists(config):
profiles = ocm.load_profiles()
assert set(profiles) == set(ocm.DEFAULT_PROFILES)
assert profiles["ollama"]["power_limit_w"] == ocm.DEFAULT_PROFILES["ollama"]["power_limit_w"]
def test_stored_values_override_defaults_key_by_key(config):
"""Only the keys present on disk change; the rest of the default profile survives.
A whole-profile replacement would silently drop keys added by a later version."""
config.write({"ollama": {"power_limit_w": 340}})
profiles = ocm.load_profiles()
assert profiles["ollama"]["power_limit_w"] == 340
assert profiles["ollama"]["label"] == ocm.DEFAULT_PROFILES["ollama"]["label"]
assert profiles["ollama"]["mem_offset_mhz"] == ocm.DEFAULT_PROFILES["ollama"]["mem_offset_mhz"]
def test_unknown_stored_profiles_are_preserved(config):
"""A user-created profile is not in DEFAULT_PROFILES and must not be dropped on load."""
config.write({"my-custom": {"label": "mine", "power_limit_w": 300}})
profiles = ocm.load_profiles()
assert profiles["my-custom"]["power_limit_w"] == 300
assert set(ocm.DEFAULT_PROFILES) <= set(profiles)
def test_loading_does_not_mutate_the_defaults(config):
"""load_profiles deep-copies DEFAULT_PROFILES; if it did not, one load with a stored
override would poison the defaults for the rest of the process."""
config.write({"ollama": {"power_limit_w": 111}})
ocm.load_profiles()
config.write({})
assert ocm.load_profiles()["ollama"]["power_limit_w"] == \
ocm.DEFAULT_PROFILES["ollama"]["power_limit_w"]
assert ocm.DEFAULT_PROFILES["ollama"]["power_limit_w"] != 111
def test_corrupt_config_falls_back_to_defaults(config):
"""A half-written JSON file must not take the whole service down at startup."""
config.path.write_text("{ not valid json")
profiles = ocm.load_profiles()
assert set(profiles) == set(ocm.DEFAULT_PROFILES)
def test_set_profile_rejects_an_unknown_name(config):
"""set_profile edits an existing profile; it is not a create-anything endpoint."""
res = ocm.set_profile("no-such-profile", {"power_limit_w": 400})
assert res["success"] is False
assert "unknown profile" in res["error"]
assert not config.path.exists()
def test_set_profile_persists_a_partial_update(config):
res = ocm.set_profile("comfy", {"power_limit_w": 350})
assert res["success"] is True
stored = json.loads(config.path.read_text())
assert stored["comfy"]["power_limit_w"] == 350
# The other profiles are written back untouched.
assert stored["ollama"]["power_limit_w"] == ocm.DEFAULT_PROFILES["ollama"]["power_limit_w"]
assert ocm.load_profiles()["comfy"]["power_limit_w"] == 350
def test_set_profile_can_edit_a_previously_stored_custom_profile(config):
config.write({"my-custom": {"label": "mine", "power_limit_w": 300}})
assert ocm.set_profile("my-custom", {"power_limit_w": 310})["success"] is True
assert ocm.load_profiles()["my-custom"]["power_limit_w"] == 310
def test_get_profiles_matches_load_profiles(config):
config.write({"ollama": {"core_offset_mhz": 42}})
assert ocm.get_profiles() == ocm.load_profiles()
def test_save_profiles_reports_failure_instead_of_raising(tmp_path, monkeypatch):
"""The dashboard calls this; an unwritable config must surface as success=False."""
monkeypatch.setattr(ocm, "CONFIG_PATH", str(tmp_path / "no-such-dir" / "p.json"))
assert ocm.save_profiles({"ollama": {}}) is False
@pytest.mark.parametrize("name", ["ollama", "comfy", "balanced"])
def test_default_profiles_declare_the_keys_the_appliers_read(name):
"""apply_profile and the thermal governor both index these keys directly; a profile
missing one would KeyError inside the actuation thread."""
cfg = ocm.DEFAULT_PROFILES[name]
for key in ("label", "power_limit_w", "core_offset_mhz", "mem_offset_mhz",
"lock_core_min", "lock_core_max"):
assert key in cfg, f"{name} is missing {key}"
def test_default_power_limits_stay_within_the_cards_range():
"""RTX 4080 SUPER: 320 W stock, 370 W maximum. A profile above that is silently
clamped by the driver and the sweep results become meaningless."""
for name, cfg in ocm.DEFAULT_PROFILES.items():
assert 100 <= cfg["power_limit_w"] <= 370, name