Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit

Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-01 13:40:30 -07:00
parent bacaf50713
commit 868d82794d
16 changed files with 1972 additions and 7 deletions

View File

@@ -0,0 +1,264 @@
"""thermal_governor: the derate state machine.
SAFETY: _step() spawns a thread that calls overclock_manager.apply_profile. The autouse
`no_gpu_mutation` fixture in conftest.py replaces that (and load_profiles is stubbed per
test), so escalation here can never reach the card.
"""
import time
import pytest
import overclock_manager
import thermal_governor as tg
@pytest.fixture
def gov(monkeypatch):
"""A fresh governor with a stubbed profile store — never the module singleton."""
monkeypatch.setattr(overclock_manager, "load_profiles", lambda: {
"ollama": {"core_offset_mhz": 100, "mem_offset_mhz": 500, "fan_speed_pct": 70,
"lock_core_min": 2500, "lock_core_max": 2800},
"stock": {"core_offset_mhz": 0, "mem_offset_mhz": 0, "fan_speed_pct": 0},
})
monkeypatch.setattr(overclock_manager, "ACTIVE_PROFILE", "ollama")
return tg.ThermalGovernor()
def _sample(temp=60.0, reasons=None, available=True):
return {"available": available, "temperature_c": temp, "throttle_reasons": reasons or []}
def _feed(gov, n, **kw):
for _ in range(n):
gov.observe(_sample(**kw), active_profile="ollama")
def _clear_cooldown(gov):
"""The REAPPLY_COOLDOWN_S gate is time-based; wind the clock back rather than sleep."""
gov.last_change = time.time() - tg.REAPPLY_COOLDOWN_S - 1.0
# --------------------------------------------------------------- escalation hysteresis
def test_does_not_escalate_before_hot_samples_consecutive_bad_readings(gov):
"""One hot spike during a diffusion step must not derate the card."""
_feed(gov, tg.HOT_SAMPLES - 1, temp=90.0)
assert gov.level == 0
assert gov.hot_streak == tg.HOT_SAMPLES - 1
def test_escalates_on_exactly_hot_samples_consecutive_bad_readings(gov, no_gpu_mutation):
_feed(gov, tg.HOT_SAMPLES, temp=90.0)
assert gov.level == 1
assert gov.hot_streak == 0 # streaks reset after a step
# Actuation runs in a daemon thread; it must reach apply_profile (the stub) with the
# derate overrides rather than blocking the 1 Hz sampler.
deadline = time.time() + 2.0
while not no_gpu_mutation["apply_profile"] and time.time() < deadline:
time.sleep(0.01)
assert no_gpu_mutation["apply_profile"], "escalation never actuated"
name, overrides = no_gpu_mutation["apply_profile"][0]
assert name == "ollama"
assert overrides["core_offset_mhz"] == int(100 * tg.DERATE_LADDER[1]["offset_scale"])
def test_alternating_hot_and_cool_samples_never_escalate(gov):
"""The core hysteresis property: a card oscillating around the threshold must not
thrash the profile. A single good sample resets the hot streak."""
for _ in range(50):
gov.observe(_sample(temp=90.0), active_profile="ollama")
gov.observe(_sample(temp=60.0), active_profile="ollama")
assert gov.level == 0
def test_temperature_between_recover_and_escalate_resets_both_streaks(gov):
"""The band between TEMP_RECOVER_C and TEMP_ESCALATE_C is neither hot nor cool; it
must not accumulate credit in either direction."""
_feed(gov, tg.HOT_SAMPLES - 1, temp=90.0)
gov.observe(_sample(temp=78.0), active_profile="ollama")
assert gov.hot_streak == 0 and gov.cool_streak == 0
assert gov.level == 0
def test_hard_throttle_counts_as_hot_even_when_cool(gov):
"""A hardware slowdown means the card is protecting itself; temperature alone is not
the whole signal."""
_feed(gov, tg.HOT_SAMPLES, temp=55.0, reasons=["hw_thermal_slowdown"])
assert gov.level == 1
def test_soft_throttle_reasons_do_not_escalate(gov):
"""Hitting a power or utilisation cap is normal operation, not distress."""
_feed(gov, tg.HOT_SAMPLES * 3, temp=55.0, reasons=["sw_power_cap", "gpu_idle"])
assert gov.level == 0
def test_escalation_stops_at_the_bottom_of_the_ladder(gov):
"""Level must never index past DERATE_LADDER."""
for _ in range(len(tg.DERATE_LADDER) + 3):
_clear_cooldown(gov)
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert gov.level == len(tg.DERATE_LADDER) - 1
def test_cooldown_gate_blocks_a_second_step_immediately_after_the_first(gov):
"""REAPPLY_COOLDOWN_S stops the governor from walking the whole ladder in one second
while the card is still responding to the previous change."""
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert gov.level == 1
_feed(gov, tg.HOT_SAMPLES * 2, temp=95.0)
assert gov.level == 1
def test_cooldown_gate_releases_after_the_window(gov):
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
_clear_cooldown(gov)
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert gov.level == 2
# --------------------------------------------------------------- recovery
def test_recovery_needs_cool_samples_consecutive_good_readings(gov):
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert gov.level == 1
_clear_cooldown(gov)
_feed(gov, tg.COOL_SAMPLES - 1, temp=60.0)
assert gov.level == 1, "recovered too early"
gov.observe(_sample(temp=60.0), active_profile="ollama")
assert gov.level == 0
def test_recovery_is_slower_than_escalation():
"""Deliberate asymmetry: react fast to heat, give the clocks back slowly."""
assert tg.COOL_SAMPLES > tg.HOT_SAMPLES
def test_recovery_stops_at_level_zero(gov):
_clear_cooldown(gov)
_feed(gov, tg.COOL_SAMPLES * 2, temp=50.0)
assert gov.level == 0
def test_a_hard_throttle_blocks_recovery_even_at_a_cool_temperature(gov):
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
_clear_cooldown(gov)
_feed(gov, tg.COOL_SAMPLES * 2, temp=50.0, reasons=["hw_power_brake_slowdown"])
assert gov.level >= 1
# --------------------------------------------------------------- ignored input
def test_disabled_governor_ignores_samples(gov):
gov.enabled = False
_feed(gov, tg.HOT_SAMPLES * 3, temp=99.0)
assert gov.level == 0
def test_unavailable_gpu_sample_is_ignored(gov):
"""A failed NVML read reports available=False with no temperature; treating that as
0 C would count as a cool sample and hand the clocks back."""
for _ in range(tg.COOL_SAMPLES * 2):
gov.observe({"available": False}, active_profile="ollama")
assert gov.level == 0 and gov.cool_streak == 0
# --------------------------------------------------------------- overrides
def test_overrides_are_empty_at_level_zero(gov):
assert gov.overrides_for("ollama") == {}
@pytest.mark.parametrize("level", [1, 2, 3])
def test_overrides_scale_offsets_by_the_ladder(gov, level):
gov.level = level
scale = tg.DERATE_LADDER[level]["offset_scale"]
ov = gov.overrides_for("ollama")
assert ov["core_offset_mhz"] == int(100 * scale)
assert ov["mem_offset_mhz"] == int(500 * scale)
@pytest.mark.parametrize("level", [1, 2, 3])
def test_overrides_raise_the_fan_floor_and_never_lower_it(gov, level):
"""The floor is a floor: a profile already running fans harder than the ladder asks
keeps its own setting."""
gov.level = level
floor = tg.DERATE_LADDER[level]["fan_floor"]
ov = gov.overrides_for("ollama")
assert ov["fan_mode"] == "manual"
assert ov["fan_speed_pct"] == max(70, floor) # profile fan_speed_pct is 70
assert ov["fan_speed_pct"] >= floor
def test_overrides_release_the_core_clock_lock_from_level_two(gov):
"""Pinning the core clock high while the card is already backing off fights the
hardware's own protection."""
gov.level = 1
assert "lock_core_max" not in gov.overrides_for("ollama")
for level in (2, 3):
gov.level = level
ov = gov.overrides_for("ollama")
assert ov["lock_core_min"] == 0 and ov["lock_core_max"] == 0
def test_top_of_ladder_is_stock_clocks_and_maximum_fans(gov):
gov.level = len(tg.DERATE_LADDER) - 1
ov = gov.overrides_for("ollama")
assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0
assert ov["fan_speed_pct"] == 100
def test_overrides_for_unknown_profile_do_not_raise(gov):
"""The active profile can be one the store does not know; a missing config must
derate to zero offsets rather than blow up in the sampler thread."""
gov.level = 2
ov = gov.overrides_for("does-not-exist")
assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0
# --------------------------------------------------------------- ladder invariants
def test_ladder_is_monotonically_more_conservative():
"""Each rung must reduce clocks and raise fans; a non-monotonic ladder would make
escalation increase heat."""
scales = [s["offset_scale"] for s in tg.DERATE_LADDER]
floors = [s["fan_floor"] for s in tg.DERATE_LADDER]
assert scales == sorted(scales, reverse=True)
assert floors == sorted(floors)
assert scales[0] == 1.0 and scales[-1] == 0.0
assert [s["level"] for s in tg.DERATE_LADDER] == list(range(len(tg.DERATE_LADDER)))
def test_escalate_and_recover_temperatures_have_a_gap():
"""Without a gap between the two thresholds the governor would oscillate."""
assert tg.TEMP_RECOVER_C < tg.TEMP_ESCALATE_C
# --------------------------------------------------------------- status / control
def test_status_reports_level_and_history(gov):
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
st = gov.get_status()
assert st["level"] == 1
assert st["label"] == tg.DERATE_LADDER[1]["label"]
assert st["escalate_at_c"] == tg.TEMP_ESCALATE_C
assert st["history"] and st["history"][0]["to_level"] == 1
assert "95" in st["history"][0]["reason"]
def test_history_is_bounded(gov):
"""The governor is long-lived inside the service; its history must not grow forever."""
for _ in range(80):
_clear_cooldown(gov)
gov.level = 0
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert len(gov.history) <= 50
def test_disabling_a_derated_governor_resets_it(gov, no_gpu_mutation):
"""Turning the governor off must give the clocks back, not freeze the derate in place."""
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
assert gov.level == 1
gov.set_enabled(False)
assert gov.level == 0
assert no_gpu_mutation["apply_profile"], "reset should have re-applied the base profile"