Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
264
tests/test_thermal_governor.py
Normal file
264
tests/test_thermal_governor.py
Normal file
@@ -0,0 +1,264 @@
|
||||
"""thermal_governor: the derate state machine.
|
||||
|
||||
SAFETY: _step() spawns a thread that calls overclock_manager.apply_profile. The autouse
|
||||
`no_gpu_mutation` fixture in conftest.py replaces that (and load_profiles is stubbed per
|
||||
test), so escalation here can never reach the card.
|
||||
"""
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
import overclock_manager
|
||||
import thermal_governor as tg
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def gov(monkeypatch):
|
||||
"""A fresh governor with a stubbed profile store — never the module singleton."""
|
||||
monkeypatch.setattr(overclock_manager, "load_profiles", lambda: {
|
||||
"ollama": {"core_offset_mhz": 100, "mem_offset_mhz": 500, "fan_speed_pct": 70,
|
||||
"lock_core_min": 2500, "lock_core_max": 2800},
|
||||
"stock": {"core_offset_mhz": 0, "mem_offset_mhz": 0, "fan_speed_pct": 0},
|
||||
})
|
||||
monkeypatch.setattr(overclock_manager, "ACTIVE_PROFILE", "ollama")
|
||||
return tg.ThermalGovernor()
|
||||
|
||||
|
||||
def _sample(temp=60.0, reasons=None, available=True):
|
||||
return {"available": available, "temperature_c": temp, "throttle_reasons": reasons or []}
|
||||
|
||||
|
||||
def _feed(gov, n, **kw):
|
||||
for _ in range(n):
|
||||
gov.observe(_sample(**kw), active_profile="ollama")
|
||||
|
||||
|
||||
def _clear_cooldown(gov):
|
||||
"""The REAPPLY_COOLDOWN_S gate is time-based; wind the clock back rather than sleep."""
|
||||
gov.last_change = time.time() - tg.REAPPLY_COOLDOWN_S - 1.0
|
||||
|
||||
|
||||
# --------------------------------------------------------------- escalation hysteresis
|
||||
|
||||
def test_does_not_escalate_before_hot_samples_consecutive_bad_readings(gov):
|
||||
"""One hot spike during a diffusion step must not derate the card."""
|
||||
_feed(gov, tg.HOT_SAMPLES - 1, temp=90.0)
|
||||
assert gov.level == 0
|
||||
assert gov.hot_streak == tg.HOT_SAMPLES - 1
|
||||
|
||||
|
||||
def test_escalates_on_exactly_hot_samples_consecutive_bad_readings(gov, no_gpu_mutation):
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=90.0)
|
||||
assert gov.level == 1
|
||||
assert gov.hot_streak == 0 # streaks reset after a step
|
||||
# Actuation runs in a daemon thread; it must reach apply_profile (the stub) with the
|
||||
# derate overrides rather than blocking the 1 Hz sampler.
|
||||
deadline = time.time() + 2.0
|
||||
while not no_gpu_mutation["apply_profile"] and time.time() < deadline:
|
||||
time.sleep(0.01)
|
||||
assert no_gpu_mutation["apply_profile"], "escalation never actuated"
|
||||
name, overrides = no_gpu_mutation["apply_profile"][0]
|
||||
assert name == "ollama"
|
||||
assert overrides["core_offset_mhz"] == int(100 * tg.DERATE_LADDER[1]["offset_scale"])
|
||||
|
||||
|
||||
def test_alternating_hot_and_cool_samples_never_escalate(gov):
|
||||
"""The core hysteresis property: a card oscillating around the threshold must not
|
||||
thrash the profile. A single good sample resets the hot streak."""
|
||||
for _ in range(50):
|
||||
gov.observe(_sample(temp=90.0), active_profile="ollama")
|
||||
gov.observe(_sample(temp=60.0), active_profile="ollama")
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_temperature_between_recover_and_escalate_resets_both_streaks(gov):
|
||||
"""The band between TEMP_RECOVER_C and TEMP_ESCALATE_C is neither hot nor cool; it
|
||||
must not accumulate credit in either direction."""
|
||||
_feed(gov, tg.HOT_SAMPLES - 1, temp=90.0)
|
||||
gov.observe(_sample(temp=78.0), active_profile="ollama")
|
||||
assert gov.hot_streak == 0 and gov.cool_streak == 0
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_hard_throttle_counts_as_hot_even_when_cool(gov):
|
||||
"""A hardware slowdown means the card is protecting itself; temperature alone is not
|
||||
the whole signal."""
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=55.0, reasons=["hw_thermal_slowdown"])
|
||||
assert gov.level == 1
|
||||
|
||||
|
||||
def test_soft_throttle_reasons_do_not_escalate(gov):
|
||||
"""Hitting a power or utilisation cap is normal operation, not distress."""
|
||||
_feed(gov, tg.HOT_SAMPLES * 3, temp=55.0, reasons=["sw_power_cap", "gpu_idle"])
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_escalation_stops_at_the_bottom_of_the_ladder(gov):
|
||||
"""Level must never index past DERATE_LADDER."""
|
||||
for _ in range(len(tg.DERATE_LADDER) + 3):
|
||||
_clear_cooldown(gov)
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert gov.level == len(tg.DERATE_LADDER) - 1
|
||||
|
||||
|
||||
def test_cooldown_gate_blocks_a_second_step_immediately_after_the_first(gov):
|
||||
"""REAPPLY_COOLDOWN_S stops the governor from walking the whole ladder in one second
|
||||
while the card is still responding to the previous change."""
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert gov.level == 1
|
||||
_feed(gov, tg.HOT_SAMPLES * 2, temp=95.0)
|
||||
assert gov.level == 1
|
||||
|
||||
|
||||
def test_cooldown_gate_releases_after_the_window(gov):
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
_clear_cooldown(gov)
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert gov.level == 2
|
||||
|
||||
|
||||
# --------------------------------------------------------------- recovery
|
||||
|
||||
def test_recovery_needs_cool_samples_consecutive_good_readings(gov):
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert gov.level == 1
|
||||
_clear_cooldown(gov)
|
||||
_feed(gov, tg.COOL_SAMPLES - 1, temp=60.0)
|
||||
assert gov.level == 1, "recovered too early"
|
||||
gov.observe(_sample(temp=60.0), active_profile="ollama")
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_recovery_is_slower_than_escalation():
|
||||
"""Deliberate asymmetry: react fast to heat, give the clocks back slowly."""
|
||||
assert tg.COOL_SAMPLES > tg.HOT_SAMPLES
|
||||
|
||||
|
||||
def test_recovery_stops_at_level_zero(gov):
|
||||
_clear_cooldown(gov)
|
||||
_feed(gov, tg.COOL_SAMPLES * 2, temp=50.0)
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_a_hard_throttle_blocks_recovery_even_at_a_cool_temperature(gov):
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
_clear_cooldown(gov)
|
||||
_feed(gov, tg.COOL_SAMPLES * 2, temp=50.0, reasons=["hw_power_brake_slowdown"])
|
||||
assert gov.level >= 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------- ignored input
|
||||
|
||||
def test_disabled_governor_ignores_samples(gov):
|
||||
gov.enabled = False
|
||||
_feed(gov, tg.HOT_SAMPLES * 3, temp=99.0)
|
||||
assert gov.level == 0
|
||||
|
||||
|
||||
def test_unavailable_gpu_sample_is_ignored(gov):
|
||||
"""A failed NVML read reports available=False with no temperature; treating that as
|
||||
0 C would count as a cool sample and hand the clocks back."""
|
||||
for _ in range(tg.COOL_SAMPLES * 2):
|
||||
gov.observe({"available": False}, active_profile="ollama")
|
||||
assert gov.level == 0 and gov.cool_streak == 0
|
||||
|
||||
|
||||
# --------------------------------------------------------------- overrides
|
||||
|
||||
def test_overrides_are_empty_at_level_zero(gov):
|
||||
assert gov.overrides_for("ollama") == {}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("level", [1, 2, 3])
|
||||
def test_overrides_scale_offsets_by_the_ladder(gov, level):
|
||||
gov.level = level
|
||||
scale = tg.DERATE_LADDER[level]["offset_scale"]
|
||||
ov = gov.overrides_for("ollama")
|
||||
assert ov["core_offset_mhz"] == int(100 * scale)
|
||||
assert ov["mem_offset_mhz"] == int(500 * scale)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("level", [1, 2, 3])
|
||||
def test_overrides_raise_the_fan_floor_and_never_lower_it(gov, level):
|
||||
"""The floor is a floor: a profile already running fans harder than the ladder asks
|
||||
keeps its own setting."""
|
||||
gov.level = level
|
||||
floor = tg.DERATE_LADDER[level]["fan_floor"]
|
||||
ov = gov.overrides_for("ollama")
|
||||
assert ov["fan_mode"] == "manual"
|
||||
assert ov["fan_speed_pct"] == max(70, floor) # profile fan_speed_pct is 70
|
||||
assert ov["fan_speed_pct"] >= floor
|
||||
|
||||
|
||||
def test_overrides_release_the_core_clock_lock_from_level_two(gov):
|
||||
"""Pinning the core clock high while the card is already backing off fights the
|
||||
hardware's own protection."""
|
||||
gov.level = 1
|
||||
assert "lock_core_max" not in gov.overrides_for("ollama")
|
||||
for level in (2, 3):
|
||||
gov.level = level
|
||||
ov = gov.overrides_for("ollama")
|
||||
assert ov["lock_core_min"] == 0 and ov["lock_core_max"] == 0
|
||||
|
||||
|
||||
def test_top_of_ladder_is_stock_clocks_and_maximum_fans(gov):
|
||||
gov.level = len(tg.DERATE_LADDER) - 1
|
||||
ov = gov.overrides_for("ollama")
|
||||
assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0
|
||||
assert ov["fan_speed_pct"] == 100
|
||||
|
||||
|
||||
def test_overrides_for_unknown_profile_do_not_raise(gov):
|
||||
"""The active profile can be one the store does not know; a missing config must
|
||||
derate to zero offsets rather than blow up in the sampler thread."""
|
||||
gov.level = 2
|
||||
ov = gov.overrides_for("does-not-exist")
|
||||
assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0
|
||||
|
||||
|
||||
# --------------------------------------------------------------- ladder invariants
|
||||
|
||||
def test_ladder_is_monotonically_more_conservative():
|
||||
"""Each rung must reduce clocks and raise fans; a non-monotonic ladder would make
|
||||
escalation increase heat."""
|
||||
scales = [s["offset_scale"] for s in tg.DERATE_LADDER]
|
||||
floors = [s["fan_floor"] for s in tg.DERATE_LADDER]
|
||||
assert scales == sorted(scales, reverse=True)
|
||||
assert floors == sorted(floors)
|
||||
assert scales[0] == 1.0 and scales[-1] == 0.0
|
||||
assert [s["level"] for s in tg.DERATE_LADDER] == list(range(len(tg.DERATE_LADDER)))
|
||||
|
||||
|
||||
def test_escalate_and_recover_temperatures_have_a_gap():
|
||||
"""Without a gap between the two thresholds the governor would oscillate."""
|
||||
assert tg.TEMP_RECOVER_C < tg.TEMP_ESCALATE_C
|
||||
|
||||
|
||||
# --------------------------------------------------------------- status / control
|
||||
|
||||
def test_status_reports_level_and_history(gov):
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
st = gov.get_status()
|
||||
assert st["level"] == 1
|
||||
assert st["label"] == tg.DERATE_LADDER[1]["label"]
|
||||
assert st["escalate_at_c"] == tg.TEMP_ESCALATE_C
|
||||
assert st["history"] and st["history"][0]["to_level"] == 1
|
||||
assert "95" in st["history"][0]["reason"]
|
||||
|
||||
|
||||
def test_history_is_bounded(gov):
|
||||
"""The governor is long-lived inside the service; its history must not grow forever."""
|
||||
for _ in range(80):
|
||||
_clear_cooldown(gov)
|
||||
gov.level = 0
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert len(gov.history) <= 50
|
||||
|
||||
|
||||
def test_disabling_a_derated_governor_resets_it(gov, no_gpu_mutation):
|
||||
"""Turning the governor off must give the clocks back, not freeze the derate in place."""
|
||||
_feed(gov, tg.HOT_SAMPLES, temp=95.0)
|
||||
assert gov.level == 1
|
||||
gov.set_enabled(False)
|
||||
assert gov.level == 0
|
||||
assert no_gpu_mutation["apply_profile"], "reset should have re-applied the base profile"
|
||||
Reference in New Issue
Block a user