"""thermal_governor: the derate state machine. SAFETY: _step() spawns a thread that calls overclock_manager.apply_profile. The autouse `no_gpu_mutation` fixture in conftest.py replaces that (and load_profiles is stubbed per test), so escalation here can never reach the card. """ import time import pytest import overclock_manager import thermal_governor as tg @pytest.fixture def gov(monkeypatch): """A fresh governor with a stubbed profile store — never the module singleton.""" monkeypatch.setattr(overclock_manager, "load_profiles", lambda: { "ollama": {"core_offset_mhz": 100, "mem_offset_mhz": 500, "fan_speed_pct": 70, "lock_core_min": 2500, "lock_core_max": 2800}, "stock": {"core_offset_mhz": 0, "mem_offset_mhz": 0, "fan_speed_pct": 0}, }) monkeypatch.setattr(overclock_manager, "ACTIVE_PROFILE", "ollama") return tg.ThermalGovernor() def _sample(temp=60.0, reasons=None, available=True): return {"available": available, "temperature_c": temp, "throttle_reasons": reasons or []} def _feed(gov, n, **kw): for _ in range(n): gov.observe(_sample(**kw), active_profile="ollama") def _clear_cooldown(gov): """The REAPPLY_COOLDOWN_S gate is time-based; wind the clock back rather than sleep.""" gov.last_change = time.time() - tg.REAPPLY_COOLDOWN_S - 1.0 # --------------------------------------------------------------- escalation hysteresis def test_does_not_escalate_before_hot_samples_consecutive_bad_readings(gov): """One hot spike during a diffusion step must not derate the card.""" _feed(gov, tg.HOT_SAMPLES - 1, temp=90.0) assert gov.level == 0 assert gov.hot_streak == tg.HOT_SAMPLES - 1 def test_escalates_on_exactly_hot_samples_consecutive_bad_readings(gov, no_gpu_mutation): _feed(gov, tg.HOT_SAMPLES, temp=90.0) assert gov.level == 1 assert gov.hot_streak == 0 # streaks reset after a step # Actuation runs in a daemon thread; it must reach apply_profile (the stub) with the # derate overrides rather than blocking the 1 Hz sampler. deadline = time.time() + 2.0 while not no_gpu_mutation["apply_profile"] and time.time() < deadline: time.sleep(0.01) assert no_gpu_mutation["apply_profile"], "escalation never actuated" name, overrides = no_gpu_mutation["apply_profile"][0] assert name == "ollama" assert overrides["core_offset_mhz"] == int(100 * tg.DERATE_LADDER[1]["offset_scale"]) def test_alternating_hot_and_cool_samples_never_escalate(gov): """The core hysteresis property: a card oscillating around the threshold must not thrash the profile. A single good sample resets the hot streak.""" for _ in range(50): gov.observe(_sample(temp=90.0), active_profile="ollama") gov.observe(_sample(temp=60.0), active_profile="ollama") assert gov.level == 0 def test_temperature_between_recover_and_escalate_resets_both_streaks(gov): """The band between TEMP_RECOVER_C and TEMP_ESCALATE_C is neither hot nor cool; it must not accumulate credit in either direction.""" _feed(gov, tg.HOT_SAMPLES - 1, temp=90.0) gov.observe(_sample(temp=78.0), active_profile="ollama") assert gov.hot_streak == 0 and gov.cool_streak == 0 assert gov.level == 0 def test_hard_throttle_counts_as_hot_even_when_cool(gov): """A hardware slowdown means the card is protecting itself; temperature alone is not the whole signal.""" _feed(gov, tg.HOT_SAMPLES, temp=55.0, reasons=["hw_thermal_slowdown"]) assert gov.level == 1 def test_soft_throttle_reasons_do_not_escalate(gov): """Hitting a power or utilisation cap is normal operation, not distress.""" _feed(gov, tg.HOT_SAMPLES * 3, temp=55.0, reasons=["sw_power_cap", "gpu_idle"]) assert gov.level == 0 def test_escalation_stops_at_the_bottom_of_the_ladder(gov): """Level must never index past DERATE_LADDER.""" for _ in range(len(tg.DERATE_LADDER) + 3): _clear_cooldown(gov) _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert gov.level == len(tg.DERATE_LADDER) - 1 def test_cooldown_gate_blocks_a_second_step_immediately_after_the_first(gov): """REAPPLY_COOLDOWN_S stops the governor from walking the whole ladder in one second while the card is still responding to the previous change.""" _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert gov.level == 1 _feed(gov, tg.HOT_SAMPLES * 2, temp=95.0) assert gov.level == 1 def test_cooldown_gate_releases_after_the_window(gov): _feed(gov, tg.HOT_SAMPLES, temp=95.0) _clear_cooldown(gov) _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert gov.level == 2 # --------------------------------------------------------------- recovery def test_recovery_needs_cool_samples_consecutive_good_readings(gov): _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert gov.level == 1 _clear_cooldown(gov) _feed(gov, tg.COOL_SAMPLES - 1, temp=60.0) assert gov.level == 1, "recovered too early" gov.observe(_sample(temp=60.0), active_profile="ollama") assert gov.level == 0 def test_recovery_is_slower_than_escalation(): """Deliberate asymmetry: react fast to heat, give the clocks back slowly.""" assert tg.COOL_SAMPLES > tg.HOT_SAMPLES def test_recovery_stops_at_level_zero(gov): _clear_cooldown(gov) _feed(gov, tg.COOL_SAMPLES * 2, temp=50.0) assert gov.level == 0 def test_a_hard_throttle_blocks_recovery_even_at_a_cool_temperature(gov): _feed(gov, tg.HOT_SAMPLES, temp=95.0) _clear_cooldown(gov) _feed(gov, tg.COOL_SAMPLES * 2, temp=50.0, reasons=["hw_power_brake_slowdown"]) assert gov.level >= 1 # --------------------------------------------------------------- ignored input def test_disabled_governor_ignores_samples(gov): gov.enabled = False _feed(gov, tg.HOT_SAMPLES * 3, temp=99.0) assert gov.level == 0 def test_unavailable_gpu_sample_is_ignored(gov): """A failed NVML read reports available=False with no temperature; treating that as 0 C would count as a cool sample and hand the clocks back.""" for _ in range(tg.COOL_SAMPLES * 2): gov.observe({"available": False}, active_profile="ollama") assert gov.level == 0 and gov.cool_streak == 0 # --------------------------------------------------------------- overrides def test_overrides_are_empty_at_level_zero(gov): assert gov.overrides_for("ollama") == {} @pytest.mark.parametrize("level", [1, 2, 3]) def test_overrides_scale_offsets_by_the_ladder(gov, level): gov.level = level scale = tg.DERATE_LADDER[level]["offset_scale"] ov = gov.overrides_for("ollama") assert ov["core_offset_mhz"] == int(100 * scale) assert ov["mem_offset_mhz"] == int(500 * scale) @pytest.mark.parametrize("level", [1, 2, 3]) def test_overrides_raise_the_fan_floor_and_never_lower_it(gov, level): """The floor is a floor: a profile already running fans harder than the ladder asks keeps its own setting.""" gov.level = level floor = tg.DERATE_LADDER[level]["fan_floor"] ov = gov.overrides_for("ollama") assert ov["fan_mode"] == "manual" assert ov["fan_speed_pct"] == max(70, floor) # profile fan_speed_pct is 70 assert ov["fan_speed_pct"] >= floor def test_overrides_release_the_core_clock_lock_from_level_two(gov): """Pinning the core clock high while the card is already backing off fights the hardware's own protection.""" gov.level = 1 assert "lock_core_max" not in gov.overrides_for("ollama") for level in (2, 3): gov.level = level ov = gov.overrides_for("ollama") assert ov["lock_core_min"] == 0 and ov["lock_core_max"] == 0 def test_top_of_ladder_is_stock_clocks_and_maximum_fans(gov): gov.level = len(tg.DERATE_LADDER) - 1 ov = gov.overrides_for("ollama") assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0 assert ov["fan_speed_pct"] == 100 def test_overrides_for_unknown_profile_do_not_raise(gov): """The active profile can be one the store does not know; a missing config must derate to zero offsets rather than blow up in the sampler thread.""" gov.level = 2 ov = gov.overrides_for("does-not-exist") assert ov["core_offset_mhz"] == 0 and ov["mem_offset_mhz"] == 0 # --------------------------------------------------------------- ladder invariants def test_ladder_is_monotonically_more_conservative(): """Each rung must reduce clocks and raise fans; a non-monotonic ladder would make escalation increase heat.""" scales = [s["offset_scale"] for s in tg.DERATE_LADDER] floors = [s["fan_floor"] for s in tg.DERATE_LADDER] assert scales == sorted(scales, reverse=True) assert floors == sorted(floors) assert scales[0] == 1.0 and scales[-1] == 0.0 assert [s["level"] for s in tg.DERATE_LADDER] == list(range(len(tg.DERATE_LADDER))) def test_escalate_and_recover_temperatures_have_a_gap(): """Without a gap between the two thresholds the governor would oscillate.""" assert tg.TEMP_RECOVER_C < tg.TEMP_ESCALATE_C # --------------------------------------------------------------- status / control def test_status_reports_level_and_history(gov): _feed(gov, tg.HOT_SAMPLES, temp=95.0) st = gov.get_status() assert st["level"] == 1 assert st["label"] == tg.DERATE_LADDER[1]["label"] assert st["escalate_at_c"] == tg.TEMP_ESCALATE_C assert st["history"] and st["history"][0]["to_level"] == 1 assert "95" in st["history"][0]["reason"] def test_history_is_bounded(gov): """The governor is long-lived inside the service; its history must not grow forever.""" for _ in range(80): _clear_cooldown(gov) gov.level = 0 _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert len(gov.history) <= 50 def test_disabling_a_derated_governor_resets_it(gov, no_gpu_mutation): """Turning the governor off must give the clocks back, not freeze the derate in place.""" _feed(gov, tg.HOT_SAMPLES, temp=95.0) assert gov.level == 1 gov.set_enabled(False) assert gov.level == 0 assert no_gpu_mutation["apply_profile"], "reset should have re-applied the base profile"