Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
83 lines
3.1 KiB
Python
83 lines
3.1 KiB
Python
"""Pure helpers in vram_arbitrator: NVML throttle-bit decoding and PID attribution.
|
|
|
|
Deliberately excludes instant_free_ollama_vram, the AutoArbitrator yield/purge paths and
|
|
the SSE broker — that contract is in flux.
|
|
"""
|
|
import vram_arbitrator as va
|
|
|
|
|
|
def test_decode_throttle_reasons_empty_when_no_bits_set():
|
|
assert va.decode_throttle_reasons(0) == []
|
|
|
|
|
|
def test_decode_throttle_reasons_maps_each_known_bit():
|
|
"""Every mask in the table must decode to exactly its own name in isolation."""
|
|
for mask, name in va.THROTTLE_REASONS.items():
|
|
assert va.decode_throttle_reasons(mask) == [name]
|
|
|
|
|
|
def test_decode_throttle_reasons_decodes_combined_bits():
|
|
"""Real NVML samples set several bits at once; all of them must come back."""
|
|
bits = 0x20 | 0x40 # sw_thermal_slowdown | hw_thermal_slowdown
|
|
assert set(va.decode_throttle_reasons(bits)) == {"sw_thermal_slowdown", "hw_thermal_slowdown"}
|
|
|
|
|
|
def test_decode_throttle_reasons_ignores_unknown_bits():
|
|
"""An undocumented bit from a future driver must not crash or invent a reason."""
|
|
assert va.decode_throttle_reasons(0x8000_0000) == []
|
|
|
|
|
|
def test_hard_throttle_names_match_thermal_governor_expectations():
|
|
"""thermal_governor escalates on a fixed set of reason strings produced here.
|
|
If a name is renamed in one module and not the other the governor silently stops
|
|
reacting to hardware slowdowns, so pin the shared vocabulary."""
|
|
import thermal_governor as tg
|
|
assert tg.HARD_THROTTLES <= set(va.THROTTLE_REASONS.values())
|
|
|
|
|
|
class _FakeProc:
|
|
def __init__(self, name, cmdline):
|
|
self._name = name
|
|
self._cmdline = cmdline
|
|
|
|
def name(self):
|
|
return self._name
|
|
|
|
def cmdline(self):
|
|
return self._cmdline
|
|
|
|
|
|
def _patch_proc(monkeypatch, proc):
|
|
monkeypatch.setattr(va.psutil, "Process", lambda pid: proc)
|
|
|
|
|
|
def test_classify_pid_detects_ollama_by_process_name(monkeypatch):
|
|
_patch_proc(monkeypatch, _FakeProc("ollama", ["/usr/local/bin/ollama", "serve"]))
|
|
assert va._classify_pid(1234) == "ollama"
|
|
|
|
|
|
def test_classify_pid_detects_ollama_runner_by_cmdline(monkeypatch):
|
|
"""Ollama's model runner is a separate llama-server process; its VRAM is Ollama's."""
|
|
_patch_proc(monkeypatch, _FakeProc("llama-server",
|
|
["/usr/lib/ollama/llama-server", "--model", "blob"]))
|
|
assert va._classify_pid(1234) == "ollama"
|
|
|
|
|
|
def test_classify_pid_detects_comfyui(monkeypatch):
|
|
_patch_proc(monkeypatch, _FakeProc("python3", ["python3", "/opt/ComfyUI/main.py", "--listen"]))
|
|
assert va._classify_pid(1234) == "comfy"
|
|
|
|
|
|
def test_classify_pid_unknown_process_is_other(monkeypatch):
|
|
_patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"]))
|
|
assert va._classify_pid(1234) == "other"
|
|
|
|
|
|
def test_classify_pid_returns_other_when_process_vanished(monkeypatch):
|
|
"""PIDs are read from NVML and can exit before psutil looks them up; that is normal
|
|
and must not raise inside the 20 ms VRAM poll loop."""
|
|
def _boom(pid):
|
|
raise va.psutil.NoSuchProcess(pid)
|
|
monkeypatch.setattr(va.psutil, "Process", _boom)
|
|
assert va._classify_pid(999999) == "other"
|