Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
342
tests/test_telemetry_store.py
Normal file
342
tests/test_telemetry_store.py
Normal file
@@ -0,0 +1,342 @@
|
||||
"""telemetry_store: schema, the background writer round-trip, and the analytics queries.
|
||||
|
||||
Every test in this module runs against a throwaway SQLite file via the `temp_db` fixture
|
||||
(which monkeypatches telemetry_store.DB_PATH). The production hyperswap.db is never
|
||||
opened: the writer thread, _read_conn() and _rows() all resolve DB_PATH at call time.
|
||||
"""
|
||||
import os
|
||||
import sqlite3
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
import telemetry_store as ts
|
||||
|
||||
|
||||
def _drain(timeout=3.0):
|
||||
"""Wait until the writer thread has dequeued everything submitted so far.
|
||||
|
||||
NOTE: stop() flushes the writer's *pending* batch but does not drain the submission
|
||||
queue, so a stop() racing a just-submitted row can drop it. Tests wait explicitly
|
||||
rather than depending on that race. (See tests/README.md.)
|
||||
"""
|
||||
w = ts._writer
|
||||
assert w is not None, "writer not started"
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
if w.q.empty():
|
||||
time.sleep(0.05) # let the in-flight item finish its execute/commit
|
||||
return
|
||||
time.sleep(0.01)
|
||||
raise AssertionError("telemetry writer never drained its queue")
|
||||
|
||||
|
||||
def _seed(db_path, events=(), telemetry=()):
|
||||
"""Insert rows directly, bypassing the writer thread, for query tests."""
|
||||
conn = sqlite3.connect(db_path)
|
||||
conn.executescript(ts.SCHEMA)
|
||||
for e in events:
|
||||
conn.execute(ts._EVENT_SQL, (
|
||||
e.get("ts", time.time()), e.get("event_type"), e.get("source"), e.get("target"),
|
||||
e.get("profile"), e.get("duration_ms"), e.get("load_duration_ms"),
|
||||
e.get("yield_confirm_ms"), e.get("tokens_per_sec"), e.get("bytes_loaded"),
|
||||
e.get("load_gbps"), e.get("cache_status"), e.get("detail"),
|
||||
))
|
||||
for t in telemetry:
|
||||
conn.execute(ts._TELEMETRY_SQL, (
|
||||
t.get("ts", time.time()), t.get("profile"), t.get("gpu_util_pct"),
|
||||
t.get("mem_util_pct"), t.get("temp_c"), t.get("power_w"), t.get("power_limit_w"),
|
||||
t.get("fan_pct"), t.get("clock_sm_mhz"), t.get("clock_mem_mhz"),
|
||||
t.get("vram_used_bytes"), t.get("ollama_bytes"), t.get("comfy_bytes"),
|
||||
t.get("system_bytes"), t.get("ram_used_bytes"), t.get("ram_cached_bytes"),
|
||||
t.get("pcie_tx_kbps"), t.get("pcie_rx_kbps"), t.get("throttle_reasons"),
|
||||
))
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
|
||||
# --------------------------------------------------------------- writer round-trip
|
||||
|
||||
def test_start_creates_the_schema(temp_db):
|
||||
"""The service starts against a database that may not exist yet."""
|
||||
ts.start()
|
||||
ts.stop()
|
||||
assert os.path.exists(temp_db)
|
||||
conn = sqlite3.connect(temp_db)
|
||||
tables = {r[0] for r in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
||||
conn.close()
|
||||
assert {"telemetry", "events", "autotune_runs"} <= tables
|
||||
|
||||
|
||||
def test_record_event_round_trips_through_the_writer(temp_db):
|
||||
"""Events are committed immediately (not batched) because they are what the analytics
|
||||
are built on."""
|
||||
ts.start()
|
||||
ts.record_event({
|
||||
"event_type": "LLM Model Switch", "source": "None", "target": "llama3:8b",
|
||||
"duration_ms": 5120.0, "load_duration_ms": 4901.0, "tokens_per_sec": 61.2,
|
||||
"bytes_loaded": 13819000000, "load_gbps": 2.63, "cache_status": "RAM Cache Hit ⚡",
|
||||
}, profile="ollama")
|
||||
_drain()
|
||||
ts.stop()
|
||||
|
||||
rows = ts.recent_events()
|
||||
assert len(rows) == 1
|
||||
assert rows[0]["target"] == "llama3:8b"
|
||||
assert rows[0]["profile"] == "ollama"
|
||||
assert rows[0]["cache_status"] == "RAM Cache Hit ⚡"
|
||||
assert rows[0]["load_gbps"] == pytest.approx(2.63)
|
||||
|
||||
|
||||
def test_record_telemetry_flushes_on_stop(temp_db):
|
||||
"""Telemetry is batched on a FLUSH_INTERVAL_S timer; stop() must not drop what is
|
||||
still pending, or the last seconds before a restart are lost."""
|
||||
ts.start()
|
||||
ts.record_telemetry(
|
||||
gpu={"available": True, "gpu_util_pct": 88.0, "temperature_c": 71.0,
|
||||
"power_w": 355.0, "clock_sm_mhz": 2790.0, "clock_mem_mhz": 10501.0,
|
||||
"vram_used_bytes": 13 * 1024 ** 3,
|
||||
"breakdown": {"ollama_gb": 12.0, "comfyui_gb": 0.0, "system_gb": 0.5}},
|
||||
ram={"used_bytes": 20 * 1024 ** 3, "cached_bytes": 30 * 1024 ** 3},
|
||||
profile="ollama", throttle_reasons="sw_power_cap")
|
||||
_drain()
|
||||
ts.stop()
|
||||
|
||||
conn = sqlite3.connect(temp_db)
|
||||
row = conn.execute("SELECT profile, temp_c, ollama_bytes, throttle_reasons "
|
||||
"FROM telemetry").fetchone()
|
||||
conn.close()
|
||||
assert row[0] == "ollama"
|
||||
assert row[1] == pytest.approx(71.0)
|
||||
assert row[2] == 12 * 1024 ** 3
|
||||
assert row[3] == "sw_power_cap"
|
||||
|
||||
|
||||
def test_record_telemetry_drops_unavailable_gpu_samples(temp_db):
|
||||
"""A failed NVML read must not write a row of nulls that then skews every average."""
|
||||
ts.start()
|
||||
ts.record_telemetry(gpu={"available": False}, ram={}, profile="ollama")
|
||||
_drain()
|
||||
ts.stop()
|
||||
conn = sqlite3.connect(temp_db)
|
||||
n = conn.execute("SELECT COUNT(*) FROM telemetry").fetchone()[0]
|
||||
conn.close()
|
||||
assert n == 0
|
||||
|
||||
|
||||
def test_recording_without_a_running_writer_is_a_no_op(temp_db):
|
||||
"""Callers on the hot path must never need to know whether the store was started."""
|
||||
ts.record_event({"event_type": "LLM Model Switch", "target": "x"})
|
||||
ts.record_telemetry(gpu={"available": True, "breakdown": {}}, ram={})
|
||||
assert not os.path.exists(temp_db)
|
||||
|
||||
|
||||
def test_record_event_never_raises_on_a_malformed_event(temp_db):
|
||||
"""This is called from the swap path; it is documented as never raising."""
|
||||
ts.start()
|
||||
ts.record_event({})
|
||||
ts.record_event({"event_type": "Weird", "duration_ms": "not-a-number"})
|
||||
_drain()
|
||||
ts.stop()
|
||||
|
||||
|
||||
def test_autotune_row_round_trips(temp_db):
|
||||
ts.start()
|
||||
ts.record_autotune({"profile": "ollama", "knob": "lock_mem_mhz", "mem_offset_mhz": 0,
|
||||
"tokens_per_sec": 62.5, "temp_c": 74.0, "stable": True})
|
||||
_drain()
|
||||
ts.stop()
|
||||
rows = ts.autotune_history()
|
||||
assert len(rows) == 1
|
||||
assert rows[0]["knob"] == "lock_mem_mhz"
|
||||
assert rows[0]["stable"] == 1 # stored as an INTEGER, not a bool
|
||||
|
||||
|
||||
# --------------------------------------------------------------- absent database
|
||||
|
||||
def test_queries_return_empty_when_the_database_does_not_exist(tmp_path, monkeypatch):
|
||||
"""A dashboard opened before the first sample must render empty, not 500."""
|
||||
monkeypatch.setattr(ts, "DB_PATH", str(tmp_path / "absent.db"))
|
||||
assert ts.profile_comparison() == []
|
||||
assert ts.model_usage_ranking() == []
|
||||
assert ts.timeseries() == []
|
||||
assert ts.recent_events() == []
|
||||
assert ts.autotune_history() == []
|
||||
stats = ts.swap_stats()
|
||||
assert stats["by_type"] == [] and stats["by_model"] == [] and stats["by_cache_status"] == []
|
||||
assert ts.db_info()["exists"] is False
|
||||
|
||||
|
||||
def test_queries_return_empty_on_a_corrupt_database(tmp_path, monkeypatch):
|
||||
db = tmp_path / "corrupt.db"
|
||||
db.write_bytes(b"this is not a sqlite file")
|
||||
monkeypatch.setattr(ts, "DB_PATH", str(db))
|
||||
assert ts.recent_events() == []
|
||||
assert ts.profile_comparison() == []
|
||||
|
||||
|
||||
# --------------------------------------------------------------- analytics
|
||||
|
||||
def test_profile_comparison_ranks_profiles_by_throughput(temp_db):
|
||||
"""The headline question this store exists to answer: which profile is actually
|
||||
faster? Ordering is by average tok/s, descending."""
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 60, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 62.0, "load_gbps": 2.6, "load_duration_ms": 4900},
|
||||
{"ts": now - 50, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 64.0, "load_gbps": 2.6, "load_duration_ms": 4900},
|
||||
{"ts": now - 40, "event_type": "LLM Model Switch", "profile": "balanced",
|
||||
"tokens_per_sec": 51.0, "load_gbps": 2.5, "load_duration_ms": 5100},
|
||||
], telemetry=[
|
||||
{"ts": now - 55, "profile": "ollama", "gpu_util_pct": 90, "temp_c": 74.0,
|
||||
"power_w": 360.0, "clock_sm_mhz": 2790, "clock_mem_mhz": 10501},
|
||||
{"ts": now - 45, "profile": "balanced", "gpu_util_pct": 90, "temp_c": 66.0,
|
||||
"power_w": 300.0, "clock_sm_mhz": 2600, "clock_mem_mhz": 9501},
|
||||
])
|
||||
out = ts.profile_comparison(days=1)
|
||||
assert [r["profile"] for r in out] == ["ollama", "balanced"]
|
||||
assert out[0]["swaps"] == 2
|
||||
assert out[0]["avg_tok_s"] == pytest.approx(63.0)
|
||||
# Thermals are joined in from the telemetry table for the same profile.
|
||||
assert out[0]["avg_temp_c"] == pytest.approx(74.0)
|
||||
|
||||
|
||||
def test_profile_comparison_ignores_events_without_measured_throughput(temp_db):
|
||||
"""A swap that produced no tokens tells us nothing about the profile's speed and
|
||||
would drag the average toward zero."""
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 10, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 60.0},
|
||||
{"ts": now - 5, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 0.0},
|
||||
])
|
||||
out = ts.profile_comparison(days=1)
|
||||
assert out[0]["swaps"] == 1 and out[0]["avg_tok_s"] == pytest.approx(60.0)
|
||||
|
||||
|
||||
def test_profile_comparison_excludes_samples_outside_the_window(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 40 * 86400, "event_type": "LLM Model Switch", "profile": "old",
|
||||
"tokens_per_sec": 99.0},
|
||||
{"ts": now - 60, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 60.0},
|
||||
])
|
||||
assert [r["profile"] for r in ts.profile_comparison(days=7)] == ["ollama"]
|
||||
|
||||
|
||||
def test_profile_comparison_ignores_idle_telemetry(temp_db):
|
||||
"""Thermals are only meaningful while the GPU is doing work; idle samples (util <= 5%)
|
||||
would make every profile look cool."""
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 10, "event_type": "LLM Model Switch", "profile": "ollama",
|
||||
"tokens_per_sec": 60.0},
|
||||
], telemetry=[
|
||||
{"ts": now - 12, "profile": "ollama", "gpu_util_pct": 0, "temp_c": 35.0},
|
||||
{"ts": now - 11, "profile": "ollama", "gpu_util_pct": 95, "temp_c": 75.0},
|
||||
])
|
||||
assert ts.profile_comparison(days=1)[0]["avg_temp_c"] == pytest.approx(75.0)
|
||||
|
||||
|
||||
def test_swap_stats_groups_by_event_type_and_cache_status(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 30, "event_type": "LLM Model Switch", "target": "llama3:8b",
|
||||
"duration_ms": 5000, "cache_status": "RAM Cache Hit ⚡", "load_gbps": 2.6,
|
||||
"tokens_per_sec": 60.0},
|
||||
{"ts": now - 20, "event_type": "LLM Model Switch", "target": "llama3:8b",
|
||||
"duration_ms": 7000, "cache_status": "Cold Disk Load 💾", "load_gbps": 0.38,
|
||||
"tokens_per_sec": 58.0},
|
||||
{"ts": now - 10, "event_type": "VRAM Yield", "duration_ms": 900,
|
||||
"yield_confirm_ms": 850},
|
||||
])
|
||||
stats = ts.swap_stats(days=1)
|
||||
by_type = {r["event_type"]: r for r in stats["by_type"]}
|
||||
assert by_type["LLM Model Switch"]["n"] == 2
|
||||
assert by_type["LLM Model Switch"]["avg_ms"] == pytest.approx(6000.0)
|
||||
assert by_type["LLM Model Switch"]["min_ms"] == pytest.approx(5000.0)
|
||||
assert by_type["VRAM Yield"]["avg_confirm_ms"] == pytest.approx(850.0)
|
||||
assert {r["cache_status"] for r in stats["by_cache_status"]} == {
|
||||
"RAM Cache Hit ⚡", "Cold Disk Load 💾"}
|
||||
assert stats["by_model"][0]["model"] == "llama3:8b"
|
||||
assert stats["by_model"][0]["loads"] == 2
|
||||
|
||||
|
||||
def test_model_usage_ranking_scores_recent_use_higher(temp_db):
|
||||
"""Recency/frequency drives the RAM warm budget: given equal load counts, the model
|
||||
used more recently must rank first (half-life ~24 h)."""
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 3600, "event_type": "LLM Model Switch", "target": "recent:latest"},
|
||||
{"ts": now - 3600, "event_type": "LLM Model Switch", "target": "recent:latest"},
|
||||
{"ts": now - 20 * 86400, "event_type": "LLM Model Switch", "target": "stale:latest"},
|
||||
{"ts": now - 20 * 86400, "event_type": "LLM Model Switch", "target": "stale:latest"},
|
||||
])
|
||||
ranking = ts.model_usage_ranking(days=30)
|
||||
assert [r["model"] for r in ranking] == ["recent:latest", "stale:latest"]
|
||||
assert ranking[0]["score"] > ranking[1]["score"]
|
||||
assert ranking[0]["loads"] == 2
|
||||
assert ranking[0]["age_hours"] == pytest.approx(1.0, abs=0.05)
|
||||
|
||||
|
||||
def test_model_usage_ranking_scores_frequent_use_higher_at_equal_recency(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 3600, "event_type": "LLM Model Switch", "target": "often:latest"},
|
||||
{"ts": now - 3601, "event_type": "LLM Model Switch", "target": "often:latest"},
|
||||
{"ts": now - 3602, "event_type": "LLM Model Switch", "target": "often:latest"},
|
||||
{"ts": now - 3600, "event_type": "LLM Model Switch", "target": "once:latest"},
|
||||
])
|
||||
ranking = ts.model_usage_ranking(days=30)
|
||||
assert ranking[0]["model"] == "often:latest"
|
||||
|
||||
|
||||
def test_model_usage_ranking_counts_warms_as_well_as_switches(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db, events=[
|
||||
{"ts": now - 60, "event_type": "Model Warm", "target": "warmed:latest"},
|
||||
{"ts": now - 60, "event_type": "Something Else", "target": "ignored:latest"},
|
||||
])
|
||||
assert [r["model"] for r in ts.model_usage_ranking(days=1)] == ["warmed:latest"]
|
||||
|
||||
|
||||
def test_timeseries_buckets_samples_by_width(temp_db):
|
||||
"""Six hours into six buckets is one bucket per hour; samples inside an hour collapse
|
||||
into a single averaged point."""
|
||||
now = time.time()
|
||||
base = now - 5.5 * 3600
|
||||
_seed(temp_db, telemetry=[
|
||||
{"ts": base + 60, "temp_c": 60.0, "gpu_util_pct": 50},
|
||||
{"ts": base + 120, "temp_c": 70.0, "gpu_util_pct": 70},
|
||||
{"ts": base + 3700, "temp_c": 80.0, "gpu_util_pct": 90},
|
||||
])
|
||||
out = ts.timeseries(hours=6, buckets=6)
|
||||
assert len(out) == 2
|
||||
assert out[0]["temp_c"] == pytest.approx(65.0) # average of 60 and 70
|
||||
assert out[1]["temp_c"] == pytest.approx(80.0)
|
||||
assert out[0]["bucket_ts"] < out[1]["bucket_ts"]
|
||||
|
||||
|
||||
def test_timeseries_excludes_samples_older_than_the_window(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db, telemetry=[
|
||||
{"ts": now - 48 * 3600, "temp_c": 99.0},
|
||||
{"ts": now - 60, "temp_c": 60.0},
|
||||
])
|
||||
out = ts.timeseries(hours=6, buckets=240)
|
||||
assert len(out) == 1 and out[0]["temp_c"] == pytest.approx(60.0)
|
||||
|
||||
|
||||
def test_db_info_reports_row_counts_and_coverage(temp_db):
|
||||
now = time.time()
|
||||
_seed(temp_db,
|
||||
events=[{"ts": now - 10, "event_type": "LLM Model Switch", "target": "m"}],
|
||||
telemetry=[{"ts": now - 7200, "temp_c": 60.0}, {"ts": now, "temp_c": 61.0}])
|
||||
info = ts.db_info()
|
||||
assert info["exists"] is True
|
||||
assert info["events_rows"] == 1
|
||||
assert info["telemetry_rows"] == 2
|
||||
assert info["coverage_hours"] == pytest.approx(2.0, abs=0.01)
|
||||
Reference in New Issue
Block a user