"""telemetry_store: schema, the background writer round-trip, and the analytics queries. Every test in this module runs against a throwaway SQLite file via the `temp_db` fixture (which monkeypatches telemetry_store.DB_PATH). The production hyperswap.db is never opened: the writer thread, _read_conn() and _rows() all resolve DB_PATH at call time. """ import os import sqlite3 import time import pytest import telemetry_store as ts def _drain(timeout=3.0): """Wait until the writer thread has dequeued everything submitted so far. NOTE: stop() flushes the writer's *pending* batch but does not drain the submission queue, so a stop() racing a just-submitted row can drop it. Tests wait explicitly rather than depending on that race. (See tests/README.md.) """ w = ts._writer assert w is not None, "writer not started" deadline = time.time() + timeout while time.time() < deadline: if w.q.empty(): time.sleep(0.05) # let the in-flight item finish its execute/commit return time.sleep(0.01) raise AssertionError("telemetry writer never drained its queue") def _seed(db_path, events=(), telemetry=()): """Insert rows directly, bypassing the writer thread, for query tests.""" conn = sqlite3.connect(db_path) conn.executescript(ts.SCHEMA) for e in events: conn.execute(ts._EVENT_SQL, ( e.get("ts", time.time()), e.get("event_type"), e.get("source"), e.get("target"), e.get("profile"), e.get("duration_ms"), e.get("load_duration_ms"), e.get("yield_confirm_ms"), e.get("tokens_per_sec"), e.get("bytes_loaded"), e.get("load_gbps"), e.get("cache_status"), e.get("detail"), )) for t in telemetry: conn.execute(ts._TELEMETRY_SQL, ( t.get("ts", time.time()), t.get("profile"), t.get("gpu_util_pct"), t.get("mem_util_pct"), t.get("temp_c"), t.get("power_w"), t.get("power_limit_w"), t.get("fan_pct"), t.get("clock_sm_mhz"), t.get("clock_mem_mhz"), t.get("vram_used_bytes"), t.get("ollama_bytes"), t.get("comfy_bytes"), t.get("system_bytes"), t.get("ram_used_bytes"), t.get("ram_cached_bytes"), t.get("pcie_tx_kbps"), t.get("pcie_rx_kbps"), t.get("throttle_reasons"), )) conn.commit() conn.close() # --------------------------------------------------------------- writer round-trip def test_start_creates_the_schema(temp_db): """The service starts against a database that may not exist yet.""" ts.start() ts.stop() assert os.path.exists(temp_db) conn = sqlite3.connect(temp_db) tables = {r[0] for r in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")} conn.close() assert {"telemetry", "events", "autotune_runs"} <= tables def test_record_event_round_trips_through_the_writer(temp_db): """Events are committed immediately (not batched) because they are what the analytics are built on.""" ts.start() ts.record_event({ "event_type": "LLM Model Switch", "source": "None", "target": "llama3:8b", "duration_ms": 5120.0, "load_duration_ms": 4901.0, "tokens_per_sec": 61.2, "bytes_loaded": 13819000000, "load_gbps": 2.63, "cache_status": "RAM Cache Hit ⚡", }, profile="ollama") _drain() ts.stop() rows = ts.recent_events() assert len(rows) == 1 assert rows[0]["target"] == "llama3:8b" assert rows[0]["profile"] == "ollama" assert rows[0]["cache_status"] == "RAM Cache Hit ⚡" assert rows[0]["load_gbps"] == pytest.approx(2.63) def test_record_telemetry_flushes_on_stop(temp_db): """Telemetry is batched on a FLUSH_INTERVAL_S timer; stop() must not drop what is still pending, or the last seconds before a restart are lost.""" ts.start() ts.record_telemetry( gpu={"available": True, "gpu_util_pct": 88.0, "temperature_c": 71.0, "power_w": 355.0, "clock_sm_mhz": 2790.0, "clock_mem_mhz": 10501.0, "vram_used_bytes": 13 * 1024 ** 3, "breakdown": {"ollama_gb": 12.0, "comfyui_gb": 0.0, "system_gb": 0.5}}, ram={"used_bytes": 20 * 1024 ** 3, "cached_bytes": 30 * 1024 ** 3}, profile="ollama", throttle_reasons="sw_power_cap") _drain() ts.stop() conn = sqlite3.connect(temp_db) row = conn.execute("SELECT profile, temp_c, ollama_bytes, throttle_reasons " "FROM telemetry").fetchone() conn.close() assert row[0] == "ollama" assert row[1] == pytest.approx(71.0) assert row[2] == 12 * 1024 ** 3 assert row[3] == "sw_power_cap" def test_record_telemetry_drops_unavailable_gpu_samples(temp_db): """A failed NVML read must not write a row of nulls that then skews every average.""" ts.start() ts.record_telemetry(gpu={"available": False}, ram={}, profile="ollama") _drain() ts.stop() conn = sqlite3.connect(temp_db) n = conn.execute("SELECT COUNT(*) FROM telemetry").fetchone()[0] conn.close() assert n == 0 def test_recording_without_a_running_writer_is_a_no_op(temp_db): """Callers on the hot path must never need to know whether the store was started.""" ts.record_event({"event_type": "LLM Model Switch", "target": "x"}) ts.record_telemetry(gpu={"available": True, "breakdown": {}}, ram={}) assert not os.path.exists(temp_db) def test_record_event_never_raises_on_a_malformed_event(temp_db): """This is called from the swap path; it is documented as never raising.""" ts.start() ts.record_event({}) ts.record_event({"event_type": "Weird", "duration_ms": "not-a-number"}) _drain() ts.stop() def test_autotune_row_round_trips(temp_db): ts.start() ts.record_autotune({"profile": "ollama", "knob": "lock_mem_mhz", "mem_offset_mhz": 0, "tokens_per_sec": 62.5, "temp_c": 74.0, "stable": True}) _drain() ts.stop() rows = ts.autotune_history() assert len(rows) == 1 assert rows[0]["knob"] == "lock_mem_mhz" assert rows[0]["stable"] == 1 # stored as an INTEGER, not a bool # --------------------------------------------------------------- absent database def test_queries_return_empty_when_the_database_does_not_exist(tmp_path, monkeypatch): """A dashboard opened before the first sample must render empty, not 500.""" monkeypatch.setattr(ts, "DB_PATH", str(tmp_path / "absent.db")) assert ts.profile_comparison() == [] assert ts.model_usage_ranking() == [] assert ts.timeseries() == [] assert ts.recent_events() == [] assert ts.autotune_history() == [] stats = ts.swap_stats() assert stats["by_type"] == [] and stats["by_model"] == [] and stats["by_cache_status"] == [] assert ts.db_info()["exists"] is False def test_queries_return_empty_on_a_corrupt_database(tmp_path, monkeypatch): db = tmp_path / "corrupt.db" db.write_bytes(b"this is not a sqlite file") monkeypatch.setattr(ts, "DB_PATH", str(db)) assert ts.recent_events() == [] assert ts.profile_comparison() == [] # --------------------------------------------------------------- analytics def test_profile_comparison_ranks_profiles_by_throughput(temp_db): """The headline question this store exists to answer: which profile is actually faster? Ordering is by average tok/s, descending.""" now = time.time() _seed(temp_db, events=[ {"ts": now - 60, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 62.0, "load_gbps": 2.6, "load_duration_ms": 4900}, {"ts": now - 50, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 64.0, "load_gbps": 2.6, "load_duration_ms": 4900}, {"ts": now - 40, "event_type": "LLM Model Switch", "profile": "balanced", "tokens_per_sec": 51.0, "load_gbps": 2.5, "load_duration_ms": 5100}, ], telemetry=[ {"ts": now - 55, "profile": "ollama", "gpu_util_pct": 90, "temp_c": 74.0, "power_w": 360.0, "clock_sm_mhz": 2790, "clock_mem_mhz": 10501}, {"ts": now - 45, "profile": "balanced", "gpu_util_pct": 90, "temp_c": 66.0, "power_w": 300.0, "clock_sm_mhz": 2600, "clock_mem_mhz": 9501}, ]) out = ts.profile_comparison(days=1) assert [r["profile"] for r in out] == ["ollama", "balanced"] assert out[0]["swaps"] == 2 assert out[0]["avg_tok_s"] == pytest.approx(63.0) # Thermals are joined in from the telemetry table for the same profile. assert out[0]["avg_temp_c"] == pytest.approx(74.0) def test_profile_comparison_ignores_events_without_measured_throughput(temp_db): """A swap that produced no tokens tells us nothing about the profile's speed and would drag the average toward zero.""" now = time.time() _seed(temp_db, events=[ {"ts": now - 10, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 60.0}, {"ts": now - 5, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 0.0}, ]) out = ts.profile_comparison(days=1) assert out[0]["swaps"] == 1 and out[0]["avg_tok_s"] == pytest.approx(60.0) def test_profile_comparison_excludes_samples_outside_the_window(temp_db): now = time.time() _seed(temp_db, events=[ {"ts": now - 40 * 86400, "event_type": "LLM Model Switch", "profile": "old", "tokens_per_sec": 99.0}, {"ts": now - 60, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 60.0}, ]) assert [r["profile"] for r in ts.profile_comparison(days=7)] == ["ollama"] def test_profile_comparison_ignores_idle_telemetry(temp_db): """Thermals are only meaningful while the GPU is doing work; idle samples (util <= 5%) would make every profile look cool.""" now = time.time() _seed(temp_db, events=[ {"ts": now - 10, "event_type": "LLM Model Switch", "profile": "ollama", "tokens_per_sec": 60.0}, ], telemetry=[ {"ts": now - 12, "profile": "ollama", "gpu_util_pct": 0, "temp_c": 35.0}, {"ts": now - 11, "profile": "ollama", "gpu_util_pct": 95, "temp_c": 75.0}, ]) assert ts.profile_comparison(days=1)[0]["avg_temp_c"] == pytest.approx(75.0) def test_swap_stats_groups_by_event_type_and_cache_status(temp_db): now = time.time() _seed(temp_db, events=[ {"ts": now - 30, "event_type": "LLM Model Switch", "target": "llama3:8b", "duration_ms": 5000, "cache_status": "RAM Cache Hit ⚡", "load_gbps": 2.6, "tokens_per_sec": 60.0}, {"ts": now - 20, "event_type": "LLM Model Switch", "target": "llama3:8b", "duration_ms": 7000, "cache_status": "Cold Disk Load 💾", "load_gbps": 0.38, "tokens_per_sec": 58.0}, {"ts": now - 10, "event_type": "VRAM Yield", "duration_ms": 900, "yield_confirm_ms": 850}, ]) stats = ts.swap_stats(days=1) by_type = {r["event_type"]: r for r in stats["by_type"]} assert by_type["LLM Model Switch"]["n"] == 2 assert by_type["LLM Model Switch"]["avg_ms"] == pytest.approx(6000.0) assert by_type["LLM Model Switch"]["min_ms"] == pytest.approx(5000.0) assert by_type["VRAM Yield"]["avg_confirm_ms"] == pytest.approx(850.0) assert {r["cache_status"] for r in stats["by_cache_status"]} == { "RAM Cache Hit ⚡", "Cold Disk Load 💾"} assert stats["by_model"][0]["model"] == "llama3:8b" assert stats["by_model"][0]["loads"] == 2 def test_model_usage_ranking_scores_recent_use_higher(temp_db): """Recency/frequency drives the RAM warm budget: given equal load counts, the model used more recently must rank first (half-life ~24 h).""" now = time.time() _seed(temp_db, events=[ {"ts": now - 3600, "event_type": "LLM Model Switch", "target": "recent:latest"}, {"ts": now - 3600, "event_type": "LLM Model Switch", "target": "recent:latest"}, {"ts": now - 20 * 86400, "event_type": "LLM Model Switch", "target": "stale:latest"}, {"ts": now - 20 * 86400, "event_type": "LLM Model Switch", "target": "stale:latest"}, ]) ranking = ts.model_usage_ranking(days=30) assert [r["model"] for r in ranking] == ["recent:latest", "stale:latest"] assert ranking[0]["score"] > ranking[1]["score"] assert ranking[0]["loads"] == 2 assert ranking[0]["age_hours"] == pytest.approx(1.0, abs=0.05) def test_model_usage_ranking_scores_frequent_use_higher_at_equal_recency(temp_db): now = time.time() _seed(temp_db, events=[ {"ts": now - 3600, "event_type": "LLM Model Switch", "target": "often:latest"}, {"ts": now - 3601, "event_type": "LLM Model Switch", "target": "often:latest"}, {"ts": now - 3602, "event_type": "LLM Model Switch", "target": "often:latest"}, {"ts": now - 3600, "event_type": "LLM Model Switch", "target": "once:latest"}, ]) ranking = ts.model_usage_ranking(days=30) assert ranking[0]["model"] == "often:latest" def test_model_usage_ranking_counts_warms_as_well_as_switches(temp_db): now = time.time() _seed(temp_db, events=[ {"ts": now - 60, "event_type": "Model Warm", "target": "warmed:latest"}, {"ts": now - 60, "event_type": "Something Else", "target": "ignored:latest"}, ]) assert [r["model"] for r in ts.model_usage_ranking(days=1)] == ["warmed:latest"] def test_timeseries_buckets_samples_by_width(temp_db): """Six hours into six buckets is one bucket per hour; samples inside an hour collapse into a single averaged point.""" now = time.time() base = now - 5.5 * 3600 _seed(temp_db, telemetry=[ {"ts": base + 60, "temp_c": 60.0, "gpu_util_pct": 50}, {"ts": base + 120, "temp_c": 70.0, "gpu_util_pct": 70}, {"ts": base + 3700, "temp_c": 80.0, "gpu_util_pct": 90}, ]) out = ts.timeseries(hours=6, buckets=6) assert len(out) == 2 assert out[0]["temp_c"] == pytest.approx(65.0) # average of 60 and 70 assert out[1]["temp_c"] == pytest.approx(80.0) assert out[0]["bucket_ts"] < out[1]["bucket_ts"] def test_timeseries_excludes_samples_older_than_the_window(temp_db): now = time.time() _seed(temp_db, telemetry=[ {"ts": now - 48 * 3600, "temp_c": 99.0}, {"ts": now - 60, "temp_c": 60.0}, ]) out = ts.timeseries(hours=6, buckets=240) assert len(out) == 1 and out[0]["temp_c"] == pytest.approx(60.0) def test_db_info_reports_row_counts_and_coverage(temp_db): now = time.time() _seed(temp_db, events=[{"ts": now - 10, "event_type": "LLM Model Switch", "target": "m"}], telemetry=[{"ts": now - 7200, "temp_c": 60.0}, {"ts": now, "temp_c": 61.0}]) info = ts.db_info() assert info["exists"] is True assert info["events_rows"] == 1 assert info["telemetry_rows"] == 2 assert info["coverage_hours"] == pytest.approx(2.0, abs=0.01)