"""ram_optimizer: page-cache residency measurement, model discovery, warm planning. All file IO here happens against files this test creates in tmp_path. Nothing reads a real model blob, and nothing calls warm_file_to_ram. """ import json import os import pytest import ram_optimizer as ro MIB = 1024 * 1024 # --------------------------------------------------------------- warm_confident rule class _FakeCachestat: """Stand-in for the kernel's struct cachestat.""" def __init__(self, nr_cache): self.nr_cache = nr_cache self.nr_dirty = 0 self.nr_evicted = 0 def _file_of_pages(tmp_path, pages): p = tmp_path / f"blob_{pages}.bin" p.write_bytes(b"\0" * (pages * ro.PAGE_SIZE)) return str(p) def _force_cachestat(monkeypatch, resident_pages): monkeypatch.setattr(ro, "_cachestat", lambda fd, offset, length: _FakeCachestat(resident_pages)) def _force_probe(monkeypatch, pct): monkeypatch.setattr(ro, "_cachestat", lambda fd, offset, length: None) monkeypatch.setattr(ro, "_throughput_probe", lambda fd, size, windows=None: {"resident_pct": pct, "windows": 12, "median_gbps": 3.0}) @pytest.mark.parametrize("resident_pages,expect_confident", [ (100, True), # 100% exact (90, True), # exactly WARM_SKIP_THRESHOLD_PCT (89, False), # just under the threshold (0, False), ]) def test_cachestat_reading_is_trusted_at_or_above_threshold(tmp_path, monkeypatch, resident_pages, expect_confident): """An exact cachestat reading >= WARM_SKIP_THRESHOLD_PCT (90) may be trusted to skip warming; below it, it may not.""" path = _file_of_pages(tmp_path, 100) _force_cachestat(monkeypatch, resident_pages) res = ro.page_residency(path) assert res["method"] == "cachestat" assert res["resident_pct"] == pytest.approx(float(resident_pages)) assert res["warm_confident"] is expect_confident @pytest.mark.parametrize("pct,expect_confident", [ (100.0, True), # only an unblemished probe is trustworthy (99.9, False), (95.0, False), # would pass `warm`, must NOT pass `warm_confident` (90.0, False), ]) def test_probe_reading_is_only_trusted_at_exactly_100_percent(tmp_path, monkeypatch, pct, expect_confident): """Correctness fix, not a style choice: a 12-window probe once cleared 90% on a file that was mostly cold (a 12.87 GB blob reported 'already resident' then loaded at 2.44 GB/s). Only a perfect probe score may skip work.""" path = _file_of_pages(tmp_path, 100) _force_probe(monkeypatch, pct) res = ro.page_residency(path) assert res["method"] == "probe" assert res["warm_confident"] is expect_confident def test_probe_at_95_percent_is_warm_but_not_confident(tmp_path, monkeypatch): """`warm` and `warm_confident` are different questions and must not be conflated: warm drives display, warm_confident drives skipping work.""" path = _file_of_pages(tmp_path, 100) _force_probe(monkeypatch, 95.0) res = ro.page_residency(path) assert res["warm"] is True assert res["warm_confident"] is False def test_warm_skip_threshold_constant_unchanged(): # Pinned: build_warm_plan and the dashboard both key off this number. assert ro.WARM_SKIP_THRESHOLD_PCT == 90.0 # --------------------------------------------------------------- page_residency basics def test_page_residency_on_missing_file_reports_not_measurable(tmp_path): res = ro.page_residency(str(tmp_path / "nope.bin")) assert res["success"] is False assert res["measurable"] is False assert res["resident_pct"] == 0.0 def test_page_residency_on_empty_file_short_circuits(tmp_path): """A zero-byte file has no pages to measure; dividing by its page count would throw.""" p = tmp_path / "empty.bin" p.write_bytes(b"") res = ro.page_residency(str(p)) assert res["method"] == "empty" assert res["resident_pct"] == 0.0 assert res["measurable"] is True def test_page_residency_refuses_to_guess_when_probing_is_disallowed(tmp_path, monkeypatch): """With cachestat unavailable and allow_probe=False the answer is 'unknown', never 0%. Reporting 0% would make the planner re-warm a file that may be fully resident.""" path = _file_of_pages(tmp_path, 8) monkeypatch.setattr(ro, "_cachestat", lambda fd, offset, length: None) res = ro.page_residency(path, allow_probe=False) assert res["measurable"] is False assert res["resident_pct"] is None assert res["warm"] is None assert res["method"] == "unavailable" def test_page_residency_reports_size_and_bytes_consistently(tmp_path, monkeypatch): path = _file_of_pages(tmp_path, 100) _force_cachestat(monkeypatch, 50) res = ro.page_residency(path) size = os.path.getsize(path) assert res["size_bytes"] == size assert res["resident_bytes"] == int(size * 0.5) def test_page_residency_against_a_real_file_holds_the_confidence_invariant(tmp_path): """End-to-end with the real kernel path (whichever method is available here): the warm_confident rule must hold for whatever the machine actually reports.""" path = _file_of_pages(tmp_path, 512) res = ro.page_residency(path) assert res["success"] is True assert res["method"] in ("cachestat", "probe") assert 0.0 <= res["resident_pct"] <= 100.0 expected = ((res["method"] == "cachestat" and res["resident_pct"] >= 90.0) or (res["method"] == "probe" and res["resident_pct"] >= 100.0)) assert res["warm_confident"] is expected # --------------------------------------------------------------- throughput probe def test_throughput_probe_window_count_is_clamped_to_file_size(tmp_path): """A file smaller than PROBE_WINDOWS * PROBE_WINDOW_BYTES must not be probed with more windows than it has, or offsets run past EOF.""" p = tmp_path / "small.bin" p.write_bytes(b"\0" * (5 * MIB)) fd = os.open(str(p), os.O_RDONLY) try: out = ro._throughput_probe(fd, 5 * MIB) finally: os.close(fd) assert 0 < out["windows"] <= 5 * MIB // ro.PROBE_WINDOW_BYTES assert 0.0 <= out["resident_pct"] <= 100.0 def test_throughput_probe_honours_window_override(tmp_path): p = tmp_path / "medium.bin" p.write_bytes(b"\0" * (32 * MIB)) fd = os.open(str(p), os.O_RDONLY) try: out = ro._throughput_probe(fd, 32 * MIB, windows_override=4) finally: os.close(fd) assert out["windows"] == 4 def test_probe_cached_threshold_sits_between_measured_disk_and_cache_rates(): # Measured on this box: cold NVMe 0.35-0.5 GB/s, page cache 3.2-13 GB/s. assert 0.5 < ro.PROBE_CACHED_GBPS < 3.2 # --------------------------------------------------------------- ollama manifest parsing MODEL_MEDIA_TYPE = "application/vnd.ollama.image.model" def _write_manifest(root, rel_dir, tag, layers): d = os.path.join(root, "manifests", rel_dir) os.makedirs(d, exist_ok=True) with open(os.path.join(d, tag), "w") as f: json.dump({"layers": layers}, f) def _write_blob(root, digest, size): blobs = os.path.join(root, "blobs") os.makedirs(blobs, exist_ok=True) path = os.path.join(blobs, digest.replace(":", "-")) with open(path, "wb") as f: f.write(b"\0" * size) return path @pytest.fixture def ollama_tree(tmp_path, monkeypatch): root = tmp_path / "ollama-models" root.mkdir() monkeypatch.setattr(ro, "OLLAMA_MODEL_DIRS", [str(root)]) return str(root) def test_find_ollama_model_files_maps_library_model_to_its_blob(ollama_tree): """registry/library// is the common case and must render as 'name:tag'.""" _write_blob(ollama_tree, "sha256:aaa111", 4096) _write_manifest(ollama_tree, "registry.ollama.ai/library/llama3", "8b", [ {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:aaa111", "size": 987654321}, ]) files = ro.find_ollama_model_files() assert len(files) == 1 entry = files[0] assert entry["model"] == "llama3:8b" assert entry["filename"] == "sha256-aaa111" assert entry["size_bytes"] == 987654321 # taken from the manifest, not the stub blob assert entry["kind"] == "ollama" assert os.path.exists(entry["full_path"]) def test_find_ollama_model_files_keeps_non_library_namespace(ollama_tree): """A model pulled from a user namespace must keep it, or two different models with the same short name collide in the warm plan.""" _write_blob(ollama_tree, "sha256:bbb222", 4096) _write_manifest(ollama_tree, "hf.co/someuser/qwen-coder", "q4", [ {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:bbb222", "size": 100}, ]) files = ro.find_ollama_model_files() assert [f["model"] for f in files] == ["someuser/qwen-coder:q4"] def test_find_ollama_model_files_ignores_non_model_layers(ollama_tree): """Manifests also list templates, params and licence layers; warming those is wasted IO and they are not the weights.""" _write_blob(ollama_tree, "sha256:ccc333", 4096) _write_blob(ollama_tree, "sha256:ddd444", 32) _write_manifest(ollama_tree, "registry.ollama.ai/library/mistral", "7b", [ {"mediaType": "application/vnd.ollama.image.template", "digest": "sha256:ddd444", "size": 32}, {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:ccc333", "size": 500}, ]) files = ro.find_ollama_model_files() assert len(files) == 1 assert files[0]["filename"] == "sha256-ccc333" def test_find_ollama_model_files_skips_layers_whose_blob_is_missing(ollama_tree): """A partially pulled model leaves a manifest with no blob; returning that path would make every downstream residency check fail.""" _write_manifest(ollama_tree, "registry.ollama.ai/library/ghost", "latest", [ {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:missing", "size": 10}, ]) assert ro.find_ollama_model_files() == [] def test_find_ollama_model_files_tolerates_corrupt_manifest_json(ollama_tree): """One unreadable manifest must not take out discovery of every other model.""" bad_dir = os.path.join(ollama_tree, "manifests", "registry.ollama.ai/library/broken") os.makedirs(bad_dir) with open(os.path.join(bad_dir, "latest"), "w") as f: f.write("{not json") _write_blob(ollama_tree, "sha256:eee555", 4096) _write_manifest(ollama_tree, "registry.ollama.ai/library/good", "latest", [ {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:eee555", "size": 10}, ]) assert [f["model"] for f in ro.find_ollama_model_files()] == ["good:latest"] def test_find_ollama_model_files_deduplicates_same_model_and_blob(ollama_tree, monkeypatch): """The same root listed twice (or a duplicated layer) must not double-count bytes against the warm budget.""" _write_blob(ollama_tree, "sha256:fff666", 4096) _write_manifest(ollama_tree, "registry.ollama.ai/library/dup", "latest", [ {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:fff666", "size": 10}, {"mediaType": MODEL_MEDIA_TYPE, "digest": "sha256:fff666", "size": 10}, ]) monkeypatch.setattr(ro, "OLLAMA_MODEL_DIRS", [ollama_tree, ollama_tree]) assert len(ro.find_ollama_model_files()) == 1 def test_find_ollama_model_files_returns_empty_when_no_manifest_dir(tmp_path, monkeypatch): monkeypatch.setattr(ro, "OLLAMA_MODEL_DIRS", [str(tmp_path / "does-not-exist")]) assert ro.find_ollama_model_files() == [] # --------------------------------------------------------------- warm planning GIB = 1024 ** 3 @pytest.fixture def planner(monkeypatch): """build_warm_plan with every IO boundary stubbed: catalog, residency, meminfo, scores.""" state = {"residency": {}, "scores": {}} def _catalog(force_refresh=False): return state["catalog"] monkeypatch.setattr(ro, "get_model_catalog", _catalog) monkeypatch.setattr(ro, "_warm_priority", lambda days=30.0: state["scores"]) monkeypatch.setattr(ro, "get_detailed_meminfo", lambda: {"available_bytes": 40 * GIB, "available_gb": 40.0}) monkeypatch.setattr(ro, "page_residency", lambda path, allow_probe=True, probe_windows=None: state["residency"].get(path, {"resident_pct": 0.0, "warm_confident": False})) return state def _ollama_entry(name, path, gb): return {"model": name, "full_path": path, "size_bytes": int(gb * GIB), "size_gb": gb, "kind": "ollama"} def _comfy_entry(rel, path, gb, mtime): return {"rel_path": rel, "full_path": path, "size_bytes": int(gb * GIB), "size_gb": gb, "kind": "comfy", "mtime": mtime} def test_warm_plan_orders_by_usage_score(planner): """The whole point of the plan is that the most-used model is warmed first, because the budget usually cannot cover everything.""" planner["catalog"] = { "ollama": [_ollama_entry("rare:latest", "/m/rare", 1.0), _ollama_entry("hot:latest", "/m/hot", 1.0)], "comfy": [], } planner["scores"] = {"hot:latest": 50.0, "rare:latest": 0.1} plan = ro.build_warm_plan(budget_gb=10.0) assert [e["name"] for e in plan["plan"]] == ["hot:latest", "rare:latest"] def test_warm_plan_skips_confidently_warm_files_without_spending_budget(planner): """Already-resident files cost nothing and must not consume budget another file needs.""" planner["catalog"] = { "ollama": [_ollama_entry("warm:latest", "/m/warm", 8.0), _ollama_entry("cold:latest", "/m/cold", 8.0)], "comfy": [], } planner["scores"] = {"warm:latest": 10.0, "cold:latest": 5.0} planner["residency"]["/m/warm"] = {"resident_pct": 100.0, "warm_confident": True} plan = ro.build_warm_plan(budget_gb=10.0) assert [e["name"] for e in plan["plan"]] == ["cold:latest"] assert [e["action"] for e in plan["skipped"]] == ["already-warm"] assert plan["planned_gb"] == pytest.approx(8.0, abs=0.01) def test_warm_plan_does_not_skip_a_high_but_unconfident_residency(planner): """95% from a probe is not permission to skip — this is the bug the warm_confident flag exists to prevent.""" planner["catalog"] = {"ollama": [_ollama_entry("m:latest", "/m/x", 4.0)], "comfy": []} planner["residency"]["/m/x"] = {"resident_pct": 95.0, "warm_confident": False} plan = ro.build_warm_plan(budget_gb=10.0) assert [e["name"] for e in plan["plan"]] == ["m:latest"] # Only the missing 5% has to be read. assert plan["plan"][0]["bytes_to_read"] == pytest.approx(int(4.0 * GIB) * 0.05, rel=0.01) def test_warm_plan_charges_only_the_non_resident_fraction(planner): planner["catalog"] = {"ollama": [_ollama_entry("m:latest", "/m/x", 10.0)], "comfy": []} planner["residency"]["/m/x"] = {"resident_pct": 50.0, "warm_confident": False} plan = ro.build_warm_plan(budget_gb=6.0) assert plan["planned_gb"] == pytest.approx(5.0, abs=0.01) def test_warm_plan_stops_at_the_budget(planner): """Warming past the budget just evicts what was warmed first, so over-budget entries are reported as skipped rather than planned.""" planner["catalog"] = { "ollama": [_ollama_entry("a", "/m/a", 6.0), _ollama_entry("b", "/m/b", 6.0)], "comfy": [], } planner["scores"] = {"a": 9.0, "b": 1.0} plan = ro.build_warm_plan(budget_gb=8.0) assert [e["name"] for e in plan["plan"]] == ["a"] assert plan["skipped"][0]["name"] == "b" assert plan["skipped"][0]["action"] == "over-budget" assert plan["planned_gb"] <= plan["budget_gb"] def test_warm_plan_deduplicates_by_path(planner): """The same file reachable from both catalogs must be planned once, or the budget is charged twice for one read.""" planner["catalog"] = { "ollama": [_ollama_entry("shared", "/m/shared", 2.0)], "comfy": [_comfy_entry("shared.safetensors", "/m/shared", 2.0, 0)], } plan = ro.build_warm_plan(budget_gb=100.0) assert len(plan["plan"]) == 1 assert plan["planned_gb"] == pytest.approx(2.0, abs=0.01) def test_warm_plan_default_budget_leaves_headroom(planner): """An unbounded budget would push the box into reclaim; the default is 70% of MemAvailable.""" planner["catalog"] = {"ollama": [], "comfy": []} plan = ro.build_warm_plan() assert plan["budget_gb"] == pytest.approx(40.0 * 0.7, abs=0.01) def test_warm_plan_ranks_recent_comfy_checkpoints_above_stale_ones(planner): """ComfyUI files have no usage history, so recency by mtime is the ranking signal.""" import time now = time.time() planner["catalog"] = { "ollama": [], "comfy": [_comfy_entry("old.safetensors", "/c/old", 1.0, now - 90 * 86400), _comfy_entry("new.safetensors", "/c/new", 1.0, now - 60)], } plan = ro.build_warm_plan(budget_gb=10.0) assert [e["name"] for e in plan["plan"]] == ["new.safetensors", "old.safetensors"] def test_warm_plan_on_empty_catalog_is_a_valid_empty_plan(planner): planner["catalog"] = {"ollama": [], "comfy": []} plan = ro.build_warm_plan(budget_gb=1.0) assert plan["warm_count"] == 0 and plan["plan"] == [] and plan["skipped"] == []