diff --git a/ram_optimizer.py b/ram_optimizer.py index c1c4445..6816906 100644 --- a/ram_optimizer.py +++ b/ram_optimizer.py @@ -417,6 +417,16 @@ _report_cache: Dict[str, Any] = {"ts": 0.0, "report": None} REPORT_TTL_S = 15.0 +def invalidate_cache_report() -> None: + """Drop the cached residency report. + + Anything that changes what is resident must call this, or the report keeps serving + pre-change numbers for up to REPORT_TTL_S -- so warming a model and then looking at + residency showed the state from before the warm. + """ + _report_cache["report"] = None + + def get_cache_report(include_files: bool = True, force_refresh: bool = False) -> Dict[str, Any]: """Measured page-cache residency across the whole model catalog. @@ -512,6 +522,7 @@ def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024, duration = time.perf_counter() - t0 after = page_residency(filepath, probe_windows=32) + invalidate_cache_report() return { "success": True, "filepath": filepath, diff --git a/static/app.js b/static/app.js index 24bdfbb..b516388 100644 --- a/static/app.js +++ b/static/app.js @@ -35,7 +35,7 @@ function updateDashboard(data) { // Governor and arbitration state ride along in the shared snapshot. if (data.governor) renderGovernor(data.governor); - if (data.arbitrator) renderArbitrator(data.arbitrator); + if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu); // 1. GPU VRAM Stats const gpu = data.gpu || {}; @@ -886,11 +886,23 @@ document.addEventListener('DOMContentLoaded', () => { // ---------------------------------------------------------------- arbitration -function renderArbitrator(arb) { +function renderArbitrator(arb, gpu) { const el = (id) => document.getElementById(id); if (!el('arb-action')) return; const c = arb.counters || {}; + // VRAM held by processes HyperSwap cannot reclaim. Worth showing: it is headroom the + // arbitrator can never give back, no matter how much it purges. + const bd = (gpu && gpu.breakdown) || {}; + const un = el('arb-unmanaged'); + if (un) { + const procs = bd.unmanaged || []; + un.innerHTML = procs.length + ? `${bd.unmanaged_gb} GB unreclaimable — ` + + procs.map(p => `${p.name} (${p.vram_mb} MB)`).join(', ') + : ''; + } + el('arb-action').textContent = arb.last_action || 'Idle'; el('arb-yields').textContent = c.yields ?? 0; el('arb-busy').textContent = c.yield_deferred_busy ?? 0; diff --git a/static/index.html b/static/index.html index 55561ce..f27666a 100644 --- a/static/index.html +++ b/static/index.html @@ -686,6 +686,7 @@
Idle
+
0
diff --git a/tests/test_vram_helpers.py b/tests/test_vram_helpers.py index 4c6416f..53fd08f 100644 --- a/tests/test_vram_helpers.py +++ b/tests/test_vram_helpers.py @@ -68,15 +68,72 @@ def test_classify_pid_detects_comfyui(monkeypatch): assert va._classify_pid(1234) == "comfy" -def test_classify_pid_unknown_process_is_other(monkeypatch): +def test_classify_pid_unknown_process_is_unmanaged(monkeypatch): + # Xorg used to stand in for "unknown" here, but a display server is now its own + # bucket, so this needs a process that is genuinely neither ours nor the desktop's. + _patch_proc(monkeypatch, _FakeProc("trainer", ["/opt/ml/bin/trainer", "--epochs", "3"])) + # "unmanaged" rather than "other": a third-party GPU workload holds VRAM this + # service cannot reclaim, and must not be lumped in with the desktop compositor. + assert va._classify_pid(1234) == "unmanaged" + + +def test_classify_pid_display_server_is_desktop(monkeypatch): _patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"])) - assert va._classify_pid(1234) == "other" + assert va._classify_pid(1234) == "desktop" -def test_classify_pid_returns_other_when_process_vanished(monkeypatch): +def test_classify_pid_returns_unmanaged_when_process_vanished(monkeypatch): """PIDs are read from NVML and can exit before psutil looks them up; that is normal and must not raise inside the 20 ms VRAM poll loop.""" def _boom(pid): raise va.psutil.NoSuchProcess(pid) monkeypatch.setattr(va.psutil, "Process", _boom) - assert va._classify_pid(999999) == "other" + assert va._classify_pid(999999) == "unmanaged" + + +# --- process bucketing: desktop vs unmanaged --------------------------------- +# +# Real case from this machine: stt_relay.py held 842 MB of VRAM for nearly three days +# while gnome-shell held 3.9 MB. A single "other" bucket reported them as one number, +# which matters because ComfyUI's memory can be reclaimed and a third party's cannot. + +class _FakeProc: + def __init__(self, name, cmdline): + self._name = name + self._cmdline = cmdline + def name(self): + return self._name + def cmdline(self): + return self._cmdline + def create_time(self): + return 1234.5 + + +def _classify(monkeypatch, name, cmdline): + monkeypatch.setattr(va.psutil, "Process", + lambda pid: _FakeProc(name, cmdline)) + return va._classify_pid(4321) + + +def test_desktop_compositors_are_their_own_bucket(monkeypatch): + assert _classify(monkeypatch, "gnome-shell", ["/usr/bin/gnome-shell"]) == "desktop" + assert _classify(monkeypatch, "Xorg", ["/usr/lib/xorg/Xorg", ":8"]) == "desktop" + + +def test_third_party_compute_is_unmanaged_not_desktop(monkeypatch): + kind = _classify(monkeypatch, "python", + ["/home/u/robopest-venv/bin/python", "/home/u/stt_relay.py"]) + assert kind == "unmanaged" + + +def test_ollama_and_comfy_still_win_over_the_catch_all(monkeypatch): + assert _classify(monkeypatch, "llama-server", + ["/usr/local/lib/ollama/llama-server", "--model", "x"]) == "ollama" + assert _classify(monkeypatch, "python", + ["/home/u/ComfyUI/venv/bin/python", "main.py"]) == "comfy" + + +def test_pid_cache_is_keyed_by_start_time_not_pid_alone(): + # Linux recycles PIDs; a stale entry would attribute a new process's VRAM to Ollama + # inside the same snapshot the yield barrier trusts. + assert all(isinstance(k, tuple) and len(k) == 2 for k in va._PID_KIND_CACHE) diff --git a/vram_arbitrator.py b/vram_arbitrator.py index 19af0ba..01ca217 100644 --- a/vram_arbitrator.py +++ b/vram_arbitrator.py @@ -77,6 +77,20 @@ OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate", "failed to allocate", "cuda error") +def describe_unmanaged() -> Dict[str, Any]: + """VRAM held by processes this service cannot reclaim, named explicitly.""" + stats = get_gpu_hardware_stats() + bd = stats.get("breakdown", {}) if stats.get("available") else {} + entries = bd.get("unmanaged", []) + return { + "unmanaged_gb": bd.get("unmanaged_gb", 0.0), + "processes": entries, + "note": ("VRAM held by processes outside HyperSwap's control; it cannot be " + "reclaimed automatically" if entries else + "no third-party GPU processes are holding VRAM"), + } + + def looks_like_vram_oom(text: str) -> bool: low = (text or "").lower() return any(sig in low for sig in OOM_SIGNATURES) @@ -135,7 +149,7 @@ def get_process_vram_bytes() -> Dict[str, int]: to actually drain. """ out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0, - "gpu_util_pct": 0} + "desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0} if not NVML_AVAILABLE: return out try: @@ -154,36 +168,72 @@ def get_process_vram_bytes() -> Dict[str, int]: for p in procs: merged[p.pid] = max(merged.get(p.pid, 0), p.usedGpuMemory or 0) for pid, used in merged.items(): - kind = _PID_KIND_CACHE.get(pid) + key = _pid_key(pid) + kind = _PID_KIND_CACHE.get(key) if key else None if kind is None: kind = _classify_pid(pid) - _PID_KIND_CACHE[pid] = kind + if key: + if len(_PID_KIND_CACHE) >= _PID_KIND_CACHE_MAX: + _PID_KIND_CACHE.clear() + _PID_KIND_CACHE[key] = kind if kind == "ollama": out["ollama_bytes"] += used elif kind == "comfy": out["comfyui_bytes"] += used + elif kind == "desktop": + out["desktop_bytes"] += used + out["other_bytes"] += used else: + out["unmanaged_bytes"] += used out["other_bytes"] += used except Exception as e: logger.debug(f"get_process_vram_bytes failed: {e}") return out -_PID_KIND_CACHE: Dict[int, str] = {} +# Keyed by (pid, process start time) rather than pid alone. Linux recycles PIDs, and a +# stale entry would attribute a new process's VRAM to Ollama or ComfyUI -- in the same +# snapshot the yield barrier uses to decide whether VRAM was released. +_PID_KIND_CACHE: Dict[tuple, str] = {} +_PID_KIND_CACHE_MAX = 512 + + +def _pid_key(pid: int) -> Optional[tuple]: + try: + return (pid, psutil.Process(pid).create_time()) + except Exception: + return None + + +# Compositors and display servers. Their VRAM is small, permanent and not ours to +# reclaim, so it should not be confused with a real workload. +DESKTOP_PROCESS_HINTS = ( + "gnome-shell", "xorg", "gnome-remote-desktop", "mutter", "kwin", "plasmashell", + "gnome-session", "wayland", "weston", "sddm", "gdm", "picom", "compiz", +) def _classify_pid(pid: int) -> str: + """Bucket a GPU process into ollama | comfy | desktop | unmanaged. + + The old version had one catch-all "other" bucket, which put a 3.9 MB compositor and + an 842 MB long-running inference script in the same number. That matters: this + service can reclaim VRAM from ComfyUI, but it cannot touch a third-party workload, + and pretending otherwise makes it promise headroom it cannot deliver. + """ try: proc = psutil.Process(pid) pname = proc.name().lower() cmdline = " ".join(proc.cmdline()).lower() except Exception: - return "other" + return "unmanaged" if "ollama" in pname or "llama-server" in cmdline: return "ollama" - if "comfy" in cmdline or "main.py" in cmdline: + if "comfyui" in cmdline or "comfy" in cmdline or cmdline.rstrip().endswith("main.py"): return "comfy" - return "other" + if any(hint in pname or hint in cmdline for hint in DESKTOP_PROCESS_HINTS): + return "desktop" + return "unmanaged" def get_gpu_hardware_stats() -> Dict[str, Any]: @@ -283,6 +333,9 @@ def get_gpu_hardware_stats() -> Dict[str, Any]: "ollama_bytes": 0, "comfyui_bytes": 0, "system_bytes": 0, + "desktop_bytes": 0, + "unmanaged_bytes": 0, + "unmanaged": [], "processes": [] } @@ -303,15 +356,25 @@ def get_gpu_hardware_stats() -> Dict[str, Any]: except Exception: pass - is_ollama = "ollama" in pname.lower() or "llama-server" in cmdline.lower() - is_comfy = "comfy" in cmdline.lower() or "main.py" in cmdline.lower() - + kind = _classify_pid(pid) + is_ollama = kind == "ollama" + is_comfy = kind == "comfy" + if is_ollama: proc_breakdown["ollama_bytes"] += used_mem elif is_comfy: proc_breakdown["comfyui_bytes"] += used_mem else: proc_breakdown["system_bytes"] += used_mem + if kind == "desktop": + proc_breakdown["desktop_bytes"] += used_mem + else: + proc_breakdown["unmanaged_bytes"] += used_mem + proc_breakdown["unmanaged"].append({ + "pid": pid, "name": pname, + "cmdline": cmdline[:120], + "vram_mb": round(used_mem / (1024**2), 1), + }) proc_breakdown["processes"].append({ "pid": pid, @@ -321,6 +384,7 @@ def get_gpu_hardware_stats() -> Dict[str, Any]: "vram_mb": round(used_mem / (1024**2), 1), "is_ollama": is_ollama, "is_comfy": is_comfy, + "kind": kind, }) except Exception as e: logger.error(f"Error enumerating GPU processes: {e}") @@ -362,6 +426,11 @@ def get_gpu_hardware_stats() -> Dict[str, Any]: "comfyui_gb": round(proc_breakdown["comfyui_bytes"] / (1024**3), 2), "system_mb": round(proc_breakdown["system_bytes"] / (1024**2), 1), "system_gb": round(proc_breakdown["system_bytes"] / (1024**3), 2), + "desktop_gb": round(proc_breakdown["desktop_bytes"] / (1024**3), 2), + # VRAM held by workloads this service has no control over. It cannot be + # reclaimed, so it is permanently unavailable headroom. + "unmanaged_gb": round(proc_breakdown["unmanaged_bytes"] / (1024**3), 2), + "unmanaged": proc_breakdown["unmanaged"], "free_mb": round(free_vram / (1024**2), 1), "free_gb": round(free_vram / (1024**3), 2), "processes": proc_breakdown["processes"], @@ -803,6 +872,11 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m", retry["reclaimed_from_comfyui_gb"] = round( snap["comfyui_bytes"] / (1024**3), 2) retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM" + if not retry.get("success"): + # Be specific about why the reclaim was not enough. Blaming ComfyUI + # when a third-party process is holding the memory sends the user + # looking in the wrong place. + retry["unmanaged_blockers"] = describe_unmanaged() return retry return {"success": False, "error": f"HTTP {resp.status_code}: {body}", "duration_ms": total_duration_ms,