Fix two more stale-cache bugs; account for VRAM this service cannot reclaim

The stale-readback bug fixed in f5917a0 was a class, not an instance. Two more:

- ram_optimizer's residency report cached for 15s and was never invalidated when
  anything warmed a file, so warming a model and then looking at residency showed
  the state from before the warm. warm_file_to_ram now invalidates it.
- _PID_KIND_CACHE was keyed on pid alone and never expired. Linux recycles PIDs, so
  a stale entry could attribute a new process's VRAM to Ollama or ComfyUI -- inside
  the very snapshot the yield barrier trusts to decide whether VRAM was released.
  Now keyed by (pid, process start time) and bounded.

Unmanaged VRAM. Investigating a persistence-mode warning turned up a third GPU
consumer this service does not model: stt_relay.py, holding 842 MB for nearly three
days. It was bucketed as "system" alongside gnome-shell's 3.9 MB. That conflation
matters, because ComfyUI's memory can be reclaimed and a third party's cannot, and
the reclaim path assumed ComfyUI was always to blame for missing headroom.

Processes are now bucketed ollama | comfy | desktop | unmanaged. The breakdown
reports desktop_gb and unmanaged_gb separately and names the unmanaged processes;
when a reclaim-and-retry still fails, the error identifies them rather than
implying ComfyUI was at fault; and the dashboard shows the unreclaimable total, so
headroom the arbitrator can never give back is visible rather than inferred.

Checked and deliberately not changed: persistence mode reads Disabled, but
nvidia-persistenced is active and two clients hold the GPU open continuously, so
the driver never unloads. The nvidia-smi warning is legacy noise here and is not a
source of the profile drift.

Tests: 169 (was 164). The new ones cover the bucketing, and one existing test used
Xorg as its "unknown process" fixture -- correct before a display server had its own
bucket, wrong after.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-05 19:11:40 -07:00
parent f5917a0464
commit 25b601e24a
5 changed files with 171 additions and 16 deletions

View File

@@ -417,6 +417,16 @@ _report_cache: Dict[str, Any] = {"ts": 0.0, "report": None}
REPORT_TTL_S = 15.0 REPORT_TTL_S = 15.0
def invalidate_cache_report() -> None:
"""Drop the cached residency report.
Anything that changes what is resident must call this, or the report keeps serving
pre-change numbers for up to REPORT_TTL_S -- so warming a model and then looking at
residency showed the state from before the warm.
"""
_report_cache["report"] = None
def get_cache_report(include_files: bool = True, force_refresh: bool = False) -> Dict[str, Any]: def get_cache_report(include_files: bool = True, force_refresh: bool = False) -> Dict[str, Any]:
"""Measured page-cache residency across the whole model catalog. """Measured page-cache residency across the whole model catalog.
@@ -512,6 +522,7 @@ def warm_file_to_ram(filepath: str, chunk_size: int = 16 * 1024 * 1024,
duration = time.perf_counter() - t0 duration = time.perf_counter() - t0
after = page_residency(filepath, probe_windows=32) after = page_residency(filepath, probe_windows=32)
invalidate_cache_report()
return { return {
"success": True, "success": True,
"filepath": filepath, "filepath": filepath,

View File

@@ -35,7 +35,7 @@ function updateDashboard(data) {
// Governor and arbitration state ride along in the shared snapshot. // Governor and arbitration state ride along in the shared snapshot.
if (data.governor) renderGovernor(data.governor); if (data.governor) renderGovernor(data.governor);
if (data.arbitrator) renderArbitrator(data.arbitrator); if (data.arbitrator) renderArbitrator(data.arbitrator, data.gpu);
// 1. GPU VRAM Stats // 1. GPU VRAM Stats
const gpu = data.gpu || {}; const gpu = data.gpu || {};
@@ -886,11 +886,23 @@ document.addEventListener('DOMContentLoaded', () => {
// ---------------------------------------------------------------- arbitration // ---------------------------------------------------------------- arbitration
function renderArbitrator(arb) { function renderArbitrator(arb, gpu) {
const el = (id) => document.getElementById(id); const el = (id) => document.getElementById(id);
if (!el('arb-action')) return; if (!el('arb-action')) return;
const c = arb.counters || {}; const c = arb.counters || {};
// VRAM held by processes HyperSwap cannot reclaim. Worth showing: it is headroom the
// arbitrator can never give back, no matter how much it purges.
const bd = (gpu && gpu.breakdown) || {};
const un = el('arb-unmanaged');
if (un) {
const procs = bd.unmanaged || [];
un.innerHTML = procs.length
? `<span class="text-amber-400">${bd.unmanaged_gb} GB unreclaimable</span> — ` +
procs.map(p => `${p.name} (${p.vram_mb} MB)`).join(', ')
: '';
}
el('arb-action').textContent = arb.last_action || 'Idle'; el('arb-action').textContent = arb.last_action || 'Idle';
el('arb-yields').textContent = c.yields ?? 0; el('arb-yields').textContent = c.yields ?? 0;
el('arb-busy').textContent = c.yield_deferred_busy ?? 0; el('arb-busy').textContent = c.yield_deferred_busy ?? 0;

View File

@@ -686,6 +686,7 @@
<div class="mt-4"> <div class="mt-4">
<div id="arb-action" class="text-sm text-slate-200 bg-slate-950/60 border border-slate-800 rounded-lg px-3 py-2 mb-3 font-mono">Idle</div> <div id="arb-action" class="text-sm text-slate-200 bg-slate-950/60 border border-slate-800 rounded-lg px-3 py-2 mb-3 font-mono">Idle</div>
<div id="arb-backoff" class="text-[11px] font-mono text-amber-400 mb-3"></div> <div id="arb-backoff" class="text-[11px] font-mono text-amber-400 mb-3"></div>
<div id="arb-unmanaged" class="text-[11px] font-mono text-slate-400 mb-3"></div>
<div class="grid grid-cols-3 sm:grid-cols-6 gap-2 text-center"> <div class="grid grid-cols-3 sm:grid-cols-6 gap-2 text-center">
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800"> <div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
<div id="arb-yields" class="text-lg font-bold text-emerald-400">0</div> <div id="arb-yields" class="text-lg font-bold text-emerald-400">0</div>

View File

@@ -68,15 +68,72 @@ def test_classify_pid_detects_comfyui(monkeypatch):
assert va._classify_pid(1234) == "comfy" assert va._classify_pid(1234) == "comfy"
def test_classify_pid_unknown_process_is_other(monkeypatch): def test_classify_pid_unknown_process_is_unmanaged(monkeypatch):
# Xorg used to stand in for "unknown" here, but a display server is now its own
# bucket, so this needs a process that is genuinely neither ours nor the desktop's.
_patch_proc(monkeypatch, _FakeProc("trainer", ["/opt/ml/bin/trainer", "--epochs", "3"]))
# "unmanaged" rather than "other": a third-party GPU workload holds VRAM this
# service cannot reclaim, and must not be lumped in with the desktop compositor.
assert va._classify_pid(1234) == "unmanaged"
def test_classify_pid_display_server_is_desktop(monkeypatch):
_patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"])) _patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"]))
assert va._classify_pid(1234) == "other" assert va._classify_pid(1234) == "desktop"
def test_classify_pid_returns_other_when_process_vanished(monkeypatch): def test_classify_pid_returns_unmanaged_when_process_vanished(monkeypatch):
"""PIDs are read from NVML and can exit before psutil looks them up; that is normal """PIDs are read from NVML and can exit before psutil looks them up; that is normal
and must not raise inside the 20 ms VRAM poll loop.""" and must not raise inside the 20 ms VRAM poll loop."""
def _boom(pid): def _boom(pid):
raise va.psutil.NoSuchProcess(pid) raise va.psutil.NoSuchProcess(pid)
monkeypatch.setattr(va.psutil, "Process", _boom) monkeypatch.setattr(va.psutil, "Process", _boom)
assert va._classify_pid(999999) == "other" assert va._classify_pid(999999) == "unmanaged"
# --- process bucketing: desktop vs unmanaged ---------------------------------
#
# Real case from this machine: stt_relay.py held 842 MB of VRAM for nearly three days
# while gnome-shell held 3.9 MB. A single "other" bucket reported them as one number,
# which matters because ComfyUI's memory can be reclaimed and a third party's cannot.
class _FakeProc:
def __init__(self, name, cmdline):
self._name = name
self._cmdline = cmdline
def name(self):
return self._name
def cmdline(self):
return self._cmdline
def create_time(self):
return 1234.5
def _classify(monkeypatch, name, cmdline):
monkeypatch.setattr(va.psutil, "Process",
lambda pid: _FakeProc(name, cmdline))
return va._classify_pid(4321)
def test_desktop_compositors_are_their_own_bucket(monkeypatch):
assert _classify(monkeypatch, "gnome-shell", ["/usr/bin/gnome-shell"]) == "desktop"
assert _classify(monkeypatch, "Xorg", ["/usr/lib/xorg/Xorg", ":8"]) == "desktop"
def test_third_party_compute_is_unmanaged_not_desktop(monkeypatch):
kind = _classify(monkeypatch, "python",
["/home/u/robopest-venv/bin/python", "/home/u/stt_relay.py"])
assert kind == "unmanaged"
def test_ollama_and_comfy_still_win_over_the_catch_all(monkeypatch):
assert _classify(monkeypatch, "llama-server",
["/usr/local/lib/ollama/llama-server", "--model", "x"]) == "ollama"
assert _classify(monkeypatch, "python",
["/home/u/ComfyUI/venv/bin/python", "main.py"]) == "comfy"
def test_pid_cache_is_keyed_by_start_time_not_pid_alone():
# Linux recycles PIDs; a stale entry would attribute a new process's VRAM to Ollama
# inside the same snapshot the yield barrier trusts.
assert all(isinstance(k, tuple) and len(k) == 2 for k in va._PID_KIND_CACHE)

View File

@@ -77,6 +77,20 @@ OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
"failed to allocate", "cuda error") "failed to allocate", "cuda error")
def describe_unmanaged() -> Dict[str, Any]:
"""VRAM held by processes this service cannot reclaim, named explicitly."""
stats = get_gpu_hardware_stats()
bd = stats.get("breakdown", {}) if stats.get("available") else {}
entries = bd.get("unmanaged", [])
return {
"unmanaged_gb": bd.get("unmanaged_gb", 0.0),
"processes": entries,
"note": ("VRAM held by processes outside HyperSwap's control; it cannot be "
"reclaimed automatically" if entries else
"no third-party GPU processes are holding VRAM"),
}
def looks_like_vram_oom(text: str) -> bool: def looks_like_vram_oom(text: str) -> bool:
low = (text or "").lower() low = (text or "").lower()
return any(sig in low for sig in OOM_SIGNATURES) return any(sig in low for sig in OOM_SIGNATURES)
@@ -135,7 +149,7 @@ def get_process_vram_bytes() -> Dict[str, int]:
to actually drain. to actually drain.
""" """
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0, out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0,
"gpu_util_pct": 0} "desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0}
if not NVML_AVAILABLE: if not NVML_AVAILABLE:
return out return out
try: try:
@@ -154,36 +168,72 @@ def get_process_vram_bytes() -> Dict[str, int]:
for p in procs: for p in procs:
merged[p.pid] = max(merged.get(p.pid, 0), p.usedGpuMemory or 0) merged[p.pid] = max(merged.get(p.pid, 0), p.usedGpuMemory or 0)
for pid, used in merged.items(): for pid, used in merged.items():
kind = _PID_KIND_CACHE.get(pid) key = _pid_key(pid)
kind = _PID_KIND_CACHE.get(key) if key else None
if kind is None: if kind is None:
kind = _classify_pid(pid) kind = _classify_pid(pid)
_PID_KIND_CACHE[pid] = kind if key:
if len(_PID_KIND_CACHE) >= _PID_KIND_CACHE_MAX:
_PID_KIND_CACHE.clear()
_PID_KIND_CACHE[key] = kind
if kind == "ollama": if kind == "ollama":
out["ollama_bytes"] += used out["ollama_bytes"] += used
elif kind == "comfy": elif kind == "comfy":
out["comfyui_bytes"] += used out["comfyui_bytes"] += used
elif kind == "desktop":
out["desktop_bytes"] += used
out["other_bytes"] += used
else: else:
out["unmanaged_bytes"] += used
out["other_bytes"] += used out["other_bytes"] += used
except Exception as e: except Exception as e:
logger.debug(f"get_process_vram_bytes failed: {e}") logger.debug(f"get_process_vram_bytes failed: {e}")
return out return out
_PID_KIND_CACHE: Dict[int, str] = {} # Keyed by (pid, process start time) rather than pid alone. Linux recycles PIDs, and a
# stale entry would attribute a new process's VRAM to Ollama or ComfyUI -- in the same
# snapshot the yield barrier uses to decide whether VRAM was released.
_PID_KIND_CACHE: Dict[tuple, str] = {}
_PID_KIND_CACHE_MAX = 512
def _pid_key(pid: int) -> Optional[tuple]:
try:
return (pid, psutil.Process(pid).create_time())
except Exception:
return None
# Compositors and display servers. Their VRAM is small, permanent and not ours to
# reclaim, so it should not be confused with a real workload.
DESKTOP_PROCESS_HINTS = (
"gnome-shell", "xorg", "gnome-remote-desktop", "mutter", "kwin", "plasmashell",
"gnome-session", "wayland", "weston", "sddm", "gdm", "picom", "compiz",
)
def _classify_pid(pid: int) -> str: def _classify_pid(pid: int) -> str:
"""Bucket a GPU process into ollama | comfy | desktop | unmanaged.
The old version had one catch-all "other" bucket, which put a 3.9 MB compositor and
an 842 MB long-running inference script in the same number. That matters: this
service can reclaim VRAM from ComfyUI, but it cannot touch a third-party workload,
and pretending otherwise makes it promise headroom it cannot deliver.
"""
try: try:
proc = psutil.Process(pid) proc = psutil.Process(pid)
pname = proc.name().lower() pname = proc.name().lower()
cmdline = " ".join(proc.cmdline()).lower() cmdline = " ".join(proc.cmdline()).lower()
except Exception: except Exception:
return "other" return "unmanaged"
if "ollama" in pname or "llama-server" in cmdline: if "ollama" in pname or "llama-server" in cmdline:
return "ollama" return "ollama"
if "comfy" in cmdline or "main.py" in cmdline: if "comfyui" in cmdline or "comfy" in cmdline or cmdline.rstrip().endswith("main.py"):
return "comfy" return "comfy"
return "other" if any(hint in pname or hint in cmdline for hint in DESKTOP_PROCESS_HINTS):
return "desktop"
return "unmanaged"
def get_gpu_hardware_stats() -> Dict[str, Any]: def get_gpu_hardware_stats() -> Dict[str, Any]:
@@ -283,6 +333,9 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"ollama_bytes": 0, "ollama_bytes": 0,
"comfyui_bytes": 0, "comfyui_bytes": 0,
"system_bytes": 0, "system_bytes": 0,
"desktop_bytes": 0,
"unmanaged_bytes": 0,
"unmanaged": [],
"processes": [] "processes": []
} }
@@ -303,8 +356,9 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
except Exception: except Exception:
pass pass
is_ollama = "ollama" in pname.lower() or "llama-server" in cmdline.lower() kind = _classify_pid(pid)
is_comfy = "comfy" in cmdline.lower() or "main.py" in cmdline.lower() is_ollama = kind == "ollama"
is_comfy = kind == "comfy"
if is_ollama: if is_ollama:
proc_breakdown["ollama_bytes"] += used_mem proc_breakdown["ollama_bytes"] += used_mem
@@ -312,6 +366,15 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
proc_breakdown["comfyui_bytes"] += used_mem proc_breakdown["comfyui_bytes"] += used_mem
else: else:
proc_breakdown["system_bytes"] += used_mem proc_breakdown["system_bytes"] += used_mem
if kind == "desktop":
proc_breakdown["desktop_bytes"] += used_mem
else:
proc_breakdown["unmanaged_bytes"] += used_mem
proc_breakdown["unmanaged"].append({
"pid": pid, "name": pname,
"cmdline": cmdline[:120],
"vram_mb": round(used_mem / (1024**2), 1),
})
proc_breakdown["processes"].append({ proc_breakdown["processes"].append({
"pid": pid, "pid": pid,
@@ -321,6 +384,7 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"vram_mb": round(used_mem / (1024**2), 1), "vram_mb": round(used_mem / (1024**2), 1),
"is_ollama": is_ollama, "is_ollama": is_ollama,
"is_comfy": is_comfy, "is_comfy": is_comfy,
"kind": kind,
}) })
except Exception as e: except Exception as e:
logger.error(f"Error enumerating GPU processes: {e}") logger.error(f"Error enumerating GPU processes: {e}")
@@ -362,6 +426,11 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"comfyui_gb": round(proc_breakdown["comfyui_bytes"] / (1024**3), 2), "comfyui_gb": round(proc_breakdown["comfyui_bytes"] / (1024**3), 2),
"system_mb": round(proc_breakdown["system_bytes"] / (1024**2), 1), "system_mb": round(proc_breakdown["system_bytes"] / (1024**2), 1),
"system_gb": round(proc_breakdown["system_bytes"] / (1024**3), 2), "system_gb": round(proc_breakdown["system_bytes"] / (1024**3), 2),
"desktop_gb": round(proc_breakdown["desktop_bytes"] / (1024**3), 2),
# VRAM held by workloads this service has no control over. It cannot be
# reclaimed, so it is permanently unavailable headroom.
"unmanaged_gb": round(proc_breakdown["unmanaged_bytes"] / (1024**3), 2),
"unmanaged": proc_breakdown["unmanaged"],
"free_mb": round(free_vram / (1024**2), 1), "free_mb": round(free_vram / (1024**2), 1),
"free_gb": round(free_vram / (1024**3), 2), "free_gb": round(free_vram / (1024**3), 2),
"processes": proc_breakdown["processes"], "processes": proc_breakdown["processes"],
@@ -803,6 +872,11 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
retry["reclaimed_from_comfyui_gb"] = round( retry["reclaimed_from_comfyui_gb"] = round(
snap["comfyui_bytes"] / (1024**3), 2) snap["comfyui_bytes"] / (1024**3), 2)
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM" retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
if not retry.get("success"):
# Be specific about why the reclaim was not enough. Blaming ComfyUI
# when a third-party process is holding the memory sends the user
# looking in the wrong place.
retry["unmanaged_blockers"] = describe_unmanaged()
return retry return retry
return {"success": False, "error": f"HTTP {resp.status_code}: {body}", return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
"duration_ms": total_duration_ms, "duration_ms": total_duration_ms,