0
diff --git a/tests/test_vram_helpers.py b/tests/test_vram_helpers.py
index 4c6416f..53fd08f 100644
--- a/tests/test_vram_helpers.py
+++ b/tests/test_vram_helpers.py
@@ -68,15 +68,72 @@ def test_classify_pid_detects_comfyui(monkeypatch):
assert va._classify_pid(1234) == "comfy"
-def test_classify_pid_unknown_process_is_other(monkeypatch):
+def test_classify_pid_unknown_process_is_unmanaged(monkeypatch):
+ # Xorg used to stand in for "unknown" here, but a display server is now its own
+ # bucket, so this needs a process that is genuinely neither ours nor the desktop's.
+ _patch_proc(monkeypatch, _FakeProc("trainer", ["/opt/ml/bin/trainer", "--epochs", "3"]))
+ # "unmanaged" rather than "other": a third-party GPU workload holds VRAM this
+ # service cannot reclaim, and must not be lumped in with the desktop compositor.
+ assert va._classify_pid(1234) == "unmanaged"
+
+
+def test_classify_pid_display_server_is_desktop(monkeypatch):
_patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"]))
- assert va._classify_pid(1234) == "other"
+ assert va._classify_pid(1234) == "desktop"
-def test_classify_pid_returns_other_when_process_vanished(monkeypatch):
+def test_classify_pid_returns_unmanaged_when_process_vanished(monkeypatch):
"""PIDs are read from NVML and can exit before psutil looks them up; that is normal
and must not raise inside the 20 ms VRAM poll loop."""
def _boom(pid):
raise va.psutil.NoSuchProcess(pid)
monkeypatch.setattr(va.psutil, "Process", _boom)
- assert va._classify_pid(999999) == "other"
+ assert va._classify_pid(999999) == "unmanaged"
+
+
+# --- process bucketing: desktop vs unmanaged ---------------------------------
+#
+# Real case from this machine: stt_relay.py held 842 MB of VRAM for nearly three days
+# while gnome-shell held 3.9 MB. A single "other" bucket reported them as one number,
+# which matters because ComfyUI's memory can be reclaimed and a third party's cannot.
+
+class _FakeProc:
+ def __init__(self, name, cmdline):
+ self._name = name
+ self._cmdline = cmdline
+ def name(self):
+ return self._name
+ def cmdline(self):
+ return self._cmdline
+ def create_time(self):
+ return 1234.5
+
+
+def _classify(monkeypatch, name, cmdline):
+ monkeypatch.setattr(va.psutil, "Process",
+ lambda pid: _FakeProc(name, cmdline))
+ return va._classify_pid(4321)
+
+
+def test_desktop_compositors_are_their_own_bucket(monkeypatch):
+ assert _classify(monkeypatch, "gnome-shell", ["/usr/bin/gnome-shell"]) == "desktop"
+ assert _classify(monkeypatch, "Xorg", ["/usr/lib/xorg/Xorg", ":8"]) == "desktop"
+
+
+def test_third_party_compute_is_unmanaged_not_desktop(monkeypatch):
+ kind = _classify(monkeypatch, "python",
+ ["/home/u/robopest-venv/bin/python", "/home/u/stt_relay.py"])
+ assert kind == "unmanaged"
+
+
+def test_ollama_and_comfy_still_win_over_the_catch_all(monkeypatch):
+ assert _classify(monkeypatch, "llama-server",
+ ["/usr/local/lib/ollama/llama-server", "--model", "x"]) == "ollama"
+ assert _classify(monkeypatch, "python",
+ ["/home/u/ComfyUI/venv/bin/python", "main.py"]) == "comfy"
+
+
+def test_pid_cache_is_keyed_by_start_time_not_pid_alone():
+ # Linux recycles PIDs; a stale entry would attribute a new process's VRAM to Ollama
+ # inside the same snapshot the yield barrier trusts.
+ assert all(isinstance(k, tuple) and len(k) == 2 for k in va._PID_KIND_CACHE)
diff --git a/vram_arbitrator.py b/vram_arbitrator.py
index 19af0ba..01ca217 100644
--- a/vram_arbitrator.py
+++ b/vram_arbitrator.py
@@ -77,6 +77,20 @@ OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
"failed to allocate", "cuda error")
+def describe_unmanaged() -> Dict[str, Any]:
+ """VRAM held by processes this service cannot reclaim, named explicitly."""
+ stats = get_gpu_hardware_stats()
+ bd = stats.get("breakdown", {}) if stats.get("available") else {}
+ entries = bd.get("unmanaged", [])
+ return {
+ "unmanaged_gb": bd.get("unmanaged_gb", 0.0),
+ "processes": entries,
+ "note": ("VRAM held by processes outside HyperSwap's control; it cannot be "
+ "reclaimed automatically" if entries else
+ "no third-party GPU processes are holding VRAM"),
+ }
+
+
def looks_like_vram_oom(text: str) -> bool:
low = (text or "").lower()
return any(sig in low for sig in OOM_SIGNATURES)
@@ -135,7 +149,7 @@ def get_process_vram_bytes() -> Dict[str, int]:
to actually drain.
"""
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0,
- "gpu_util_pct": 0}
+ "desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0}
if not NVML_AVAILABLE:
return out
try:
@@ -154,36 +168,72 @@ def get_process_vram_bytes() -> Dict[str, int]:
for p in procs:
merged[p.pid] = max(merged.get(p.pid, 0), p.usedGpuMemory or 0)
for pid, used in merged.items():
- kind = _PID_KIND_CACHE.get(pid)
+ key = _pid_key(pid)
+ kind = _PID_KIND_CACHE.get(key) if key else None
if kind is None:
kind = _classify_pid(pid)
- _PID_KIND_CACHE[pid] = kind
+ if key:
+ if len(_PID_KIND_CACHE) >= _PID_KIND_CACHE_MAX:
+ _PID_KIND_CACHE.clear()
+ _PID_KIND_CACHE[key] = kind
if kind == "ollama":
out["ollama_bytes"] += used
elif kind == "comfy":
out["comfyui_bytes"] += used
+ elif kind == "desktop":
+ out["desktop_bytes"] += used
+ out["other_bytes"] += used
else:
+ out["unmanaged_bytes"] += used
out["other_bytes"] += used
except Exception as e:
logger.debug(f"get_process_vram_bytes failed: {e}")
return out
-_PID_KIND_CACHE: Dict[int, str] = {}
+# Keyed by (pid, process start time) rather than pid alone. Linux recycles PIDs, and a
+# stale entry would attribute a new process's VRAM to Ollama or ComfyUI -- in the same
+# snapshot the yield barrier uses to decide whether VRAM was released.
+_PID_KIND_CACHE: Dict[tuple, str] = {}
+_PID_KIND_CACHE_MAX = 512
+
+
+def _pid_key(pid: int) -> Optional[tuple]:
+ try:
+ return (pid, psutil.Process(pid).create_time())
+ except Exception:
+ return None
+
+
+# Compositors and display servers. Their VRAM is small, permanent and not ours to
+# reclaim, so it should not be confused with a real workload.
+DESKTOP_PROCESS_HINTS = (
+ "gnome-shell", "xorg", "gnome-remote-desktop", "mutter", "kwin", "plasmashell",
+ "gnome-session", "wayland", "weston", "sddm", "gdm", "picom", "compiz",
+)
def _classify_pid(pid: int) -> str:
+ """Bucket a GPU process into ollama | comfy | desktop | unmanaged.
+
+ The old version had one catch-all "other" bucket, which put a 3.9 MB compositor and
+ an 842 MB long-running inference script in the same number. That matters: this
+ service can reclaim VRAM from ComfyUI, but it cannot touch a third-party workload,
+ and pretending otherwise makes it promise headroom it cannot deliver.
+ """
try:
proc = psutil.Process(pid)
pname = proc.name().lower()
cmdline = " ".join(proc.cmdline()).lower()
except Exception:
- return "other"
+ return "unmanaged"
if "ollama" in pname or "llama-server" in cmdline:
return "ollama"
- if "comfy" in cmdline or "main.py" in cmdline:
+ if "comfyui" in cmdline or "comfy" in cmdline or cmdline.rstrip().endswith("main.py"):
return "comfy"
- return "other"
+ if any(hint in pname or hint in cmdline for hint in DESKTOP_PROCESS_HINTS):
+ return "desktop"
+ return "unmanaged"
def get_gpu_hardware_stats() -> Dict[str, Any]:
@@ -283,6 +333,9 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"ollama_bytes": 0,
"comfyui_bytes": 0,
"system_bytes": 0,
+ "desktop_bytes": 0,
+ "unmanaged_bytes": 0,
+ "unmanaged": [],
"processes": []
}
@@ -303,15 +356,25 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
except Exception:
pass
- is_ollama = "ollama" in pname.lower() or "llama-server" in cmdline.lower()
- is_comfy = "comfy" in cmdline.lower() or "main.py" in cmdline.lower()
-
+ kind = _classify_pid(pid)
+ is_ollama = kind == "ollama"
+ is_comfy = kind == "comfy"
+
if is_ollama:
proc_breakdown["ollama_bytes"] += used_mem
elif is_comfy:
proc_breakdown["comfyui_bytes"] += used_mem
else:
proc_breakdown["system_bytes"] += used_mem
+ if kind == "desktop":
+ proc_breakdown["desktop_bytes"] += used_mem
+ else:
+ proc_breakdown["unmanaged_bytes"] += used_mem
+ proc_breakdown["unmanaged"].append({
+ "pid": pid, "name": pname,
+ "cmdline": cmdline[:120],
+ "vram_mb": round(used_mem / (1024**2), 1),
+ })
proc_breakdown["processes"].append({
"pid": pid,
@@ -321,6 +384,7 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"vram_mb": round(used_mem / (1024**2), 1),
"is_ollama": is_ollama,
"is_comfy": is_comfy,
+ "kind": kind,
})
except Exception as e:
logger.error(f"Error enumerating GPU processes: {e}")
@@ -362,6 +426,11 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
"comfyui_gb": round(proc_breakdown["comfyui_bytes"] / (1024**3), 2),
"system_mb": round(proc_breakdown["system_bytes"] / (1024**2), 1),
"system_gb": round(proc_breakdown["system_bytes"] / (1024**3), 2),
+ "desktop_gb": round(proc_breakdown["desktop_bytes"] / (1024**3), 2),
+ # VRAM held by workloads this service has no control over. It cannot be
+ # reclaimed, so it is permanently unavailable headroom.
+ "unmanaged_gb": round(proc_breakdown["unmanaged_bytes"] / (1024**3), 2),
+ "unmanaged": proc_breakdown["unmanaged"],
"free_mb": round(free_vram / (1024**2), 1),
"free_gb": round(free_vram / (1024**3), 2),
"processes": proc_breakdown["processes"],
@@ -803,6 +872,11 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
retry["reclaimed_from_comfyui_gb"] = round(
snap["comfyui_bytes"] / (1024**3), 2)
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
+ if not retry.get("success"):
+ # Be specific about why the reclaim was not enough. Blaming ComfyUI
+ # when a third-party process is holding the memory sends the user
+ # looking in the wrong place.
+ retry["unmanaged_blockers"] = describe_unmanaged()
return retry
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
"duration_ms": total_duration_ms,