Fix two more stale-cache bugs; account for VRAM this service cannot reclaim
The stale-readback bug fixed in f5917a0 was a class, not an instance. Two more:
- ram_optimizer's residency report cached for 15s and was never invalidated when
anything warmed a file, so warming a model and then looking at residency showed
the state from before the warm. warm_file_to_ram now invalidates it.
- _PID_KIND_CACHE was keyed on pid alone and never expired. Linux recycles PIDs, so
a stale entry could attribute a new process's VRAM to Ollama or ComfyUI -- inside
the very snapshot the yield barrier trusts to decide whether VRAM was released.
Now keyed by (pid, process start time) and bounded.
Unmanaged VRAM. Investigating a persistence-mode warning turned up a third GPU
consumer this service does not model: stt_relay.py, holding 842 MB for nearly three
days. It was bucketed as "system" alongside gnome-shell's 3.9 MB. That conflation
matters, because ComfyUI's memory can be reclaimed and a third party's cannot, and
the reclaim path assumed ComfyUI was always to blame for missing headroom.
Processes are now bucketed ollama | comfy | desktop | unmanaged. The breakdown
reports desktop_gb and unmanaged_gb separately and names the unmanaged processes;
when a reclaim-and-retry still fails, the error identifies them rather than
implying ComfyUI was at fault; and the dashboard shows the unreclaimable total, so
headroom the arbitrator can never give back is visible rather than inferred.
Checked and deliberately not changed: persistence mode reads Disabled, but
nvidia-persistenced is active and two clients hold the GPU open continuously, so
the driver never unloads. The nvidia-smi warning is legacy noise here and is not a
source of the profile drift.
Tests: 169 (was 164). The new ones cover the bucketing, and one existing test used
Xorg as its "unknown process" fixture -- correct before a display server had its own
bucket, wrong after.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -77,6 +77,20 @@ OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
|
||||
"failed to allocate", "cuda error")
|
||||
|
||||
|
||||
def describe_unmanaged() -> Dict[str, Any]:
|
||||
"""VRAM held by processes this service cannot reclaim, named explicitly."""
|
||||
stats = get_gpu_hardware_stats()
|
||||
bd = stats.get("breakdown", {}) if stats.get("available") else {}
|
||||
entries = bd.get("unmanaged", [])
|
||||
return {
|
||||
"unmanaged_gb": bd.get("unmanaged_gb", 0.0),
|
||||
"processes": entries,
|
||||
"note": ("VRAM held by processes outside HyperSwap's control; it cannot be "
|
||||
"reclaimed automatically" if entries else
|
||||
"no third-party GPU processes are holding VRAM"),
|
||||
}
|
||||
|
||||
|
||||
def looks_like_vram_oom(text: str) -> bool:
|
||||
low = (text or "").lower()
|
||||
return any(sig in low for sig in OOM_SIGNATURES)
|
||||
@@ -135,7 +149,7 @@ def get_process_vram_bytes() -> Dict[str, int]:
|
||||
to actually drain.
|
||||
"""
|
||||
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0,
|
||||
"gpu_util_pct": 0}
|
||||
"desktop_bytes": 0, "unmanaged_bytes": 0, "gpu_util_pct": 0}
|
||||
if not NVML_AVAILABLE:
|
||||
return out
|
||||
try:
|
||||
@@ -154,36 +168,72 @@ def get_process_vram_bytes() -> Dict[str, int]:
|
||||
for p in procs:
|
||||
merged[p.pid] = max(merged.get(p.pid, 0), p.usedGpuMemory or 0)
|
||||
for pid, used in merged.items():
|
||||
kind = _PID_KIND_CACHE.get(pid)
|
||||
key = _pid_key(pid)
|
||||
kind = _PID_KIND_CACHE.get(key) if key else None
|
||||
if kind is None:
|
||||
kind = _classify_pid(pid)
|
||||
_PID_KIND_CACHE[pid] = kind
|
||||
if key:
|
||||
if len(_PID_KIND_CACHE) >= _PID_KIND_CACHE_MAX:
|
||||
_PID_KIND_CACHE.clear()
|
||||
_PID_KIND_CACHE[key] = kind
|
||||
if kind == "ollama":
|
||||
out["ollama_bytes"] += used
|
||||
elif kind == "comfy":
|
||||
out["comfyui_bytes"] += used
|
||||
elif kind == "desktop":
|
||||
out["desktop_bytes"] += used
|
||||
out["other_bytes"] += used
|
||||
else:
|
||||
out["unmanaged_bytes"] += used
|
||||
out["other_bytes"] += used
|
||||
except Exception as e:
|
||||
logger.debug(f"get_process_vram_bytes failed: {e}")
|
||||
return out
|
||||
|
||||
|
||||
_PID_KIND_CACHE: Dict[int, str] = {}
|
||||
# Keyed by (pid, process start time) rather than pid alone. Linux recycles PIDs, and a
|
||||
# stale entry would attribute a new process's VRAM to Ollama or ComfyUI -- in the same
|
||||
# snapshot the yield barrier uses to decide whether VRAM was released.
|
||||
_PID_KIND_CACHE: Dict[tuple, str] = {}
|
||||
_PID_KIND_CACHE_MAX = 512
|
||||
|
||||
|
||||
def _pid_key(pid: int) -> Optional[tuple]:
|
||||
try:
|
||||
return (pid, psutil.Process(pid).create_time())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# Compositors and display servers. Their VRAM is small, permanent and not ours to
|
||||
# reclaim, so it should not be confused with a real workload.
|
||||
DESKTOP_PROCESS_HINTS = (
|
||||
"gnome-shell", "xorg", "gnome-remote-desktop", "mutter", "kwin", "plasmashell",
|
||||
"gnome-session", "wayland", "weston", "sddm", "gdm", "picom", "compiz",
|
||||
)
|
||||
|
||||
|
||||
def _classify_pid(pid: int) -> str:
|
||||
"""Bucket a GPU process into ollama | comfy | desktop | unmanaged.
|
||||
|
||||
The old version had one catch-all "other" bucket, which put a 3.9 MB compositor and
|
||||
an 842 MB long-running inference script in the same number. That matters: this
|
||||
service can reclaim VRAM from ComfyUI, but it cannot touch a third-party workload,
|
||||
and pretending otherwise makes it promise headroom it cannot deliver.
|
||||
"""
|
||||
try:
|
||||
proc = psutil.Process(pid)
|
||||
pname = proc.name().lower()
|
||||
cmdline = " ".join(proc.cmdline()).lower()
|
||||
except Exception:
|
||||
return "other"
|
||||
return "unmanaged"
|
||||
if "ollama" in pname or "llama-server" in cmdline:
|
||||
return "ollama"
|
||||
if "comfy" in cmdline or "main.py" in cmdline:
|
||||
if "comfyui" in cmdline or "comfy" in cmdline or cmdline.rstrip().endswith("main.py"):
|
||||
return "comfy"
|
||||
return "other"
|
||||
if any(hint in pname or hint in cmdline for hint in DESKTOP_PROCESS_HINTS):
|
||||
return "desktop"
|
||||
return "unmanaged"
|
||||
|
||||
|
||||
def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
@@ -283,6 +333,9 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
"ollama_bytes": 0,
|
||||
"comfyui_bytes": 0,
|
||||
"system_bytes": 0,
|
||||
"desktop_bytes": 0,
|
||||
"unmanaged_bytes": 0,
|
||||
"unmanaged": [],
|
||||
"processes": []
|
||||
}
|
||||
|
||||
@@ -303,15 +356,25 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
is_ollama = "ollama" in pname.lower() or "llama-server" in cmdline.lower()
|
||||
is_comfy = "comfy" in cmdline.lower() or "main.py" in cmdline.lower()
|
||||
|
||||
kind = _classify_pid(pid)
|
||||
is_ollama = kind == "ollama"
|
||||
is_comfy = kind == "comfy"
|
||||
|
||||
if is_ollama:
|
||||
proc_breakdown["ollama_bytes"] += used_mem
|
||||
elif is_comfy:
|
||||
proc_breakdown["comfyui_bytes"] += used_mem
|
||||
else:
|
||||
proc_breakdown["system_bytes"] += used_mem
|
||||
if kind == "desktop":
|
||||
proc_breakdown["desktop_bytes"] += used_mem
|
||||
else:
|
||||
proc_breakdown["unmanaged_bytes"] += used_mem
|
||||
proc_breakdown["unmanaged"].append({
|
||||
"pid": pid, "name": pname,
|
||||
"cmdline": cmdline[:120],
|
||||
"vram_mb": round(used_mem / (1024**2), 1),
|
||||
})
|
||||
|
||||
proc_breakdown["processes"].append({
|
||||
"pid": pid,
|
||||
@@ -321,6 +384,7 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
"vram_mb": round(used_mem / (1024**2), 1),
|
||||
"is_ollama": is_ollama,
|
||||
"is_comfy": is_comfy,
|
||||
"kind": kind,
|
||||
})
|
||||
except Exception as e:
|
||||
logger.error(f"Error enumerating GPU processes: {e}")
|
||||
@@ -362,6 +426,11 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
"comfyui_gb": round(proc_breakdown["comfyui_bytes"] / (1024**3), 2),
|
||||
"system_mb": round(proc_breakdown["system_bytes"] / (1024**2), 1),
|
||||
"system_gb": round(proc_breakdown["system_bytes"] / (1024**3), 2),
|
||||
"desktop_gb": round(proc_breakdown["desktop_bytes"] / (1024**3), 2),
|
||||
# VRAM held by workloads this service has no control over. It cannot be
|
||||
# reclaimed, so it is permanently unavailable headroom.
|
||||
"unmanaged_gb": round(proc_breakdown["unmanaged_bytes"] / (1024**3), 2),
|
||||
"unmanaged": proc_breakdown["unmanaged"],
|
||||
"free_mb": round(free_vram / (1024**2), 1),
|
||||
"free_gb": round(free_vram / (1024**3), 2),
|
||||
"processes": proc_breakdown["processes"],
|
||||
@@ -803,6 +872,11 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
|
||||
retry["reclaimed_from_comfyui_gb"] = round(
|
||||
snap["comfyui_bytes"] / (1024**3), 2)
|
||||
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
|
||||
if not retry.get("success"):
|
||||
# Be specific about why the reclaim was not enough. Blaming ComfyUI
|
||||
# when a third-party process is holding the memory sends the user
|
||||
# looking in the wrong place.
|
||||
retry["unmanaged_blockers"] = describe_unmanaged()
|
||||
return retry
|
||||
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
|
||||
"duration_ms": total_duration_ms,
|
||||
|
||||
Reference in New Issue
Block a user