Make unreclaimable VRAM actionable, and account for ComfyUI's CUDA context
The health check now reports what unmanaged VRAM actually costs rather than just how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit within 15.42 GB but not the 14.60 GB actually available", naming them. Getting that arithmetic right took a correction. The first version subtracted only the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB -- the same model the service had just refused with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process lives, which a purge does not free, so it is not available either. The floor is taken from the minimum ComfyUI VRAM in recent telemetry rather than its current value, which could be a 7 GB checkpoint mid-generation. The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the 30 s idle window, since a large model takes longer than that to load and the purge was freeing ComfyUI mid-load, so the reclaim path was never reached. Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model too large to fit on the card at all is not blamed on the third-party process. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
72
health.py
72
health.py
@@ -10,6 +10,7 @@ when it is missing and how to fix it. A degraded dependency should be loud.
|
||||
"""
|
||||
import asyncio
|
||||
import logging
|
||||
import time
|
||||
import os
|
||||
import time
|
||||
from typing import Any, Dict, List
|
||||
@@ -139,6 +140,75 @@ def _check_residency() -> Dict[str, Any]:
|
||||
cap.get("hint", ""))
|
||||
|
||||
|
||||
def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float:
|
||||
"""Lowest VRAM ComfyUI has been observed holding while alive.
|
||||
|
||||
A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred
|
||||
MB for as long as the process runs. The minimum seen in recent telemetry is a better
|
||||
estimate of that floor than whatever it happens to hold right now, which could be a
|
||||
7 GB checkpoint mid-generation.
|
||||
"""
|
||||
try:
|
||||
rows = telemetry_store._rows(
|
||||
"SELECT MIN(comfy_bytes) AS floor FROM telemetry "
|
||||
"WHERE ts > ? AND comfy_bytes > 0",
|
||||
(time.time() - days * 86400,))
|
||||
if rows and rows[0].get("floor"):
|
||||
return round(rows[0]["floor"] / (1024 ** 3), 2)
|
||||
except Exception as e:
|
||||
logger.debug(f"comfy floor lookup failed: {e}")
|
||||
return default
|
||||
|
||||
|
||||
def _check_unmanaged_vram() -> Dict[str, Any]:
|
||||
"""Report unreclaimable VRAM in terms of what it actually costs.
|
||||
|
||||
"0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your
|
||||
models can no longer fit" is something you can act on.
|
||||
"""
|
||||
stats = vram_arbitrator.get_gpu_hardware_stats()
|
||||
if not stats.get("available"):
|
||||
return _check("unmanaged VRAM", DEGRADED, "GPU unavailable")
|
||||
bd = stats.get("breakdown", {})
|
||||
unmanaged_gb = bd.get("unmanaged_gb", 0.0)
|
||||
procs = bd.get("unmanaged", [])
|
||||
if not procs:
|
||||
return _check("unmanaged VRAM", OK, "no third-party GPU processes")
|
||||
|
||||
total_gb = stats.get("vram_total_gb", 0)
|
||||
# What HyperSwap could offer at best. Three things are never available: the desktop,
|
||||
# processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge.
|
||||
# Omitting that last one made this check claim a 14.93 GB model would fit against a
|
||||
# real ceiling of 14.60 GB -- the model that had just returned 507.
|
||||
comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0))
|
||||
ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb
|
||||
try:
|
||||
blobs = ram_optimizer.find_ollama_model_files()
|
||||
except Exception:
|
||||
blobs = []
|
||||
# Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache
|
||||
# are allocated.
|
||||
VRAM_OVERHEAD = 1.16
|
||||
blocked = sorted(
|
||||
{b["model"]: b for b in blobs
|
||||
if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb
|
||||
and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(),
|
||||
key=lambda b: -b["size_gb"])
|
||||
|
||||
names = ", ".join(b["model"] for b in blocked[:3])
|
||||
who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2])
|
||||
if blocked:
|
||||
return _check("unmanaged VRAM", DEGRADED,
|
||||
f"{unmanaged_gb} GB held by {who}",
|
||||
f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB "
|
||||
f"but not the {ceiling_gb:.2f} GB actually available: {names}",
|
||||
"Stop that process to reclaim the difference, or accept that "
|
||||
"these models cannot load")
|
||||
return _check("unmanaged VRAM", OK,
|
||||
f"{unmanaged_gb} GB held by {who}; no model is blocked by it",
|
||||
"", "")
|
||||
|
||||
|
||||
def _check_model_dirs() -> Dict[str, Any]:
|
||||
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
|
||||
if not os.path.isdir(comfy_dir):
|
||||
@@ -158,7 +228,7 @@ async def run_health_checks() -> Dict[str, Any]:
|
||||
|
||||
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
|
||||
_check_profile_drift, _check_store, _check_residency,
|
||||
_check_model_dirs, _check_comfy_ws]
|
||||
_check_model_dirs, _check_comfy_ws, _check_unmanaged_vram]
|
||||
results: List[Dict[str, Any]] = []
|
||||
for fn in sync_checks:
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user