The health check now reports what unmanaged VRAM actually costs rather than just how much of it there is: "0.82 GB held by python (842 MB)" becomes "5 model(s) fit within 15.42 GB but not the 14.60 GB actually available", naming them. Getting that arithmetic right took a correction. The first version subtracted only the desktop and the unmanaged process, and so reported a 14.93 GB model as fitting against a real ceiling of 14.60 GB -- the same model the service had just refused with 507. ComfyUI keeps a few hundred MB of CUDA context for as long as the process lives, which a purge does not free, so it is not available either. The floor is taken from the minimum ComfyUI VRAM in recent telemetry rather than its current value, which could be a 7 GB checkpoint mid-generation. The verifier's reclaim stage now re-runs a graph immediately beforehand to reset the 30 s idle window, since a large model takes longer than that to load and the purge was freeing ComfyUI mid-load, so the reclaim path was never reached. Tests: 199 (was 192). The new ones pin the ceiling arithmetic, including that a model too large to fit on the card at all is not blamed on the third-party process. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
259 lines
12 KiB
Python
259 lines
12 KiB
Python
"""Dependency self-check.
|
|
|
|
Motivation: fan control failed for an entire session because the service started before
|
|
the headless X server that owns the GPU was accepting connections. The failure was real,
|
|
recoverable and completely invisible -- it appeared once, inside one field of one log
|
|
line, and nothing ever asked "is fan control actually working?"
|
|
|
|
Everything HyperSwap needs is checked here, each with a plain statement of what breaks
|
|
when it is missing and how to fix it. A degraded dependency should be loud.
|
|
"""
|
|
import asyncio
|
|
import logging
|
|
import time
|
|
import os
|
|
import time
|
|
from typing import Any, Dict, List
|
|
|
|
import httpx
|
|
|
|
import overclock_manager
|
|
import ram_optimizer
|
|
import telemetry_store
|
|
import vram_arbitrator
|
|
|
|
logger = logging.getLogger("health")
|
|
|
|
OK, DEGRADED, FAILED = "ok", "degraded", "failed"
|
|
|
|
|
|
def _check(name: str, status: str, detail: str, impact: str = "",
|
|
fix: str = "") -> Dict[str, Any]:
|
|
return {"name": name, "status": status, "detail": detail,
|
|
"impact": impact, "fix": fix}
|
|
|
|
|
|
def _check_nvml() -> Dict[str, Any]:
|
|
if not vram_arbitrator.NVML_AVAILABLE:
|
|
return _check("nvml", FAILED, "pynvml did not initialise",
|
|
"No GPU telemetry, and VRAM yields cannot be confirmed",
|
|
"Check the NVIDIA driver and that pynvml is installed in the venv")
|
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
|
if not stats.get("available"):
|
|
return _check("nvml", FAILED, stats.get("error", "unavailable"),
|
|
"No GPU telemetry", "Check the NVIDIA driver")
|
|
return _check("nvml", OK, f"{stats.get('device_name')}, "
|
|
f"{stats.get('vram_total_gb')} GB")
|
|
|
|
|
|
def _check_sudo_smi() -> Dict[str, Any]:
|
|
r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader")
|
|
if r["rc"] != 0:
|
|
return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit",
|
|
"Power limits and clock locks cannot be applied",
|
|
"Passwordless sudo for /usr/bin/nvidia-smi is required "
|
|
"(see /etc/sudoers.d/)")
|
|
return _check("nvidia-smi (sudo)", OK, "passwordless sudo works")
|
|
|
|
|
|
def _check_fan_control() -> Dict[str, Any]:
|
|
"""The check that would have caught the startup race."""
|
|
if not overclock_manager.is_headless_x_running():
|
|
return _check("fan control", FAILED,
|
|
f"no X server found on {overclock_manager.HEADLESS_DISPLAY}",
|
|
"Fan speed cannot be read or set; the thermal governor cannot "
|
|
"raise the fan floor when the card gets hot",
|
|
f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}")
|
|
status = overclock_manager.get_fan_status(force=True)
|
|
if status.get("target_speed_pct") is None and not status.get("manual"):
|
|
# Auto mode legitimately reports no target; probe the control attribute instead.
|
|
probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState")
|
|
if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe):
|
|
return _check("fan control", FAILED,
|
|
probe.get("err") or "GPU target not resolvable",
|
|
"Fan control unavailable; the governor cannot cool the card",
|
|
"Check Coolbits and that X on "
|
|
f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU")
|
|
return _check("fan control", OK, f"mode={status.get('mode')}")
|
|
|
|
|
|
def _check_profile_drift() -> Dict[str, Any]:
|
|
drift = overclock_manager.profile_drift()
|
|
if not drift.get("applied_since_start"):
|
|
return _check("overclock profile", DEGRADED,
|
|
"no profile has been successfully applied since startup",
|
|
"The card may not be running the settings this app reports",
|
|
"Apply a profile, or check the nvidia-smi/fan checks above")
|
|
if drift.get("drifted"):
|
|
return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"),
|
|
"Reported settings do not match the hardware",
|
|
"The sampler reconciles once a minute; POST /api/overclock/apply "
|
|
"to force it now")
|
|
return _check("overclock profile", OK,
|
|
f"{drift['profile']} @ {drift['power_limit_actual_w']}W")
|
|
|
|
|
|
async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]:
|
|
try:
|
|
async with httpx.AsyncClient(timeout=3.0) as c:
|
|
r = await c.get(url)
|
|
if r.status_code == 200:
|
|
return _check(name, OK, "reachable")
|
|
return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix)
|
|
except Exception as e:
|
|
return _check(name, FAILED, str(e)[:120], impact, fix)
|
|
|
|
|
|
def _check_comfy_ws() -> Dict[str, Any]:
|
|
arb = vram_arbitrator.arbitrator
|
|
if not arb.running:
|
|
return _check("arbitrator", FAILED, "background engine not running",
|
|
"No automatic VRAM handoff between Ollama and ComfyUI",
|
|
"Restart the service")
|
|
if not arb.connected_ws:
|
|
return _check("comfyui websocket", DEGRADED, "not connected",
|
|
"Falling back to 1 Hz polling; handoffs react more slowly",
|
|
"Check that ComfyUI is running and reachable on :8188")
|
|
return _check("comfyui websocket", OK, "subscribed")
|
|
|
|
|
|
def _check_store() -> Dict[str, Any]:
|
|
info = telemetry_store.db_info()
|
|
if not info.get("exists"):
|
|
return _check("telemetry store", DEGRADED, "database not created yet",
|
|
"No persisted history, so profile comparison cannot be computed",
|
|
"It is created on first write; check the directory is writable")
|
|
if not os.access(telemetry_store.DB_PATH, os.W_OK):
|
|
return _check("telemetry store", FAILED, "database not writable",
|
|
"Telemetry and swap events are being dropped",
|
|
f"Check permissions on {telemetry_store.DB_PATH}")
|
|
return _check("telemetry store", OK,
|
|
f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history")
|
|
|
|
|
|
def _check_residency() -> Dict[str, Any]:
|
|
cap = ram_optimizer.residency_capability()
|
|
if cap.get("exact_everywhere"):
|
|
return _check("residency measurement", OK, "cachestat available for all models")
|
|
return _check("residency measurement", DEGRADED, cap.get("reason", ""),
|
|
"Ollama weight residency is estimated by read-rate probe, not measured",
|
|
cap.get("hint", ""))
|
|
|
|
|
|
def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float:
|
|
"""Lowest VRAM ComfyUI has been observed holding while alive.
|
|
|
|
A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred
|
|
MB for as long as the process runs. The minimum seen in recent telemetry is a better
|
|
estimate of that floor than whatever it happens to hold right now, which could be a
|
|
7 GB checkpoint mid-generation.
|
|
"""
|
|
try:
|
|
rows = telemetry_store._rows(
|
|
"SELECT MIN(comfy_bytes) AS floor FROM telemetry "
|
|
"WHERE ts > ? AND comfy_bytes > 0",
|
|
(time.time() - days * 86400,))
|
|
if rows and rows[0].get("floor"):
|
|
return round(rows[0]["floor"] / (1024 ** 3), 2)
|
|
except Exception as e:
|
|
logger.debug(f"comfy floor lookup failed: {e}")
|
|
return default
|
|
|
|
|
|
def _check_unmanaged_vram() -> Dict[str, Any]:
|
|
"""Report unreclaimable VRAM in terms of what it actually costs.
|
|
|
|
"0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your
|
|
models can no longer fit" is something you can act on.
|
|
"""
|
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
|
if not stats.get("available"):
|
|
return _check("unmanaged VRAM", DEGRADED, "GPU unavailable")
|
|
bd = stats.get("breakdown", {})
|
|
unmanaged_gb = bd.get("unmanaged_gb", 0.0)
|
|
procs = bd.get("unmanaged", [])
|
|
if not procs:
|
|
return _check("unmanaged VRAM", OK, "no third-party GPU processes")
|
|
|
|
total_gb = stats.get("vram_total_gb", 0)
|
|
# What HyperSwap could offer at best. Three things are never available: the desktop,
|
|
# processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge.
|
|
# Omitting that last one made this check claim a 14.93 GB model would fit against a
|
|
# real ceiling of 14.60 GB -- the model that had just returned 507.
|
|
comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0))
|
|
ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb
|
|
try:
|
|
blobs = ram_optimizer.find_ollama_model_files()
|
|
except Exception:
|
|
blobs = []
|
|
# Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache
|
|
# are allocated.
|
|
VRAM_OVERHEAD = 1.16
|
|
blocked = sorted(
|
|
{b["model"]: b for b in blobs
|
|
if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb
|
|
and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(),
|
|
key=lambda b: -b["size_gb"])
|
|
|
|
names = ", ".join(b["model"] for b in blocked[:3])
|
|
who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2])
|
|
if blocked:
|
|
return _check("unmanaged VRAM", DEGRADED,
|
|
f"{unmanaged_gb} GB held by {who}",
|
|
f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB "
|
|
f"but not the {ceiling_gb:.2f} GB actually available: {names}",
|
|
"Stop that process to reclaim the difference, or accept that "
|
|
"these models cannot load")
|
|
return _check("unmanaged VRAM", OK,
|
|
f"{unmanaged_gb} GB held by {who}; no model is blocked by it",
|
|
"", "")
|
|
|
|
|
|
def _check_model_dirs() -> Dict[str, Any]:
|
|
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
|
|
if not os.path.isdir(comfy_dir):
|
|
return _check("model directories", DEGRADED,
|
|
f"ComfyUI model directory not found: {comfy_dir}",
|
|
"ComfyUI checkpoints cannot be catalogued or pre-warmed",
|
|
"Set HYPERSWAP_COMFY_MODELS to the right path")
|
|
catalog = ram_optimizer.get_model_catalog()
|
|
return _check("model directories", OK,
|
|
f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files")
|
|
|
|
|
|
async def run_health_checks() -> Dict[str, Any]:
|
|
"""Run every dependency check. Never raises."""
|
|
t0 = time.perf_counter()
|
|
loop = asyncio.get_running_loop()
|
|
|
|
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
|
|
_check_profile_drift, _check_store, _check_residency,
|
|
_check_model_dirs, _check_comfy_ws, _check_unmanaged_vram]
|
|
results: List[Dict[str, Any]] = []
|
|
for fn in sync_checks:
|
|
try:
|
|
results.append(await loop.run_in_executor(None, fn))
|
|
except Exception as e:
|
|
results.append(_check(fn.__name__, FAILED, f"check raised: {e}"))
|
|
|
|
results.extend(await asyncio.gather(
|
|
_check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags",
|
|
"No LLM orchestration", "Start the ollama service"),
|
|
_check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats",
|
|
"No diffusion arbitration", "Start ComfyUI on :8188"),
|
|
))
|
|
|
|
failed = [r for r in results if r["status"] == FAILED]
|
|
degraded = [r for r in results if r["status"] == DEGRADED]
|
|
overall = FAILED if failed else (DEGRADED if degraded else OK)
|
|
return {
|
|
"status": overall,
|
|
"checked_at": time.time(),
|
|
"duration_ms": round((time.perf_counter() - t0) * 1000, 1),
|
|
"summary": (f"{len(results) - len(failed) - len(degraded)} ok, "
|
|
f"{len(degraded)} degraded, {len(failed)} failed"),
|
|
"failed": [r["name"] for r in failed],
|
|
"degraded": [r["name"] for r in degraded],
|
|
"checks": results,
|
|
}
|