Chasing why the reverse-direction reclaim never fired turned up something worse than the reclaim itself. The starvation check was never running. Instrumenting the watchdog showed busy=6, idle_check=0: every poll took the "ComfyUI is busy" branch. ComfyUI's /queue was reporting a WAN 2.1 i2v job in queue_running while the GPU sat at 0% and ComfyUI held 0.56 GB. The job was dead; ComfyUI had simply never cleared the row. Believing that flag meant this service thought ComfyUI was permanently busy, so it yielded the LLM's VRAM on every poll, never ran the idle purge, and never checked whether the LLM had been squeezed onto the CPU. One stale row disabled half of the arbitration, and it very likely explains the earlier burst of yields against a cron-driven model. A running entry is now corroborated before it is believed. The first attempt used GPU utilisation, which does not work: utilisation is shared with Ollama and with the third-party process on this box, so peak utilisation stayed above any sensible threshold and a stuck entry never looked stale. ComfyUI's own VRAM is the right signal -- a real diffusion job loads gigabytes of checkpoint, a dead one holds only its CUDA context. After the fix the same watchdog reports busy=3, idle_check=32. Every early return in the starvation check now records why it bailed, because with four of them there was no way to tell which had fired. /api/health reports a stale queue entry with its impact and how to clear it. Also confirmed, contradicting an earlier conclusion in this branch: Ollama on this box *does* spill to the CPU. smtek/Qwen3.8-27B:Q2_K_XL held steady at 29.2% on GPU (size=15.59 GB, size_vram=4.56 GB) across twelve seconds of polling -- a stable placement, not a progressive load. Both failure modes are real; which one occurs depends on the model. Tests: 206 (was 199). Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
278 lines
13 KiB
Python
278 lines
13 KiB
Python
"""Dependency self-check.
|
|
|
|
Motivation: fan control failed for an entire session because the service started before
|
|
the headless X server that owns the GPU was accepting connections. The failure was real,
|
|
recoverable and completely invisible -- it appeared once, inside one field of one log
|
|
line, and nothing ever asked "is fan control actually working?"
|
|
|
|
Everything HyperSwap needs is checked here, each with a plain statement of what breaks
|
|
when it is missing and how to fix it. A degraded dependency should be loud.
|
|
"""
|
|
import asyncio
|
|
import logging
|
|
import time
|
|
import os
|
|
import time
|
|
from typing import Any, Dict, List
|
|
|
|
import httpx
|
|
|
|
import overclock_manager
|
|
import ram_optimizer
|
|
import telemetry_store
|
|
import vram_arbitrator
|
|
|
|
logger = logging.getLogger("health")
|
|
|
|
OK, DEGRADED, FAILED = "ok", "degraded", "failed"
|
|
|
|
|
|
def _check(name: str, status: str, detail: str, impact: str = "",
|
|
fix: str = "") -> Dict[str, Any]:
|
|
return {"name": name, "status": status, "detail": detail,
|
|
"impact": impact, "fix": fix}
|
|
|
|
|
|
def _check_nvml() -> Dict[str, Any]:
|
|
if not vram_arbitrator.NVML_AVAILABLE:
|
|
return _check("nvml", FAILED, "pynvml did not initialise",
|
|
"No GPU telemetry, and VRAM yields cannot be confirmed",
|
|
"Check the NVIDIA driver and that pynvml is installed in the venv")
|
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
|
if not stats.get("available"):
|
|
return _check("nvml", FAILED, stats.get("error", "unavailable"),
|
|
"No GPU telemetry", "Check the NVIDIA driver")
|
|
return _check("nvml", OK, f"{stats.get('device_name')}, "
|
|
f"{stats.get('vram_total_gb')} GB")
|
|
|
|
|
|
def _check_sudo_smi() -> Dict[str, Any]:
|
|
r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader")
|
|
if r["rc"] != 0:
|
|
return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit",
|
|
"Power limits and clock locks cannot be applied",
|
|
"Passwordless sudo for /usr/bin/nvidia-smi is required "
|
|
"(see /etc/sudoers.d/)")
|
|
return _check("nvidia-smi (sudo)", OK, "passwordless sudo works")
|
|
|
|
|
|
def _check_fan_control() -> Dict[str, Any]:
|
|
"""The check that would have caught the startup race."""
|
|
if not overclock_manager.is_headless_x_running():
|
|
return _check("fan control", FAILED,
|
|
f"no X server found on {overclock_manager.HEADLESS_DISPLAY}",
|
|
"Fan speed cannot be read or set; the thermal governor cannot "
|
|
"raise the fan floor when the card gets hot",
|
|
f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}")
|
|
status = overclock_manager.get_fan_status(force=True)
|
|
if status.get("target_speed_pct") is None and not status.get("manual"):
|
|
# Auto mode legitimately reports no target; probe the control attribute instead.
|
|
probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState")
|
|
if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe):
|
|
return _check("fan control", FAILED,
|
|
probe.get("err") or "GPU target not resolvable",
|
|
"Fan control unavailable; the governor cannot cool the card",
|
|
"Check Coolbits and that X on "
|
|
f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU")
|
|
return _check("fan control", OK, f"mode={status.get('mode')}")
|
|
|
|
|
|
def _check_profile_drift() -> Dict[str, Any]:
|
|
drift = overclock_manager.profile_drift()
|
|
if not drift.get("applied_since_start"):
|
|
return _check("overclock profile", DEGRADED,
|
|
"no profile has been successfully applied since startup",
|
|
"The card may not be running the settings this app reports",
|
|
"Apply a profile, or check the nvidia-smi/fan checks above")
|
|
if drift.get("drifted"):
|
|
return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"),
|
|
"Reported settings do not match the hardware",
|
|
"The sampler reconciles once a minute; POST /api/overclock/apply "
|
|
"to force it now")
|
|
return _check("overclock profile", OK,
|
|
f"{drift['profile']} @ {drift['power_limit_actual_w']}W")
|
|
|
|
|
|
async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]:
|
|
try:
|
|
async with httpx.AsyncClient(timeout=3.0) as c:
|
|
r = await c.get(url)
|
|
if r.status_code == 200:
|
|
return _check(name, OK, "reachable")
|
|
return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix)
|
|
except Exception as e:
|
|
return _check(name, FAILED, str(e)[:120], impact, fix)
|
|
|
|
|
|
def _check_comfy_ws() -> Dict[str, Any]:
|
|
arb = vram_arbitrator.arbitrator
|
|
if not arb.running:
|
|
return _check("arbitrator", FAILED, "background engine not running",
|
|
"No automatic VRAM handoff between Ollama and ComfyUI",
|
|
"Restart the service")
|
|
if not arb.connected_ws:
|
|
return _check("comfyui websocket", DEGRADED, "not connected",
|
|
"Falling back to 1 Hz polling; handoffs react more slowly",
|
|
"Check that ComfyUI is running and reachable on :8188")
|
|
return _check("comfyui websocket", OK, "subscribed")
|
|
|
|
|
|
def _check_comfy_queue() -> Dict[str, Any]:
|
|
"""A stale ComfyUI queue entry disables half of this service's logic."""
|
|
arb = vram_arbitrator.arbitrator
|
|
if arb.comfy_stale_job:
|
|
return _check("comfyui queue", DEGRADED,
|
|
f"prompt {arb.comfy_stale_job} claims to be running but the GPU is idle",
|
|
"ComfyUI looks permanently busy, so the LLM is evicted repeatedly, "
|
|
"the idle purge never runs and CPU-spill is never checked",
|
|
"Clear it from the ComfyUI queue, or POST /queue with "
|
|
"{\"clear\": true} to ComfyUI")
|
|
branches = arb.watchdog_branches
|
|
if branches.get("idle_check", 0) == 0 and branches.get("busy", 0) > 20:
|
|
return _check("comfyui queue", DEGRADED,
|
|
"the watchdog has only ever seen ComfyUI as busy",
|
|
"The idle purge and starvation check are not running",
|
|
"Check the ComfyUI queue for a stuck entry")
|
|
return _check("comfyui queue", OK, "queue state corroborated against GPU activity")
|
|
|
|
|
|
def _check_store() -> Dict[str, Any]:
|
|
info = telemetry_store.db_info()
|
|
if not info.get("exists"):
|
|
return _check("telemetry store", DEGRADED, "database not created yet",
|
|
"No persisted history, so profile comparison cannot be computed",
|
|
"It is created on first write; check the directory is writable")
|
|
if not os.access(telemetry_store.DB_PATH, os.W_OK):
|
|
return _check("telemetry store", FAILED, "database not writable",
|
|
"Telemetry and swap events are being dropped",
|
|
f"Check permissions on {telemetry_store.DB_PATH}")
|
|
return _check("telemetry store", OK,
|
|
f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history")
|
|
|
|
|
|
def _check_residency() -> Dict[str, Any]:
|
|
cap = ram_optimizer.residency_capability()
|
|
if cap.get("exact_everywhere"):
|
|
return _check("residency measurement", OK, "cachestat available for all models")
|
|
return _check("residency measurement", DEGRADED, cap.get("reason", ""),
|
|
"Ollama weight residency is estimated by read-rate probe, not measured",
|
|
cap.get("hint", ""))
|
|
|
|
|
|
def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float:
|
|
"""Lowest VRAM ComfyUI has been observed holding while alive.
|
|
|
|
A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred
|
|
MB for as long as the process runs. The minimum seen in recent telemetry is a better
|
|
estimate of that floor than whatever it happens to hold right now, which could be a
|
|
7 GB checkpoint mid-generation.
|
|
"""
|
|
try:
|
|
rows = telemetry_store._rows(
|
|
"SELECT MIN(comfy_bytes) AS floor FROM telemetry "
|
|
"WHERE ts > ? AND comfy_bytes > 0",
|
|
(time.time() - days * 86400,))
|
|
if rows and rows[0].get("floor"):
|
|
return round(rows[0]["floor"] / (1024 ** 3), 2)
|
|
except Exception as e:
|
|
logger.debug(f"comfy floor lookup failed: {e}")
|
|
return default
|
|
|
|
|
|
def _check_unmanaged_vram() -> Dict[str, Any]:
|
|
"""Report unreclaimable VRAM in terms of what it actually costs.
|
|
|
|
"0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your
|
|
models can no longer fit" is something you can act on.
|
|
"""
|
|
stats = vram_arbitrator.get_gpu_hardware_stats()
|
|
if not stats.get("available"):
|
|
return _check("unmanaged VRAM", DEGRADED, "GPU unavailable")
|
|
bd = stats.get("breakdown", {})
|
|
unmanaged_gb = bd.get("unmanaged_gb", 0.0)
|
|
procs = bd.get("unmanaged", [])
|
|
if not procs:
|
|
return _check("unmanaged VRAM", OK, "no third-party GPU processes")
|
|
|
|
total_gb = stats.get("vram_total_gb", 0)
|
|
# What HyperSwap could offer at best. Three things are never available: the desktop,
|
|
# processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge.
|
|
# Omitting that last one made this check claim a 14.93 GB model would fit against a
|
|
# real ceiling of 14.60 GB -- the model that had just returned 507.
|
|
comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0))
|
|
ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb
|
|
try:
|
|
blobs = ram_optimizer.find_ollama_model_files()
|
|
except Exception:
|
|
blobs = []
|
|
# Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache
|
|
# are allocated.
|
|
VRAM_OVERHEAD = 1.16
|
|
blocked = sorted(
|
|
{b["model"]: b for b in blobs
|
|
if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb
|
|
and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(),
|
|
key=lambda b: -b["size_gb"])
|
|
|
|
names = ", ".join(b["model"] for b in blocked[:3])
|
|
who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2])
|
|
if blocked:
|
|
return _check("unmanaged VRAM", DEGRADED,
|
|
f"{unmanaged_gb} GB held by {who}",
|
|
f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB "
|
|
f"but not the {ceiling_gb:.2f} GB actually available: {names}",
|
|
"Stop that process to reclaim the difference, or accept that "
|
|
"these models cannot load")
|
|
return _check("unmanaged VRAM", OK,
|
|
f"{unmanaged_gb} GB held by {who}; no model is blocked by it",
|
|
"", "")
|
|
|
|
|
|
def _check_model_dirs() -> Dict[str, Any]:
|
|
comfy_dir = ram_optimizer.COMFY_MODELS_DIR
|
|
if not os.path.isdir(comfy_dir):
|
|
return _check("model directories", DEGRADED,
|
|
f"ComfyUI model directory not found: {comfy_dir}",
|
|
"ComfyUI checkpoints cannot be catalogued or pre-warmed",
|
|
"Set HYPERSWAP_COMFY_MODELS to the right path")
|
|
catalog = ram_optimizer.get_model_catalog()
|
|
return _check("model directories", OK,
|
|
f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files")
|
|
|
|
|
|
async def run_health_checks() -> Dict[str, Any]:
|
|
"""Run every dependency check. Never raises."""
|
|
t0 = time.perf_counter()
|
|
loop = asyncio.get_running_loop()
|
|
|
|
sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control,
|
|
_check_profile_drift, _check_store, _check_residency,
|
|
_check_model_dirs, _check_comfy_ws, _check_unmanaged_vram]
|
|
results: List[Dict[str, Any]] = []
|
|
for fn in sync_checks:
|
|
try:
|
|
results.append(await loop.run_in_executor(None, fn))
|
|
except Exception as e:
|
|
results.append(_check(fn.__name__, FAILED, f"check raised: {e}"))
|
|
|
|
results.extend(await asyncio.gather(
|
|
_check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags",
|
|
"No LLM orchestration", "Start the ollama service"),
|
|
_check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats",
|
|
"No diffusion arbitration", "Start ComfyUI on :8188"),
|
|
))
|
|
|
|
failed = [r for r in results if r["status"] == FAILED]
|
|
degraded = [r for r in results if r["status"] == DEGRADED]
|
|
overall = FAILED if failed else (DEGRADED if degraded else OK)
|
|
return {
|
|
"status": overall,
|
|
"checked_at": time.time(),
|
|
"duration_ms": round((time.perf_counter() - t0) * 1000, 1),
|
|
"summary": (f"{len(results) - len(failed) - len(degraded)} ok, "
|
|
f"{len(degraded)} degraded, {len(failed)} failed"),
|
|
"failed": [r["name"] for r in failed],
|
|
"degraded": [r["name"] for r in degraded],
|
|
"checks": results,
|
|
}
|