"""Dependency self-check. Motivation: fan control failed for an entire session because the service started before the headless X server that owns the GPU was accepting connections. The failure was real, recoverable and completely invisible -- it appeared once, inside one field of one log line, and nothing ever asked "is fan control actually working?" Everything HyperSwap needs is checked here, each with a plain statement of what breaks when it is missing and how to fix it. A degraded dependency should be loud. """ import asyncio import logging import time import os import time from typing import Any, Dict, List import httpx import overclock_manager import ram_optimizer import telemetry_store import vram_arbitrator logger = logging.getLogger("health") OK, DEGRADED, FAILED = "ok", "degraded", "failed" def _check(name: str, status: str, detail: str, impact: str = "", fix: str = "") -> Dict[str, Any]: return {"name": name, "status": status, "detail": detail, "impact": impact, "fix": fix} def _check_nvml() -> Dict[str, Any]: if not vram_arbitrator.NVML_AVAILABLE: return _check("nvml", FAILED, "pynvml did not initialise", "No GPU telemetry, and VRAM yields cannot be confirmed", "Check the NVIDIA driver and that pynvml is installed in the venv") stats = vram_arbitrator.get_gpu_hardware_stats() if not stats.get("available"): return _check("nvml", FAILED, stats.get("error", "unavailable"), "No GPU telemetry", "Check the NVIDIA driver") return _check("nvml", OK, f"{stats.get('device_name')}, " f"{stats.get('vram_total_gb')} GB") def _check_sudo_smi() -> Dict[str, Any]: r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader") if r["rc"] != 0: return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit", "Power limits and clock locks cannot be applied", "Passwordless sudo for /usr/bin/nvidia-smi is required " "(see /etc/sudoers.d/)") return _check("nvidia-smi (sudo)", OK, "passwordless sudo works") def _check_fan_control() -> Dict[str, Any]: """The check that would have caught the startup race.""" if not overclock_manager.is_headless_x_running(): return _check("fan control", FAILED, f"no X server found on {overclock_manager.HEADLESS_DISPLAY}", "Fan speed cannot be read or set; the thermal governor cannot " "raise the fan floor when the card gets hot", f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}") status = overclock_manager.get_fan_status(force=True) if status.get("target_speed_pct") is None and not status.get("manual"): # Auto mode legitimately reports no target; probe the control attribute instead. probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState") if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe): return _check("fan control", FAILED, probe.get("err") or "GPU target not resolvable", "Fan control unavailable; the governor cannot cool the card", "Check Coolbits and that X on " f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU") return _check("fan control", OK, f"mode={status.get('mode')}") def _check_profile_drift() -> Dict[str, Any]: drift = overclock_manager.profile_drift() if not drift.get("applied_since_start"): return _check("overclock profile", DEGRADED, "no profile has been successfully applied since startup", "The card may not be running the settings this app reports", "Apply a profile, or check the nvidia-smi/fan checks above") if drift.get("drifted"): return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"), "Reported settings do not match the hardware", "The sampler reconciles once a minute; POST /api/overclock/apply " "to force it now") return _check("overclock profile", OK, f"{drift['profile']} @ {drift['power_limit_actual_w']}W") async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]: try: async with httpx.AsyncClient(timeout=3.0) as c: r = await c.get(url) if r.status_code == 200: return _check(name, OK, "reachable") return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix) except Exception as e: return _check(name, FAILED, str(e)[:120], impact, fix) def _check_comfy_ws() -> Dict[str, Any]: arb = vram_arbitrator.arbitrator if not arb.running: return _check("arbitrator", FAILED, "background engine not running", "No automatic VRAM handoff between Ollama and ComfyUI", "Restart the service") if not arb.connected_ws: return _check("comfyui websocket", DEGRADED, "not connected", "Falling back to 1 Hz polling; handoffs react more slowly", "Check that ComfyUI is running and reachable on :8188") return _check("comfyui websocket", OK, "subscribed") def _check_comfy_queue() -> Dict[str, Any]: """A stale ComfyUI queue entry disables half of this service's logic.""" arb = vram_arbitrator.arbitrator if arb.comfy_stale_job: return _check("comfyui queue", DEGRADED, f"prompt {arb.comfy_stale_job} claims to be running but the GPU is idle", "ComfyUI looks permanently busy, so the LLM is evicted repeatedly, " "the idle purge never runs and CPU-spill is never checked", "Clear it from the ComfyUI queue, or POST /queue with " "{\"clear\": true} to ComfyUI") branches = arb.watchdog_branches if branches.get("idle_check", 0) == 0 and branches.get("busy", 0) > 20: return _check("comfyui queue", DEGRADED, "the watchdog has only ever seen ComfyUI as busy", "The idle purge and starvation check are not running", "Check the ComfyUI queue for a stuck entry") return _check("comfyui queue", OK, "queue state corroborated against GPU activity") def _check_store() -> Dict[str, Any]: info = telemetry_store.db_info() if not info.get("exists"): return _check("telemetry store", DEGRADED, "database not created yet", "No persisted history, so profile comparison cannot be computed", "It is created on first write; check the directory is writable") if not os.access(telemetry_store.DB_PATH, os.W_OK): return _check("telemetry store", FAILED, "database not writable", "Telemetry and swap events are being dropped", f"Check permissions on {telemetry_store.DB_PATH}") return _check("telemetry store", OK, f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history") def _check_residency() -> Dict[str, Any]: cap = ram_optimizer.residency_capability() if cap.get("exact_everywhere"): return _check("residency measurement", OK, "cachestat available for all models") return _check("residency measurement", DEGRADED, cap.get("reason", ""), "Ollama weight residency is estimated by read-rate probe, not measured", cap.get("hint", "")) def _comfy_vram_floor_gb(default: float = 0.0, days: float = 1.0) -> float: """Lowest VRAM ComfyUI has been observed holding while alive. A purge frees checkpoints but not the CUDA context, so ComfyUI keeps a few hundred MB for as long as the process runs. The minimum seen in recent telemetry is a better estimate of that floor than whatever it happens to hold right now, which could be a 7 GB checkpoint mid-generation. """ try: rows = telemetry_store._rows( "SELECT MIN(comfy_bytes) AS floor FROM telemetry " "WHERE ts > ? AND comfy_bytes > 0", (time.time() - days * 86400,)) if rows and rows[0].get("floor"): return round(rows[0]["floor"] / (1024 ** 3), 2) except Exception as e: logger.debug(f"comfy floor lookup failed: {e}") return default def _check_unmanaged_vram() -> Dict[str, Any]: """Report unreclaimable VRAM in terms of what it actually costs. "0.82 GB unmanaged" is a number. "0.82 GB unmanaged, which is why three of your models can no longer fit" is something you can act on. """ stats = vram_arbitrator.get_gpu_hardware_stats() if not stats.get("available"): return _check("unmanaged VRAM", DEGRADED, "GPU unavailable") bd = stats.get("breakdown", {}) unmanaged_gb = bd.get("unmanaged_gb", 0.0) procs = bd.get("unmanaged", []) if not procs: return _check("unmanaged VRAM", OK, "no third-party GPU processes") total_gb = stats.get("vram_total_gb", 0) # What HyperSwap could offer at best. Three things are never available: the desktop, # processes it cannot touch, and ComfyUI's own CUDA context, which survives a purge. # Omitting that last one made this check claim a 14.93 GB model would fit against a # real ceiling of 14.60 GB -- the model that had just returned 507. comfy_floor_gb = _comfy_vram_floor_gb(default=bd.get("comfyui_gb", 0.0)) ceiling_gb = total_gb - unmanaged_gb - bd.get("desktop_gb", 0.0) - comfy_floor_gb try: blobs = ram_optimizer.find_ollama_model_files() except Exception: blobs = [] # Measured on this box: a 12.87 GB blob occupies 14.9 GB once context and KV cache # are allocated. VRAM_OVERHEAD = 1.16 blocked = sorted( {b["model"]: b for b in blobs if b["size_gb"] * VRAM_OVERHEAD > ceiling_gb and b["size_gb"] * VRAM_OVERHEAD <= ceiling_gb + unmanaged_gb}.values(), key=lambda b: -b["size_gb"]) names = ", ".join(b["model"] for b in blocked[:3]) who = ", ".join(f"{p['name']} ({p['vram_mb']} MB)" for p in procs[:2]) if blocked: return _check("unmanaged VRAM", DEGRADED, f"{unmanaged_gb} GB held by {who}", f"{len(blocked)} model(s) fit within {ceiling_gb + unmanaged_gb:.2f} GB " f"but not the {ceiling_gb:.2f} GB actually available: {names}", "Stop that process to reclaim the difference, or accept that " "these models cannot load") return _check("unmanaged VRAM", OK, f"{unmanaged_gb} GB held by {who}; no model is blocked by it", "", "") def _check_model_dirs() -> Dict[str, Any]: comfy_dir = ram_optimizer.COMFY_MODELS_DIR if not os.path.isdir(comfy_dir): return _check("model directories", DEGRADED, f"ComfyUI model directory not found: {comfy_dir}", "ComfyUI checkpoints cannot be catalogued or pre-warmed", "Set HYPERSWAP_COMFY_MODELS to the right path") catalog = ram_optimizer.get_model_catalog() return _check("model directories", OK, f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files") async def run_health_checks() -> Dict[str, Any]: """Run every dependency check. Never raises.""" t0 = time.perf_counter() loop = asyncio.get_running_loop() sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control, _check_profile_drift, _check_store, _check_residency, _check_model_dirs, _check_comfy_ws, _check_unmanaged_vram] results: List[Dict[str, Any]] = [] for fn in sync_checks: try: results.append(await loop.run_in_executor(None, fn)) except Exception as e: results.append(_check(fn.__name__, FAILED, f"check raised: {e}")) results.extend(await asyncio.gather( _check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags", "No LLM orchestration", "Start the ollama service"), _check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats", "No diffusion arbitration", "Start ComfyUI on :8188"), )) failed = [r for r in results if r["status"] == FAILED] degraded = [r for r in results if r["status"] == DEGRADED] overall = FAILED if failed else (DEGRADED if degraded else OK) return { "status": overall, "checked_at": time.time(), "duration_ms": round((time.perf_counter() - t0) * 1000, 1), "summary": (f"{len(results) - len(failed) - len(degraded)} ok, " f"{len(degraded)} degraded, {len(failed)} failed"), "failed": [r["name"] for r in failed], "degraded": [r["name"] for r in degraded], "checks": results, }