"""Dependency self-check. Motivation: fan control failed for an entire session because the service started before the headless X server that owns the GPU was accepting connections. The failure was real, recoverable and completely invisible -- it appeared once, inside one field of one log line, and nothing ever asked "is fan control actually working?" Everything HyperSwap needs is checked here, each with a plain statement of what breaks when it is missing and how to fix it. A degraded dependency should be loud. """ import asyncio import logging import os import time from typing import Any, Dict, List import httpx import overclock_manager import ram_optimizer import telemetry_store import vram_arbitrator logger = logging.getLogger("health") OK, DEGRADED, FAILED = "ok", "degraded", "failed" def _check(name: str, status: str, detail: str, impact: str = "", fix: str = "") -> Dict[str, Any]: return {"name": name, "status": status, "detail": detail, "impact": impact, "fix": fix} def _check_nvml() -> Dict[str, Any]: if not vram_arbitrator.NVML_AVAILABLE: return _check("nvml", FAILED, "pynvml did not initialise", "No GPU telemetry, and VRAM yields cannot be confirmed", "Check the NVIDIA driver and that pynvml is installed in the venv") stats = vram_arbitrator.get_gpu_hardware_stats() if not stats.get("available"): return _check("nvml", FAILED, stats.get("error", "unavailable"), "No GPU telemetry", "Check the NVIDIA driver") return _check("nvml", OK, f"{stats.get('device_name')}, " f"{stats.get('vram_total_gb')} GB") def _check_sudo_smi() -> Dict[str, Any]: r = overclock_manager._smi("--query-gpu=name", "--format=csv,noheader") if r["rc"] != 0: return _check("nvidia-smi (sudo)", FAILED, r.get("err") or "non-zero exit", "Power limits and clock locks cannot be applied", "Passwordless sudo for /usr/bin/nvidia-smi is required " "(see /etc/sudoers.d/)") return _check("nvidia-smi (sudo)", OK, "passwordless sudo works") def _check_fan_control() -> Dict[str, Any]: """The check that would have caught the startup race.""" if not overclock_manager.is_headless_x_running(): return _check("fan control", FAILED, f"no X server found on {overclock_manager.HEADLESS_DISPLAY}", "Fan speed cannot be read or set; the thermal governor cannot " "raise the fan floor when the card gets hot", f"Start the headless X server on {overclock_manager.HEADLESS_DISPLAY}") status = overclock_manager.get_fan_status(force=True) if status.get("target_speed_pct") is None and not status.get("manual"): # Auto mode legitimately reports no target; probe the control attribute instead. probe = overclock_manager._nvidia_settings("-q", "[gpu:0]/GPUFanControlState") if probe.get("rc") != 0 or overclock_manager._fan_target_missing(probe): return _check("fan control", FAILED, probe.get("err") or "GPU target not resolvable", "Fan control unavailable; the governor cannot cool the card", "Check Coolbits and that X on " f"{overclock_manager.HEADLESS_DISPLAY} owns the GPU") return _check("fan control", OK, f"mode={status.get('mode')}") def _check_profile_drift() -> Dict[str, Any]: drift = overclock_manager.profile_drift() if not drift.get("applied_since_start"): return _check("overclock profile", DEGRADED, "no profile has been successfully applied since startup", "The card may not be running the settings this app reports", "Apply a profile, or check the nvidia-smi/fan checks above") if drift.get("drifted"): return _check("overclock profile", DEGRADED, drift.get("reason", "drifted"), "Reported settings do not match the hardware", "The sampler reconciles once a minute; POST /api/overclock/apply " "to force it now") return _check("overclock profile", OK, f"{drift['profile']} @ {drift['power_limit_actual_w']}W") async def _check_http(name: str, url: str, impact: str, fix: str) -> Dict[str, Any]: try: async with httpx.AsyncClient(timeout=3.0) as c: r = await c.get(url) if r.status_code == 200: return _check(name, OK, "reachable") return _check(name, DEGRADED, f"HTTP {r.status_code}", impact, fix) except Exception as e: return _check(name, FAILED, str(e)[:120], impact, fix) def _check_comfy_ws() -> Dict[str, Any]: arb = vram_arbitrator.arbitrator if not arb.running: return _check("arbitrator", FAILED, "background engine not running", "No automatic VRAM handoff between Ollama and ComfyUI", "Restart the service") if not arb.connected_ws: return _check("comfyui websocket", DEGRADED, "not connected", "Falling back to 1 Hz polling; handoffs react more slowly", "Check that ComfyUI is running and reachable on :8188") return _check("comfyui websocket", OK, "subscribed") def _check_store() -> Dict[str, Any]: info = telemetry_store.db_info() if not info.get("exists"): return _check("telemetry store", DEGRADED, "database not created yet", "No persisted history, so profile comparison cannot be computed", "It is created on first write; check the directory is writable") if not os.access(telemetry_store.DB_PATH, os.W_OK): return _check("telemetry store", FAILED, "database not writable", "Telemetry and swap events are being dropped", f"Check permissions on {telemetry_store.DB_PATH}") return _check("telemetry store", OK, f"{info.get('size_mb')} MB, {info.get('coverage_hours')} h of history") def _check_residency() -> Dict[str, Any]: cap = ram_optimizer.residency_capability() if cap.get("exact_everywhere"): return _check("residency measurement", OK, "cachestat available for all models") return _check("residency measurement", DEGRADED, cap.get("reason", ""), "Ollama weight residency is estimated by read-rate probe, not measured", cap.get("hint", "")) def _check_model_dirs() -> Dict[str, Any]: comfy_dir = ram_optimizer.COMFY_MODELS_DIR if not os.path.isdir(comfy_dir): return _check("model directories", DEGRADED, f"ComfyUI model directory not found: {comfy_dir}", "ComfyUI checkpoints cannot be catalogued or pre-warmed", "Set HYPERSWAP_COMFY_MODELS to the right path") catalog = ram_optimizer.get_model_catalog() return _check("model directories", OK, f"{len(catalog['ollama'])} Ollama blobs, {len(catalog['comfy'])} ComfyUI files") async def run_health_checks() -> Dict[str, Any]: """Run every dependency check. Never raises.""" t0 = time.perf_counter() loop = asyncio.get_running_loop() sync_checks = [_check_nvml, _check_sudo_smi, _check_fan_control, _check_profile_drift, _check_store, _check_residency, _check_model_dirs, _check_comfy_ws] results: List[Dict[str, Any]] = [] for fn in sync_checks: try: results.append(await loop.run_in_executor(None, fn)) except Exception as e: results.append(_check(fn.__name__, FAILED, f"check raised: {e}")) results.extend(await asyncio.gather( _check_http("ollama", f"{vram_arbitrator.OLLAMA_API_BASE}/api/tags", "No LLM orchestration", "Start the ollama service"), _check_http("comfyui", f"{vram_arbitrator.COMFY_API_BASE}/system_stats", "No diffusion arbitration", "Start ComfyUI on :8188"), )) failed = [r for r in results if r["status"] == FAILED] degraded = [r for r in results if r["status"] == DEGRADED] overall = FAILED if failed else (DEGRADED if degraded else OK) return { "status": overall, "checked_at": time.time(), "duration_ms": round((time.perf_counter() - t0) * 1000, 1), "summary": (f"{len(results) - len(failed) - len(degraded)} ok, " f"{len(degraded)} degraded, {len(failed)} failed"), "failed": [r["name"] for r in failed], "degraded": [r["name"] for r in degraded], "checks": results, }