[ { "name": "ollama", "kind": "llm", "priority": 50, "match": { "names": [ "ollama" ], "cmdline": [ "llama-server", "ollama" ] }, "busy": { "type": "http_count", "url": "http://localhost:11434/api/ps", "count_keys": [ "models" ] }, "release": { "type": "http_post", "url": "http://localhost:11434/api/generate", "body": { "keep_alive": 0 }, "per_model": true, "timeout_s": 120.0 }, "notes": "Unloads per model. With OLLAMA_NUM_PARALLEL=1 the request queues behind any running generation and applies when it finishes." }, { "name": "comfyui", "kind": "diffusion", "priority": 60, "match": { "cmdline": [ "comfyui", "comfy" ], "cmdline_endswith": [ "main.py" ] }, "busy": { "type": "http_count", "url": "http://127.0.0.1:8188/queue", "count_keys": [ "queue_running", "queue_pending" ], "vram_floor_gb": 1.5, "stale_after_s": 90.0 }, "release": { "type": "http_post", "url": "http://127.0.0.1:8188/free", "body": { "unload_models": true, "free_memory": true }, "timeout_s": 30.0 }, "notes": "Leaves dead jobs in queue_running; the queue flag is corroborated against its own VRAM before being believed." }, { "name": "stt-relay", "kind": "other", "priority": 70, "match": { "cmdline": [ "stt_relay.py" ] }, "busy": { "type": "vram", "vram_busy_gb": 1.0 }, "release": { "type": "none" }, "notes": "Long-running speech relay. Holds ~0.8 GB permanently and exposes no release API, so its VRAM is headroom this service can never offer. Declared so it is named rather than lumped into 'unmanaged'." }, { "name": "desktop", "kind": "desktop", "priority": 90, "match": { "names": [ "gnome-shell", "xorg", "mutter", "kwin", "plasmashell", "gnome-remote-desktop", "sddm", "gdm", "picom", "weston" ] }, "release": { "type": "none" }, "notes": "Compositor and display server. Small, permanent, never reclaimable." } ]