Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit

Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network.
An autouse fixture stubs overclock_manager._sh -- the single choke point for every
nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately
pin the empirically measured constants that would otherwise rot silently: the cold and
warm load figures behind the cache-hit thresholds, the warm_confident residency rule,
and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the
measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot
come back.

Three bugs the suite surfaced, now fixed:
- autotune._subsample(values, 1) divided by zero; the early return only covered
  len(values) <= max_steps.
- telemetry_store.stop() flushed its local pending list but never drained the queue,
  silently losing rows submitted just before a shutdown -- exactly when the last
  events matter.
- ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other
  return path provides, so a 0-byte file was planned for warming.

Reclaim. The README has claimed bidirectional arbitration from the start, but only one
direction was ever automatic. Establishing what actually happens took a controlled test
with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU
on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is
pinned to 99 and it will not reduce the layer count. So both failure modes are handled:
_check_ollama_starved watches size_vram < size for the default configuration where Ollama
does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle
ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now
succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-01 13:40:30 -07:00
parent bacaf50713
commit 868d82794d
16 changed files with 1972 additions and 7 deletions

View File

@@ -61,6 +61,25 @@ YIELD_CONFIRM_TIMEOUT_BLOCKING_S = 30.0
# A model still holding VRAM while the GPU is pinned is generating, not wedged.
BUSY_UTIL_PCT = 50
BUSY_PROBE_S = 0.6
# Fraction of a model that may sit outside VRAM before we call it starved. A little
# slack absorbs rounding and KV-cache accounting; beyond it, layers are on the CPU.
CPU_OFFLOAD_TOLERANCE = 0.02
# Only intervene when ComfyUI is actually holding enough VRAM to be the cause.
RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2
# Ollama's response when a model will not fit. Which of the two failure modes you get
# depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU
# and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and
# returns a hard CUDA OOM instead. Both are handled -- the spill by
# AutoArbitrator._check_ollama_starved, the hard failure by the retry below.
OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
"failed to allocate", "cuda error")
def looks_like_vram_oom(text: str) -> bool:
low = (text or "").lower()
return any(sig in low for sig in OOM_SIGNATURES)
YIELD_CONFIRM_POLL_S = 0.02
YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released"
@@ -360,7 +379,13 @@ async def get_ollama_live_state() -> Dict[str, Any]:
"active_model_vram_gb": 0.0,
"active_context": 0,
"expires_at": None,
"installed_models": []
"installed_models": [],
# Ollama silently spills layers to CPU when VRAM is short. size_vram < size is the
# only externally visible sign, and the cost is roughly an order of magnitude in
# decode speed, so it is worth surfacing loudly.
"gpu_fraction": 1.0,
"cpu_offload_pct": 0.0,
"partially_offloaded": False,
}
try:
client = _client(OLLAMA_API_BASE, 3.0)
@@ -694,8 +719,13 @@ def classify_load(size_bytes: int, load_duration_ms: float) -> Dict[str, Any]:
return {"cache_status": status, "load_gbps": round(gbps, 2), "is_ram_hit": gbps >= RAM_HIT_GBPS}
async def switch_ollama_model(target_model: str, keep_alive: str = "30m") -> Dict[str, Any]:
"""High-speed hot-swap to target Ollama model, tracking swap metrics."""
async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
_retrying: bool = False) -> Dict[str, Any]:
"""High-speed hot-swap to target Ollama model, tracking swap metrics.
If the load fails because the model will not fit, reclaims VRAM from an idle ComfyUI
and retries once. `_retrying` guards against recursing more than one level.
"""
t0 = time.perf_counter()
cur_state = await get_ollama_live_state()
prev_model = cur_state.get("active_model_name") or "None"
@@ -745,8 +775,38 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m") -> Dic
"is_ram_hit": cls["is_ram_hit"],
"response": data.get("response", ""),
}
return {"success": False, "error": f"HTTP {resp.status_code}: {resp.text}",
"duration_ms": total_duration_ms}
# A model that will not fit is the exact contention this service exists to
# resolve. Rather than handing the caller a CUDA OOM, take the VRAM back from an
# idle ComfyUI and try once more.
body = resp.text
if looks_like_vram_oom(body) and not _retrying:
snap = get_process_vram_bytes()
if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES:
logger.warning(
f"Ollama could not fit '{target_model}' with ComfyUI holding "
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying")
purge = await instant_free_comfyui_vram()
arbitrator.stats["reclaims_for_ollama"] += 1
arbitrator.last_action = (
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so "
f"'{target_model}' could load")
_record({
"event_type": "VRAM Reclaim for Ollama",
"source": "ComfyUI Pipeline",
"target": target_model,
"duration_ms": purge.get("duration_ms"),
"cache_status": "Reclaimed",
"detail": f"Ollama OOM: {body[:160]}",
})
await asyncio.sleep(0.3)
retry = await switch_ollama_model(target_model, keep_alive, _retrying=True)
retry["reclaimed_from_comfyui_gb"] = round(
snap["comfyui_bytes"] / (1024**3), 2)
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
return retry
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
"duration_ms": total_duration_ms,
"vram_oom": looks_like_vram_oom(body)}
except Exception as e:
return {"success": False, "error": str(e),
"duration_ms": round((time.perf_counter() - t0) * 1000, 2)}
@@ -795,6 +855,7 @@ class AutoArbitrator:
# again every second just blocks the loop repeatedly for no benefit.
self._yield_backoff_until: Dict[str, float] = {}
self._yield_busy_streak: Dict[str, int] = {}
self.last_reclaim_time = 0.0
self.stats = {
"yields": 0, # release confirmed
"yield_deferred_busy": 0, # model mid-generation; unload queued behind it
@@ -802,6 +863,7 @@ class AutoArbitrator:
"deferred_releases": 0, # queued unloads that later landed
"purges": 0,
"deferred_purges": 0,
"reclaims_for_ollama": 0, # ComfyUI purged because the LLM was spilling to CPU
}
async def start(self):
@@ -984,6 +1046,57 @@ class AutoArbitrator:
await asyncio.sleep(backoff)
backoff = min(backoff * 1.5, 15.0)
RECLAIM_COOLDOWN_S = 30.0
async def _check_ollama_starved(self) -> None:
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the
README calling the arbitration bidirectional. When Ollama cannot fit a model it
does not fail, it silently places layers on the CPU and runs about an order of
magnitude slower -- so this is the failure mode a user is least likely to notice
and most likely to feel.
If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
checkpoints are the thing to give up.
"""
now = time.time()
if self.comfy_was_active or (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
return
ollama = await get_ollama_live_state()
if not ollama.get("partially_offloaded"):
return
snap = get_process_vram_bytes()
if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
return # ComfyUI is not the one holding the memory; nothing we can do here
self.last_reclaim_time = now
model = ollama.get("active_model_name")
offload = ollama.get("cpu_offload_pct")
logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds "
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM")
await self._purge_comfy_now(f"LLM spilling {offload}% to CPU")
self.stats["reclaims_for_ollama"] += 1
# Freeing VRAM does not move layers back; only a reload re-places the model. Do
# that only when the model is idle, never mid-generation.
after = get_process_vram_bytes()
if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model:
logger.info(f"Reloading '{model}' to place it fully on the GPU...")
await instant_free_ollama_vram(model, confirm=True)
res = await switch_ollama_model(model, keep_alive="30m")
recheck = await get_ollama_live_state()
self.last_action = (
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and "
f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU"
if res.get("success") else
f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}")
else:
self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will "
f"stay partly on CPU until its next load")
async def _poll_watchdog(self):
"""Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz.
@@ -1002,6 +1115,8 @@ class AutoArbitrator:
await self.trigger_comfy_priority("Watchdog saw an active queue")
elif self.comfy_was_active:
await self.trigger_comfy_completed()
else:
await self._check_ollama_starved()
except Exception:
pass
await asyncio.sleep(interval)