Add test suite (164 tests); reclaim VRAM from ComfyUI when an LLM will not fit
Tests. First automated coverage for the project: 164 tests, 2.7s, no GPU or network. An autouse fixture stubs overclock_manager._sh -- the single choke point for every nvidia-smi/nvidia-settings write -- so no test can mutate the card. They deliberately pin the empirically measured constants that would otherwise rot silently: the cold and warm load figures behind the cache-hit thresholds, the warm_confident residency rule, and the busy/stalled yield split. One test asserts RAM_HIT_GBPS stays at or below the measured 2.63 GB/s warm load, so the old physically unreachable 5.0 GB/s bar cannot come back. Three bugs the suite surfaced, now fixed: - autotune._subsample(values, 1) divided by zero; the early return only covered len(values) <= max_steps. - telemetry_store.stop() flushed its local pending list but never drained the queue, silently losing rows submitted just before a shutdown -- exactly when the last events matter. - ram_optimizer.page_residency's zero-byte short-circuit omitted keys every other return path provides, so a 0-byte file was planned for warming. Reclaim. The README has claimed bidirectional arbitration from the start, but only one direction was ever automatic. Establishing what actually happens took a controlled test with the service stopped: with ComfyUI holding 6.83 GB, Ollama does not spill to the CPU on this box -- it aborts with "cudaMalloc failed: out of memory", because n_gpu_layers is pinned to 99 and it will not reduce the layer count. So both failure modes are handled: _check_ollama_starved watches size_vram < size for the default configuration where Ollama does spill, and switch_ollama_model catches the hard OOM, reclaims VRAM from an idle ComfyUI and retries once. The request that returned HTTP 500 from Ollama directly now succeeds through HyperSwap, loading at 3.85 GB/s after reclaiming 6.83 GB. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -61,6 +61,25 @@ YIELD_CONFIRM_TIMEOUT_BLOCKING_S = 30.0
|
||||
# A model still holding VRAM while the GPU is pinned is generating, not wedged.
|
||||
BUSY_UTIL_PCT = 50
|
||||
BUSY_PROBE_S = 0.6
|
||||
|
||||
# Fraction of a model that may sit outside VRAM before we call it starved. A little
|
||||
# slack absorbs rounding and KV-cache accounting; beyond it, layers are on the CPU.
|
||||
CPU_OFFLOAD_TOLERANCE = 0.02
|
||||
# Only intervene when ComfyUI is actually holding enough VRAM to be the cause.
|
||||
RECLAIM_MIN_COMFY_BYTES = 512 * 1024 ** 2
|
||||
|
||||
# Ollama's response when a model will not fit. Which of the two failure modes you get
|
||||
# depends on configuration: with n_gpu_layers left to Ollama it spills layers to the CPU
|
||||
# and reports size_vram < size; with n_gpu_layers pinned (99 on this box) it refuses and
|
||||
# returns a hard CUDA OOM instead. Both are handled -- the spill by
|
||||
# AutoArbitrator._check_ollama_starved, the hard failure by the retry below.
|
||||
OOM_SIGNATURES = ("out of memory", "cudamalloc", "unable to allocate",
|
||||
"failed to allocate", "cuda error")
|
||||
|
||||
|
||||
def looks_like_vram_oom(text: str) -> bool:
|
||||
low = (text or "").lower()
|
||||
return any(sig in low for sig in OOM_SIGNATURES)
|
||||
YIELD_CONFIRM_POLL_S = 0.02
|
||||
YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released"
|
||||
|
||||
@@ -360,7 +379,13 @@ async def get_ollama_live_state() -> Dict[str, Any]:
|
||||
"active_model_vram_gb": 0.0,
|
||||
"active_context": 0,
|
||||
"expires_at": None,
|
||||
"installed_models": []
|
||||
"installed_models": [],
|
||||
# Ollama silently spills layers to CPU when VRAM is short. size_vram < size is the
|
||||
# only externally visible sign, and the cost is roughly an order of magnitude in
|
||||
# decode speed, so it is worth surfacing loudly.
|
||||
"gpu_fraction": 1.0,
|
||||
"cpu_offload_pct": 0.0,
|
||||
"partially_offloaded": False,
|
||||
}
|
||||
try:
|
||||
client = _client(OLLAMA_API_BASE, 3.0)
|
||||
@@ -694,8 +719,13 @@ def classify_load(size_bytes: int, load_duration_ms: float) -> Dict[str, Any]:
|
||||
return {"cache_status": status, "load_gbps": round(gbps, 2), "is_ram_hit": gbps >= RAM_HIT_GBPS}
|
||||
|
||||
|
||||
async def switch_ollama_model(target_model: str, keep_alive: str = "30m") -> Dict[str, Any]:
|
||||
"""High-speed hot-swap to target Ollama model, tracking swap metrics."""
|
||||
async def switch_ollama_model(target_model: str, keep_alive: str = "30m",
|
||||
_retrying: bool = False) -> Dict[str, Any]:
|
||||
"""High-speed hot-swap to target Ollama model, tracking swap metrics.
|
||||
|
||||
If the load fails because the model will not fit, reclaims VRAM from an idle ComfyUI
|
||||
and retries once. `_retrying` guards against recursing more than one level.
|
||||
"""
|
||||
t0 = time.perf_counter()
|
||||
cur_state = await get_ollama_live_state()
|
||||
prev_model = cur_state.get("active_model_name") or "None"
|
||||
@@ -745,8 +775,38 @@ async def switch_ollama_model(target_model: str, keep_alive: str = "30m") -> Dic
|
||||
"is_ram_hit": cls["is_ram_hit"],
|
||||
"response": data.get("response", ""),
|
||||
}
|
||||
return {"success": False, "error": f"HTTP {resp.status_code}: {resp.text}",
|
||||
"duration_ms": total_duration_ms}
|
||||
# A model that will not fit is the exact contention this service exists to
|
||||
# resolve. Rather than handing the caller a CUDA OOM, take the VRAM back from an
|
||||
# idle ComfyUI and try once more.
|
||||
body = resp.text
|
||||
if looks_like_vram_oom(body) and not _retrying:
|
||||
snap = get_process_vram_bytes()
|
||||
if snap["comfyui_bytes"] >= RECLAIM_MIN_COMFY_BYTES:
|
||||
logger.warning(
|
||||
f"Ollama could not fit '{target_model}' with ComfyUI holding "
|
||||
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming and retrying")
|
||||
purge = await instant_free_comfyui_vram()
|
||||
arbitrator.stats["reclaims_for_ollama"] += 1
|
||||
arbitrator.last_action = (
|
||||
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI so "
|
||||
f"'{target_model}' could load")
|
||||
_record({
|
||||
"event_type": "VRAM Reclaim for Ollama",
|
||||
"source": "ComfyUI Pipeline",
|
||||
"target": target_model,
|
||||
"duration_ms": purge.get("duration_ms"),
|
||||
"cache_status": "Reclaimed",
|
||||
"detail": f"Ollama OOM: {body[:160]}",
|
||||
})
|
||||
await asyncio.sleep(0.3)
|
||||
retry = await switch_ollama_model(target_model, keep_alive, _retrying=True)
|
||||
retry["reclaimed_from_comfyui_gb"] = round(
|
||||
snap["comfyui_bytes"] / (1024**3), 2)
|
||||
retry["first_attempt_error"] = "CUDA OOM; retried after reclaiming VRAM"
|
||||
return retry
|
||||
return {"success": False, "error": f"HTTP {resp.status_code}: {body}",
|
||||
"duration_ms": total_duration_ms,
|
||||
"vram_oom": looks_like_vram_oom(body)}
|
||||
except Exception as e:
|
||||
return {"success": False, "error": str(e),
|
||||
"duration_ms": round((time.perf_counter() - t0) * 1000, 2)}
|
||||
@@ -795,6 +855,7 @@ class AutoArbitrator:
|
||||
# again every second just blocks the loop repeatedly for no benefit.
|
||||
self._yield_backoff_until: Dict[str, float] = {}
|
||||
self._yield_busy_streak: Dict[str, int] = {}
|
||||
self.last_reclaim_time = 0.0
|
||||
self.stats = {
|
||||
"yields": 0, # release confirmed
|
||||
"yield_deferred_busy": 0, # model mid-generation; unload queued behind it
|
||||
@@ -802,6 +863,7 @@ class AutoArbitrator:
|
||||
"deferred_releases": 0, # queued unloads that later landed
|
||||
"purges": 0,
|
||||
"deferred_purges": 0,
|
||||
"reclaims_for_ollama": 0, # ComfyUI purged because the LLM was spilling to CPU
|
||||
}
|
||||
|
||||
async def start(self):
|
||||
@@ -984,6 +1046,57 @@ class AutoArbitrator:
|
||||
await asyncio.sleep(backoff)
|
||||
backoff = min(backoff * 1.5, 15.0)
|
||||
|
||||
RECLAIM_COOLDOWN_S = 30.0
|
||||
|
||||
async def _check_ollama_starved(self) -> None:
|
||||
"""The other direction: rescue an LLM that ComfyUI has squeezed onto the CPU.
|
||||
|
||||
Yielding Ollama for ComfyUI was automatic; the reverse never was, despite the
|
||||
README calling the arbitration bidirectional. When Ollama cannot fit a model it
|
||||
does not fail, it silently places layers on the CPU and runs about an order of
|
||||
magnitude slower -- so this is the failure mode a user is least likely to notice
|
||||
and most likely to feel.
|
||||
|
||||
If the LLM is spilling while ComfyUI sits idle holding VRAM, ComfyUI's cached
|
||||
checkpoints are the thing to give up.
|
||||
"""
|
||||
now = time.time()
|
||||
if self.comfy_was_active or (now - self.last_reclaim_time) < self.RECLAIM_COOLDOWN_S:
|
||||
return
|
||||
|
||||
ollama = await get_ollama_live_state()
|
||||
if not ollama.get("partially_offloaded"):
|
||||
return
|
||||
|
||||
snap = get_process_vram_bytes()
|
||||
if snap["comfyui_bytes"] < RECLAIM_MIN_COMFY_BYTES:
|
||||
return # ComfyUI is not the one holding the memory; nothing we can do here
|
||||
|
||||
self.last_reclaim_time = now
|
||||
model = ollama.get("active_model_name")
|
||||
offload = ollama.get("cpu_offload_pct")
|
||||
logger.warning(f"⚠ '{model}' is {offload}% on CPU while ComfyUI holds "
|
||||
f"{round(snap['comfyui_bytes'] / (1024**3), 2)} GB — reclaiming for the LLM")
|
||||
await self._purge_comfy_now(f"LLM spilling {offload}% to CPU")
|
||||
self.stats["reclaims_for_ollama"] += 1
|
||||
|
||||
# Freeing VRAM does not move layers back; only a reload re-places the model. Do
|
||||
# that only when the model is idle, never mid-generation.
|
||||
after = get_process_vram_bytes()
|
||||
if after.get("gpu_util_pct", 0) < BUSY_UTIL_PCT and model:
|
||||
logger.info(f"Reloading '{model}' to place it fully on the GPU...")
|
||||
await instant_free_ollama_vram(model, confirm=True)
|
||||
res = await switch_ollama_model(model, keep_alive="30m")
|
||||
recheck = await get_ollama_live_state()
|
||||
self.last_action = (
|
||||
f"Reclaimed {round(snap['comfyui_bytes'] / (1024**3), 2)}GB from ComfyUI and "
|
||||
f"reloaded '{model}' — now {round(recheck.get('gpu_fraction', 0) * 100)}% on GPU"
|
||||
if res.get("success") else
|
||||
f"Reclaimed VRAM from ComfyUI but reloading '{model}' failed: {res.get('error')}")
|
||||
else:
|
||||
self.last_action = (f"Reclaimed VRAM from ComfyUI; '{model}' is busy, so it will "
|
||||
f"stay partly on CPU until its next load")
|
||||
|
||||
async def _poll_watchdog(self):
|
||||
"""Fallback for when the WebSocket is down. One cheap /queue call, 1 Hz.
|
||||
|
||||
@@ -1002,6 +1115,8 @@ class AutoArbitrator:
|
||||
await self.trigger_comfy_priority("Watchdog saw an active queue")
|
||||
elif self.comfy_was_active:
|
||||
await self.trigger_comfy_completed()
|
||||
else:
|
||||
await self._check_ollama_starved()
|
||||
except Exception:
|
||||
pass
|
||||
await asyncio.sleep(interval)
|
||||
|
||||
Reference in New Issue
Block a user