Unload every resident Ollama model on yield; widen the confirm window

The confirm barrier surfaced two yields that left 8.2 GB allocated after 3s. Two
separate issues behind that class of failure:

- The yield only unloaded loaded_models[0]. Ollama can hold several models resident
  (OLLAMA_MAX_LOADED_MODELS), so releasing the first left the rest allocated. It now
  unloads every resident model concurrently. This box runs with the limit at 1, so
  the change is defensive here rather than a fix for the observed case.
- The observed 8.2 GB stalls happened while ComfyUI was starting and Ollama had a
  generation in flight; Ollama will not unload mid-request. A 3s ceiling reported a
  timeout for a model that was simply busy finishing. Raised to 10s -- waiting longer
  is the safer failure mode, since the alternative is diffusion allocating into VRAM
  that is still occupied.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-08-28 09:14:58 -07:00
parent 63297dc49d
commit 27dcfd0ada

View File

@@ -35,7 +35,12 @@ RAM_HIT_GBPS = 5.0
PARTIAL_HIT_GBPS = 1.5 PARTIAL_HIT_GBPS = 1.5
# How long Ollama's VRAM may take to actually drain before we stop waiting. # How long Ollama's VRAM may take to actually drain before we stop waiting.
YIELD_CONFIRM_TIMEOUT_S = 3.0 # Ollama will not unload a model while a generation is in flight, so a short ceiling
# reports a timeout for what is really just a busy model finishing its request. Observed
# here: two yields hit the old 3 s limit with 8.2 GB still held while ComfyUI was starting.
# Waiting longer is the safer failure mode -- the alternative is diffusion allocating into
# VRAM that is still occupied.
YIELD_CONFIRM_TIMEOUT_S = 10.0
YIELD_CONFIRM_POLL_S = 0.02 YIELD_CONFIRM_POLL_S = 0.02
YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released" YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released"
@@ -445,18 +450,29 @@ async def instant_free_ollama_vram(model_name: Optional[str] = None,
the HTTP POST took. the HTTP POST took.
""" """
t0 = time.perf_counter() t0 = time.perf_counter()
if not model_name: if model_name:
targets = [model_name]
else:
# Unload *every* resident model, not just loaded_models[0]. Ollama will happily
# keep several models in VRAM at once; releasing only the first left the rest
# allocated, which the confirm barrier caught as "still holding 8.2 GB after 3s".
ollama_state = await get_ollama_live_state() ollama_state = await get_ollama_live_state()
model_name = ollama_state.get("active_model_name") targets = [m.get("name") for m in ollama_state.get("loaded_models", []) if m.get("name")]
if not targets and ollama_state.get("active_model_name"):
targets = [ollama_state["active_model_name"]]
if not model_name: if not targets:
return {"success": True, "message": "No active Ollama model in VRAM", return {"success": True, "message": "No active Ollama model in VRAM",
"duration_ms": 0, "confirmed": True} "duration_ms": 0, "confirmed": True}
model_name = targets[0] if len(targets) == 1 else f"{len(targets)} models"
baseline = get_process_vram_bytes()["ollama_bytes"] baseline = get_process_vram_bytes()["ollama_bytes"]
try: try:
client = _client(OLLAMA_API_BASE, 5.0) client = _client(OLLAMA_API_BASE, 5.0)
await client.post("/api/generate", json={"model": model_name, "keep_alive": 0}) await asyncio.gather(*[
client.post("/api/generate", json={"model": t, "keep_alive": 0})
for t in targets
], return_exceptions=True)
request_ms = round((time.perf_counter() - t0) * 1000, 2) request_ms = round((time.perf_counter() - t0) * 1000, 2)
barrier = {"confirmed": None, "confirm_ms": 0.0, "residual_bytes": baseline} barrier = {"confirmed": None, "confirm_ms": 0.0, "residual_bytes": baseline}
@@ -468,7 +484,7 @@ async def instant_free_ollama_vram(model_name: Optional[str] = None,
_record({ _record({
"event_type": "Ollama VRAM Yield", "event_type": "Ollama VRAM Yield",
"source": model_name, "source": ", ".join(targets)[:200],
"target": "VRAM 0MB (Kept in RAM)", "target": "VRAM 0MB (Kept in RAM)",
"duration_ms": duration_ms, "duration_ms": duration_ms,
"yield_confirm_ms": barrier.get("confirm_ms"), "yield_confirm_ms": barrier.get("confirm_ms"),
@@ -478,6 +494,7 @@ async def instant_free_ollama_vram(model_name: Optional[str] = None,
return { return {
"success": True, "success": True,
"model": model_name, "model": model_name,
"models_unloaded": targets,
"duration_ms": duration_ms, "duration_ms": duration_ms,
"request_ms": request_ms, "request_ms": request_ms,
"confirm_ms": barrier.get("confirm_ms"), "confirm_ms": barrier.get("confirm_ms"),