Treat a mid-generation LLM as busy, not as a failed yield
The persisted counters showed 19 timeouts in 20 yields. All 19 were one model, ornith-1.5:9b-cron, in two bursts at 06:25 and 06:33. Telemetry for that window shows the GPU pinned at 96-97% with Ollama holding 14.92 GB throughout: the model was mid-generation. Ollama will not unload a model that is inferencing, so every request failed, and with a 1s trigger debounce against a 10s blocking wait the arbitrator simply asked again, four times per burst, blocking the loop for 40s. Ollama's behaviour is correct. Ours was wrong in three ways. Busy is now a distinct outcome. _await_vram_release returns "released", "busy" or "stuck": VRAM that has not moved while the GPU is pinned means a generation is in flight, which is not a failure. With OLLAMA_NUM_PARALLEL=1 our keep_alive:0 request queues behind the running one and applies the moment it finishes, so the correct response is to stop waiting, not to retry. Only "stuck" -- VRAM held with an idle GPU -- is a real fault. The wait is short again (2s, from 10s) because blocking helps nobody: ComfyUI is not gated on our return value, and every blocked second stalls the watchdog and profile switching. Callers who genuinely want to wait out an inference can pass wait_for_generation=true. The unload POST itself now gets a 120s client timeout, since a 5s one could drop the connection before Ollama ever processed a request queued behind a long generation, losing the unload entirely. A busy model gets per-model backoff (5s, 15s, 30s, 60s) instead of being asked again every second, and a detached watcher confirms and logs the release when the generation ends, so the event log tells the whole story rather than stopping at "deferred". Measured: a mid-generation yield now returns busy in 610ms instead of blocking 10s, and the queued unload lands on its own 3s later when the generation completes. Counters are honest: yields (released), yield_deferred_busy, deferred_releases, yield_stalled. The old yield_timeouts conflated a healthy cron job with a fault and implied a 95% failure rate. Also adds a VRAM Arbitration panel to the dashboard. The arbitrator is the core of this application and its state was not displayed anywhere -- there was no way to see whether handoffs were working, which is why this went unnoticed until the persisted counters were read by hand. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
17
server.py
17
server.py
@@ -336,9 +336,20 @@ async def api_switch_model(req: SwitchRequest):
|
|||||||
return res
|
return res
|
||||||
|
|
||||||
@app.post("/api/free-vram", summary="Soft-Yield Ollama VRAM", tags=["Orchestration"])
|
@app.post("/api/free-vram", summary="Soft-Yield Ollama VRAM", tags=["Orchestration"])
|
||||||
async def api_free_vram(confirm: bool = Query(True, description="Wait for the driver to actually release the allocation")):
|
async def api_free_vram(
|
||||||
"""Yield Ollama's VRAM and wait for the release to be confirmed by NVML."""
|
confirm: bool = Query(True, description="Wait for the driver to actually release the allocation"),
|
||||||
return await vram_arbitrator.instant_free_ollama_vram(confirm=confirm)
|
wait_for_generation: bool = Query(False, description="Block until an in-flight generation finishes, instead of returning 'busy'"),
|
||||||
|
):
|
||||||
|
"""Yield Ollama's VRAM and wait for NVML to confirm the release.
|
||||||
|
|
||||||
|
Returns `outcome`: "released", "busy" (model is mid-generation; the unload is queued
|
||||||
|
and will apply when it finishes) or "stuck" (VRAM held with an idle GPU).
|
||||||
|
"""
|
||||||
|
return await vram_arbitrator.instant_free_ollama_vram(
|
||||||
|
confirm=confirm,
|
||||||
|
timeout_s=(vram_arbitrator.YIELD_CONFIRM_TIMEOUT_BLOCKING_S
|
||||||
|
if wait_for_generation else None),
|
||||||
|
)
|
||||||
|
|
||||||
@app.post("/api/comfy-free", summary="Purge ComfyUI VRAM Cache", tags=["Orchestration"])
|
@app.post("/api/comfy-free", summary="Purge ComfyUI VRAM Cache", tags=["Orchestration"])
|
||||||
async def api_comfy_free():
|
async def api_comfy_free():
|
||||||
|
|||||||
@@ -33,8 +33,9 @@ function initSSE() {
|
|||||||
function updateDashboard(data) {
|
function updateDashboard(data) {
|
||||||
if (!data) return;
|
if (!data) return;
|
||||||
|
|
||||||
// Governor state rides along in the shared snapshot — no extra polling needed.
|
// Governor and arbitration state ride along in the shared snapshot.
|
||||||
if (data.governor) renderGovernor(data.governor);
|
if (data.governor) renderGovernor(data.governor);
|
||||||
|
if (data.arbitrator) renderArbitrator(data.arbitrator);
|
||||||
|
|
||||||
// 1. GPU VRAM Stats
|
// 1. GPU VRAM Stats
|
||||||
const gpu = data.gpu || {};
|
const gpu = data.gpu || {};
|
||||||
@@ -881,3 +882,36 @@ document.addEventListener('DOMContentLoaded', () => {
|
|||||||
if (d.last_result) renderSweep(d.last_result);
|
if (d.last_result) renderSweep(d.last_result);
|
||||||
}).catch(() => {});
|
}).catch(() => {});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------- arbitration
|
||||||
|
|
||||||
|
function renderArbitrator(arb) {
|
||||||
|
const el = (id) => document.getElementById(id);
|
||||||
|
if (!el('arb-action')) return;
|
||||||
|
const c = arb.counters || {};
|
||||||
|
|
||||||
|
el('arb-action').textContent = arb.last_action || 'Idle';
|
||||||
|
el('arb-yields').textContent = c.yields ?? 0;
|
||||||
|
el('arb-busy').textContent = c.yield_deferred_busy ?? 0;
|
||||||
|
el('arb-later').textContent = c.deferred_releases ?? 0;
|
||||||
|
el('arb-stalled').textContent = c.yield_stalled ?? 0;
|
||||||
|
el('arb-purges').textContent = c.purges ?? 0;
|
||||||
|
el('arb-defpurge').textContent = c.deferred_purges ?? 0;
|
||||||
|
|
||||||
|
const ws = el('arb-ws');
|
||||||
|
ws.textContent = arb.connected_ws ? 'ComfyUI WS live' : 'WS down — polling';
|
||||||
|
ws.className = 'text-xs font-mono ' + (arb.connected_ws ? 'text-emerald-400' : 'text-amber-400');
|
||||||
|
|
||||||
|
// Show why we are holding off, and the idle countdown before ComfyUI is purged.
|
||||||
|
const parts = [];
|
||||||
|
const backoff = arb.yield_backoff || {};
|
||||||
|
for (const [model, secs] of Object.entries(backoff)) {
|
||||||
|
parts.push(`waiting ${secs}s before asking '${model}' again`);
|
||||||
|
}
|
||||||
|
if (arb.pending_purge && arb.comfy_idle_s != null) {
|
||||||
|
const left = Math.max((arb.idle_purge_after_s || 0) - arb.comfy_idle_s, 0).toFixed(0);
|
||||||
|
parts.push(`ComfyUI idle ${arb.comfy_idle_s}s — holding its checkpoints ${left}s longer`);
|
||||||
|
}
|
||||||
|
el('arb-backoff').textContent = parts.join(' · ');
|
||||||
|
}
|
||||||
|
|||||||
@@ -668,6 +668,58 @@
|
|||||||
<!-- ============ NEXT-LEVEL PANELS: governor / residency / analytics / autotune ============ -->
|
<!-- ============ NEXT-LEVEL PANELS: governor / residency / analytics / autotune ============ -->
|
||||||
<div class="grid grid-cols-1 xl:grid-cols-2 gap-5 mt-5">
|
<div class="grid grid-cols-1 xl:grid-cols-2 gap-5 mt-5">
|
||||||
|
|
||||||
|
|
||||||
|
<!-- VRAM Arbitration — the core handoff, previously invisible -->
|
||||||
|
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5 xl:col-span-2">
|
||||||
|
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
||||||
|
<div class="flex items-center space-x-2">
|
||||||
|
<div class="p-2 rounded-lg bg-emerald-950/80 border border-emerald-800 text-emerald-400">
|
||||||
|
<i class="fa-solid fa-right-left text-sm"></i>
|
||||||
|
</div>
|
||||||
|
<div>
|
||||||
|
<h3 class="font-bold text-slate-100 text-sm">VRAM Arbitration</h3>
|
||||||
|
<p class="text-xs text-slate-400">Who holds the GPU, and how handoffs are going</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<span id="arb-ws" class="text-xs font-mono text-slate-500">—</span>
|
||||||
|
</div>
|
||||||
|
<div class="mt-4">
|
||||||
|
<div id="arb-action" class="text-sm text-slate-200 bg-slate-950/60 border border-slate-800 rounded-lg px-3 py-2 mb-3 font-mono">Idle</div>
|
||||||
|
<div id="arb-backoff" class="text-[11px] font-mono text-amber-400 mb-3"></div>
|
||||||
|
<div class="grid grid-cols-3 sm:grid-cols-6 gap-2 text-center">
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-yields" class="text-lg font-bold text-emerald-400">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Released</div>
|
||||||
|
</div>
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-busy" class="text-lg font-bold text-cyan-400">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Deferred<br>(busy)</div>
|
||||||
|
</div>
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-later" class="text-lg font-bold text-cyan-400">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Landed<br>later</div>
|
||||||
|
</div>
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-stalled" class="text-lg font-bold text-rose-400">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Stalled</div>
|
||||||
|
</div>
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-purges" class="text-lg font-bold text-fuchsia-400">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Comfy<br>purges</div>
|
||||||
|
</div>
|
||||||
|
<div class="bg-slate-950/60 rounded-lg p-2 border border-slate-800">
|
||||||
|
<div id="arb-defpurge" class="text-lg font-bold text-slate-300">0</div>
|
||||||
|
<div class="text-[10px] text-slate-500 uppercase leading-tight">Purges<br>deferred</div>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<p class="text-[11px] text-slate-500 mt-3">
|
||||||
|
<span class="text-cyan-400">Deferred</span> is healthy — an LLM mid-generation cannot unload, so the
|
||||||
|
request queues and applies the moment it finishes. Only <span class="text-rose-400">stalled</span>
|
||||||
|
(VRAM held while the GPU sits idle) indicates a real problem.
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
<!-- Thermal Governor -->
|
<!-- Thermal Governor -->
|
||||||
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5">
|
<div class="bg-slate-900/80 border border-slate-800 rounded-2xl p-5">
|
||||||
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
<div class="flex items-center justify-between pb-3 border-b border-slate-800">
|
||||||
|
|||||||
@@ -44,13 +44,23 @@ SWITCH_HISTORY = deque(maxlen=50)
|
|||||||
RAM_HIT_GBPS = 2.0
|
RAM_HIT_GBPS = 2.0
|
||||||
PARTIAL_HIT_GBPS = 0.8
|
PARTIAL_HIT_GBPS = 0.8
|
||||||
|
|
||||||
# How long Ollama's VRAM may take to actually drain before we stop waiting.
|
# How long to wait for Ollama's VRAM to actually drain.
|
||||||
# Ollama will not unload a model while a generation is in flight, so a short ceiling
|
#
|
||||||
# reports a timeout for what is really just a busy model finishing its request. Observed
|
# Ollama will not unload a model mid-generation. With OLLAMA_NUM_PARALLEL=1 our
|
||||||
# here: two yields hit the old 3 s limit with 8.2 GB still held while ComfyUI was starting.
|
# keep_alive:0 request queues behind the running one and takes effect the moment it
|
||||||
# Waiting longer is the safer failure mode -- the alternative is diffusion allocating into
|
# finishes, so a model that is busy is not failing -- it is finishing, and it will
|
||||||
# VRAM that is still occupied.
|
# release on its own. Blocking the arbitrator for the length of someone's inference
|
||||||
YIELD_CONFIRM_TIMEOUT_S = 10.0
|
# helps nobody: ComfyUI is not gated on our return value, and every second spent
|
||||||
|
# blocked is a second the watchdog and profile switching are stalled.
|
||||||
|
#
|
||||||
|
# So: wait briefly for the common case (an idle model releases in 40-110 ms here),
|
||||||
|
# then classify. A caller who genuinely wants to block can ask for a longer wait.
|
||||||
|
YIELD_CONFIRM_TIMEOUT_S = 2.0
|
||||||
|
YIELD_CONFIRM_TIMEOUT_BLOCKING_S = 30.0
|
||||||
|
|
||||||
|
# A model still holding VRAM while the GPU is pinned is generating, not wedged.
|
||||||
|
BUSY_UTIL_PCT = 50
|
||||||
|
BUSY_PROBE_S = 0.6
|
||||||
YIELD_CONFIRM_POLL_S = 0.02
|
YIELD_CONFIRM_POLL_S = 0.02
|
||||||
YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released"
|
YIELD_RESIDUAL_BYTES = 256 * 1024 ** 2 # treat <256 MB as "released"
|
||||||
|
|
||||||
@@ -105,12 +115,17 @@ def get_process_vram_bytes() -> Dict[str, int]:
|
|||||||
Deliberately avoids psutil lookups: this runs every 20 ms while we wait for VRAM
|
Deliberately avoids psutil lookups: this runs every 20 ms while we wait for VRAM
|
||||||
to actually drain.
|
to actually drain.
|
||||||
"""
|
"""
|
||||||
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0}
|
out = {"ollama_bytes": 0, "comfyui_bytes": 0, "other_bytes": 0, "free_bytes": 0,
|
||||||
|
"gpu_util_pct": 0}
|
||||||
if not NVML_AVAILABLE:
|
if not NVML_AVAILABLE:
|
||||||
return out
|
return out
|
||||||
try:
|
try:
|
||||||
handle = pynvml.nvmlDeviceGetHandleByIndex(0)
|
handle = pynvml.nvmlDeviceGetHandleByIndex(0)
|
||||||
out["free_bytes"] = pynvml.nvmlDeviceGetMemoryInfo(handle).free
|
out["free_bytes"] = pynvml.nvmlDeviceGetMemoryInfo(handle).free
|
||||||
|
try:
|
||||||
|
out["gpu_util_pct"] = pynvml.nvmlDeviceGetUtilizationRates(handle).gpu
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
procs = list(pynvml.nvmlDeviceGetComputeRunningProcesses(handle))
|
procs = list(pynvml.nvmlDeviceGetComputeRunningProcesses(handle))
|
||||||
try:
|
try:
|
||||||
procs += list(pynvml.nvmlDeviceGetGraphicsRunningProcesses(handle))
|
procs += list(pynvml.nvmlDeviceGetGraphicsRunningProcesses(handle))
|
||||||
@@ -415,35 +430,106 @@ async def get_comfyui_live_state() -> Dict[str, Any]:
|
|||||||
|
|
||||||
async def _await_vram_release(baseline_bytes: int,
|
async def _await_vram_release(baseline_bytes: int,
|
||||||
timeout_s: float = YIELD_CONFIRM_TIMEOUT_S) -> Dict[str, Any]:
|
timeout_s: float = YIELD_CONFIRM_TIMEOUT_S) -> Dict[str, Any]:
|
||||||
"""Block until Ollama's VRAM has actually drained, or we give up.
|
"""Wait for Ollama's VRAM to drain, distinguishing "busy" from "stuck".
|
||||||
|
|
||||||
Posting keep_alive:0 only *asks* Ollama to unload; the driver frees the allocation
|
Posting keep_alive:0 only *asks* Ollama to unload; the driver frees the allocation
|
||||||
some milliseconds later. Returning before that happens is how ComfyUI ends up
|
some milliseconds later, and returning before that happens is how ComfyUI ends up
|
||||||
allocating into VRAM that is still occupied, which shows up as a CUDA OOM mid-graph.
|
allocating into VRAM that is still occupied.
|
||||||
|
|
||||||
|
But there is a second case the first version of this got wrong. If the model is
|
||||||
|
mid-generation it cannot unload at all, and reporting that as a timeout made a
|
||||||
|
perfectly healthy cron job look like a 95% failure rate. When the VRAM has not
|
||||||
|
moved and the GPU is pinned, the model is working; the queued unload will fire when
|
||||||
|
it finishes. That is `busy`, not a failure.
|
||||||
|
|
||||||
|
Returns an `outcome` of "released", "busy" or "stuck".
|
||||||
"""
|
"""
|
||||||
t0 = time.perf_counter()
|
t0 = time.perf_counter()
|
||||||
last = baseline_bytes
|
peak_util = 0
|
||||||
while True:
|
while True:
|
||||||
snap = get_process_vram_bytes()
|
snap = get_process_vram_bytes()
|
||||||
last = snap["ollama_bytes"]
|
last = snap["ollama_bytes"]
|
||||||
|
peak_util = max(peak_util, snap.get("gpu_util_pct", 0))
|
||||||
|
elapsed = time.perf_counter() - t0
|
||||||
|
|
||||||
if last <= YIELD_RESIDUAL_BYTES:
|
if last <= YIELD_RESIDUAL_BYTES:
|
||||||
return {
|
return {
|
||||||
|
"outcome": "released",
|
||||||
"confirmed": True,
|
"confirmed": True,
|
||||||
"confirm_ms": round((time.perf_counter() - t0) * 1000, 2),
|
"confirm_ms": round(elapsed * 1000, 2),
|
||||||
"residual_bytes": last,
|
"residual_bytes": last,
|
||||||
"free_bytes": snap["free_bytes"],
|
"free_bytes": snap["free_bytes"],
|
||||||
|
"gpu_util_pct": snap.get("gpu_util_pct", 0),
|
||||||
}
|
}
|
||||||
if (time.perf_counter() - t0) >= timeout_s:
|
|
||||||
|
# Unmoved VRAM plus a pinned GPU means a generation is in flight.
|
||||||
|
busy = (elapsed >= BUSY_PROBE_S
|
||||||
|
and last >= baseline_bytes - YIELD_RESIDUAL_BYTES
|
||||||
|
and peak_util >= BUSY_UTIL_PCT)
|
||||||
|
|
||||||
|
if busy or elapsed >= timeout_s:
|
||||||
|
outcome = "busy" if busy else "stuck"
|
||||||
return {
|
return {
|
||||||
|
"outcome": outcome,
|
||||||
"confirmed": False,
|
"confirmed": False,
|
||||||
"confirm_ms": round((time.perf_counter() - t0) * 1000, 2),
|
"confirm_ms": round(elapsed * 1000, 2),
|
||||||
"residual_bytes": last,
|
"residual_bytes": last,
|
||||||
"free_bytes": snap["free_bytes"],
|
"free_bytes": snap["free_bytes"],
|
||||||
"error": f"Ollama still holding {round(last / (1024**3), 2)} GB after {timeout_s}s",
|
"gpu_util_pct": snap.get("gpu_util_pct", 0),
|
||||||
|
"peak_util_pct": peak_util,
|
||||||
|
"error": (
|
||||||
|
f"Ollama is mid-generation ({peak_util}% GPU, "
|
||||||
|
f"{round(last / (1024**3), 2)} GB held); the queued unload will apply "
|
||||||
|
f"when it finishes"
|
||||||
|
if outcome == "busy" else
|
||||||
|
f"Ollama still holding {round(last / (1024**3), 2)} GB after "
|
||||||
|
f"{timeout_s}s with the GPU idle"
|
||||||
|
),
|
||||||
}
|
}
|
||||||
await asyncio.sleep(YIELD_CONFIRM_POLL_S)
|
await asyncio.sleep(YIELD_CONFIRM_POLL_S)
|
||||||
|
|
||||||
|
|
||||||
|
# Detached tasks need a strong reference or the loop may garbage-collect them mid-flight.
|
||||||
|
_DETACHED: set = set()
|
||||||
|
|
||||||
|
|
||||||
|
def _spawn_detached(coro) -> None:
|
||||||
|
task = asyncio.ensure_future(coro)
|
||||||
|
_DETACHED.add(task)
|
||||||
|
task.add_done_callback(_DETACHED.discard)
|
||||||
|
|
||||||
|
|
||||||
|
async def _confirm_release_later(targets: List[str], baseline_bytes: int,
|
||||||
|
max_wait_s: float = 900.0) -> None:
|
||||||
|
"""Watch for a queued unload to land after the in-flight generation finishes.
|
||||||
|
|
||||||
|
Runs detached so the caller is never held for the length of an inference. Logs the
|
||||||
|
eventual release so the event log tells the whole story rather than stopping at
|
||||||
|
"deferred".
|
||||||
|
"""
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
while (time.perf_counter() - t0) < max_wait_s:
|
||||||
|
await asyncio.sleep(0.5)
|
||||||
|
snap = get_process_vram_bytes()
|
||||||
|
if snap["ollama_bytes"] <= YIELD_RESIDUAL_BYTES:
|
||||||
|
waited_ms = round((time.perf_counter() - t0) * 1000, 2)
|
||||||
|
_record({
|
||||||
|
"event_type": "Ollama VRAM Yield",
|
||||||
|
"source": ", ".join(targets)[:200],
|
||||||
|
"target": "VRAM 0MB (Kept in RAM)",
|
||||||
|
"duration_ms": waited_ms,
|
||||||
|
"yield_confirm_ms": waited_ms,
|
||||||
|
"cache_status": "RAM-Cached",
|
||||||
|
"detail": "released after the in-flight generation completed",
|
||||||
|
})
|
||||||
|
arbitrator.note_deferred_release(waited_ms)
|
||||||
|
logger.info(f"Deferred VRAM yield completed after {round(waited_ms / 1000, 1)}s "
|
||||||
|
f"({round(snap['free_bytes'] / (1024**3), 2)} GB free)")
|
||||||
|
return
|
||||||
|
logger.warning("Deferred VRAM yield never landed within "
|
||||||
|
f"{max_wait_s}s for {', '.join(targets)}")
|
||||||
|
|
||||||
|
|
||||||
def _record(event: Dict[str, Any]) -> None:
|
def _record(event: Dict[str, Any]) -> None:
|
||||||
"""Push an event to both the in-memory ring and the durable store."""
|
"""Push an event to both the in-memory ring and the durable store."""
|
||||||
event.setdefault("ts", time.time())
|
event.setdefault("ts", time.time())
|
||||||
@@ -453,7 +539,8 @@ def _record(event: Dict[str, Any]) -> None:
|
|||||||
|
|
||||||
|
|
||||||
async def instant_free_ollama_vram(model_name: Optional[str] = None,
|
async def instant_free_ollama_vram(model_name: Optional[str] = None,
|
||||||
confirm: bool = True) -> Dict[str, Any]:
|
confirm: bool = True,
|
||||||
|
timeout_s: Optional[float] = None) -> Dict[str, Any]:
|
||||||
"""Yield Ollama's VRAM and wait for the driver to actually release it.
|
"""Yield Ollama's VRAM and wait for the driver to actually release it.
|
||||||
|
|
||||||
The returned duration_ms is now the real end-to-end release time, not just how long
|
The returned duration_ms is now the real end-to-end release time, not just how long
|
||||||
@@ -478,31 +565,51 @@ async def instant_free_ollama_vram(model_name: Optional[str] = None,
|
|||||||
model_name = targets[0] if len(targets) == 1 else f"{len(targets)} models"
|
model_name = targets[0] if len(targets) == 1 else f"{len(targets)} models"
|
||||||
baseline = get_process_vram_bytes()["ollama_bytes"]
|
baseline = get_process_vram_bytes()["ollama_bytes"]
|
||||||
try:
|
try:
|
||||||
client = _client(OLLAMA_API_BASE, 5.0)
|
# Generous client timeout: with OLLAMA_NUM_PARALLEL=1 this request queues behind
|
||||||
await asyncio.gather(*[
|
# any running generation, and a short timeout would drop the connection before
|
||||||
|
# Ollama ever processed the unload -- losing it entirely.
|
||||||
|
client = _client(OLLAMA_API_BASE, 120.0)
|
||||||
|
unload_calls = [
|
||||||
client.post("/api/generate", json={"model": t, "keep_alive": 0})
|
client.post("/api/generate", json={"model": t, "keep_alive": 0})
|
||||||
for t in targets
|
for t in targets
|
||||||
], return_exceptions=True)
|
]
|
||||||
|
# Do not await the queued unloads; a busy model would block us for the length of
|
||||||
|
# its inference. They are fire-and-confirm: the barrier below watches the VRAM.
|
||||||
|
_spawn_detached(asyncio.gather(*unload_calls, return_exceptions=True))
|
||||||
request_ms = round((time.perf_counter() - t0) * 1000, 2)
|
request_ms = round((time.perf_counter() - t0) * 1000, 2)
|
||||||
|
|
||||||
barrier = {"confirmed": None, "confirm_ms": 0.0, "residual_bytes": baseline}
|
barrier: Dict[str, Any] = {"outcome": "unconfirmed", "confirmed": None,
|
||||||
|
"confirm_ms": 0.0, "residual_bytes": baseline}
|
||||||
if confirm:
|
if confirm:
|
||||||
barrier = await _await_vram_release(baseline)
|
barrier = await _await_vram_release(
|
||||||
|
baseline, timeout_s if timeout_s is not None else YIELD_CONFIRM_TIMEOUT_S)
|
||||||
|
|
||||||
|
if barrier.get("outcome") == "busy":
|
||||||
|
# The unload is queued and will fire when the generation ends. Keep watching
|
||||||
|
# in the background so the release is still logged and the counters stay true,
|
||||||
|
# without holding the caller here for the length of someone's inference.
|
||||||
|
_spawn_detached(_confirm_release_later(targets, baseline))
|
||||||
|
|
||||||
duration_ms = round((time.perf_counter() - t0) * 1000, 2)
|
duration_ms = round((time.perf_counter() - t0) * 1000, 2)
|
||||||
freed_gb = round(max(baseline - barrier.get("residual_bytes", 0), 0) / (1024**3), 2)
|
freed_gb = round(max(baseline - barrier.get("residual_bytes", 0), 0) / (1024**3), 2)
|
||||||
|
|
||||||
|
outcome = barrier.get("outcome", "unconfirmed")
|
||||||
_record({
|
_record({
|
||||||
"event_type": "Ollama VRAM Yield",
|
"event_type": "Ollama VRAM Yield",
|
||||||
"source": ", ".join(targets)[:200],
|
"source": ", ".join(targets)[:200],
|
||||||
"target": "VRAM 0MB (Kept in RAM)",
|
"target": "VRAM 0MB (Kept in RAM)",
|
||||||
"duration_ms": duration_ms,
|
"duration_ms": duration_ms,
|
||||||
"yield_confirm_ms": barrier.get("confirm_ms"),
|
"yield_confirm_ms": barrier.get("confirm_ms"),
|
||||||
"cache_status": "RAM-Cached" if barrier.get("confirmed") else "Yield Timeout",
|
"cache_status": {
|
||||||
|
"released": "RAM-Cached",
|
||||||
|
"busy": "Deferred — LLM generating",
|
||||||
|
"stuck": "Yield Stalled",
|
||||||
|
}.get(outcome, "Yield Unconfirmed"),
|
||||||
"detail": barrier.get("error"),
|
"detail": barrier.get("error"),
|
||||||
})
|
})
|
||||||
return {
|
return {
|
||||||
"success": True,
|
"success": True,
|
||||||
|
"outcome": outcome,
|
||||||
"model": model_name,
|
"model": model_name,
|
||||||
"models_unloaded": targets,
|
"models_unloaded": targets,
|
||||||
"duration_ms": duration_ms,
|
"duration_ms": duration_ms,
|
||||||
@@ -512,6 +619,7 @@ async def instant_free_ollama_vram(model_name: Optional[str] = None,
|
|||||||
"freed_gb": freed_gb,
|
"freed_gb": freed_gb,
|
||||||
"residual_gb": round(barrier.get("residual_bytes", 0) / (1024**3), 2),
|
"residual_gb": round(barrier.get("residual_bytes", 0) / (1024**3), 2),
|
||||||
"free_vram_gb": round(barrier.get("free_bytes", 0) / (1024**3), 2),
|
"free_vram_gb": round(barrier.get("free_bytes", 0) / (1024**3), 2),
|
||||||
|
"gpu_util_pct": barrier.get("gpu_util_pct"),
|
||||||
"error": barrier.get("error"),
|
"error": barrier.get("error"),
|
||||||
}
|
}
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -683,7 +791,18 @@ class AutoArbitrator:
|
|||||||
# benchmark would otherwise trip trigger_comfy_priority, which reapplies the whole
|
# benchmark would otherwise trip trigger_comfy_priority, which reapplies the whole
|
||||||
# 'comfy' profile and silently overwrites the clock the sweep is measuring.
|
# 'comfy' profile and silently overwrites the clock the sweep is measuring.
|
||||||
self.oc_suspended = False
|
self.oc_suspended = False
|
||||||
self.stats = {"yields": 0, "purges": 0, "yield_timeouts": 0, "deferred_purges": 0}
|
# Per-model backoff. A model that is mid-generation cannot yield, and asking it
|
||||||
|
# again every second just blocks the loop repeatedly for no benefit.
|
||||||
|
self._yield_backoff_until: Dict[str, float] = {}
|
||||||
|
self._yield_busy_streak: Dict[str, int] = {}
|
||||||
|
self.stats = {
|
||||||
|
"yields": 0, # release confirmed
|
||||||
|
"yield_deferred_busy": 0, # model mid-generation; unload queued behind it
|
||||||
|
"yield_stalled": 0, # VRAM held with an idle GPU -- the real failure
|
||||||
|
"deferred_releases": 0, # queued unloads that later landed
|
||||||
|
"purges": 0,
|
||||||
|
"deferred_purges": 0,
|
||||||
|
}
|
||||||
|
|
||||||
async def start(self):
|
async def start(self):
|
||||||
if self.running:
|
if self.running:
|
||||||
@@ -708,8 +827,19 @@ class AutoArbitrator:
|
|||||||
await close_clients()
|
await close_clients()
|
||||||
logger.info("AutoArbitrator background engine stopped.")
|
logger.info("AutoArbitrator background engine stopped.")
|
||||||
|
|
||||||
|
# Backoff schedule for a model that keeps reporting busy, in seconds.
|
||||||
|
BUSY_BACKOFF_S = (5.0, 15.0, 30.0, 60.0)
|
||||||
|
|
||||||
|
def note_deferred_release(self, waited_ms: float) -> None:
|
||||||
|
"""Called when a queued unload finally lands after a generation finished."""
|
||||||
|
self.stats["deferred_releases"] += 1
|
||||||
|
self._yield_backoff_until.clear()
|
||||||
|
self._yield_busy_streak.clear()
|
||||||
|
self.last_action = (f"VRAM released after the LLM finished "
|
||||||
|
f"({round(waited_ms / 1000, 1)}s) — ComfyUI can proceed")
|
||||||
|
|
||||||
async def trigger_comfy_priority(self, reason: str = "ComfyUI prompt detected"):
|
async def trigger_comfy_priority(self, reason: str = "ComfyUI prompt detected"):
|
||||||
"""Yield Ollama's VRAM — and confirm it is gone — before diffusion allocates."""
|
"""Yield Ollama's VRAM before diffusion allocates, without fighting a busy model."""
|
||||||
self.comfy_was_active = True
|
self.comfy_was_active = True
|
||||||
self.comfy_idle_since = None
|
self.comfy_idle_since = None
|
||||||
self._apply_oc_profile("comfy")
|
self._apply_oc_profile("comfy")
|
||||||
@@ -718,21 +848,39 @@ class AutoArbitrator:
|
|||||||
return
|
return
|
||||||
|
|
||||||
ollama_state = await get_ollama_live_state()
|
ollama_state = await get_ollama_live_state()
|
||||||
if ollama_state.get("active_model_name"):
|
model = ollama_state.get("active_model_name")
|
||||||
model = ollama_state["active_model_name"]
|
if not model:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Still finishing an inference we already asked to unload: leave it alone.
|
||||||
|
until = self._yield_backoff_until.get(model, 0.0)
|
||||||
|
if now < until:
|
||||||
|
return
|
||||||
|
|
||||||
logger.info(f"⚡ ComfyUI active ({reason}) -> Auto-yielding Ollama model '{model}'...")
|
logger.info(f"⚡ ComfyUI active ({reason}) -> Auto-yielding Ollama model '{model}'...")
|
||||||
self.last_yield_time = time.time()
|
self.last_yield_time = time.time()
|
||||||
res = await instant_free_ollama_vram(model, confirm=True)
|
res = await instant_free_ollama_vram(model, confirm=True)
|
||||||
|
outcome = res.get("outcome")
|
||||||
|
|
||||||
|
if outcome == "released":
|
||||||
self.stats["yields"] += 1
|
self.stats["yields"] += 1
|
||||||
if res.get("confirmed"):
|
self._yield_backoff_until.pop(model, None)
|
||||||
|
self._yield_busy_streak.pop(model, None)
|
||||||
self.last_action = (f"Yielded '{model}' for ComfyUI in "
|
self.last_action = (f"Yielded '{model}' for ComfyUI in "
|
||||||
f"{res.get('confirm_ms')}ms (confirmed {res.get('freed_gb')}GB free)")
|
f"{res.get('confirm_ms')}ms ({res.get('freed_gb')}GB freed)")
|
||||||
|
elif outcome == "busy":
|
||||||
|
streak = self._yield_busy_streak.get(model, 0)
|
||||||
|
delay = self.BUSY_BACKOFF_S[min(streak, len(self.BUSY_BACKOFF_S) - 1)]
|
||||||
|
self._yield_busy_streak[model] = streak + 1
|
||||||
|
self._yield_backoff_until[model] = time.time() + delay
|
||||||
|
self.stats["yield_deferred_busy"] += 1
|
||||||
|
self.last_action = (f"'{model}' is mid-generation ({res.get('gpu_util_pct')}% GPU); "
|
||||||
|
f"unload is queued and will apply when it finishes")
|
||||||
|
logger.info(f"Yield deferred: {res.get('error')} — backing off {delay}s")
|
||||||
else:
|
else:
|
||||||
self.stats["yield_timeouts"] += 1
|
self.stats["yield_stalled"] += 1
|
||||||
self.last_action = (f"⚠ Yield of '{model}' NOT confirmed: "
|
self.last_action = (f"⚠ '{model}' holding {res.get('residual_gb')}GB with an idle GPU")
|
||||||
f"{res.get('residual_gb')}GB still held")
|
logger.warning(f"VRAM yield stalled: {res.get('error')}")
|
||||||
logger.warning(f"VRAM yield barrier timed out: {res.get('error')}")
|
|
||||||
logger.info(f"Ollama auto-yield completed: {res}")
|
|
||||||
|
|
||||||
async def trigger_comfy_completed(self, immediate: bool = False):
|
async def trigger_comfy_completed(self, immediate: bool = False):
|
||||||
"""Mark the end of a generation. The actual purge is deferred unless forced."""
|
"""Mark the end of a generation. The actual purge is deferred unless forced."""
|
||||||
@@ -893,6 +1041,9 @@ class AutoArbitrator:
|
|||||||
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
|
"idle_purge_after_s": self.COMFY_IDLE_PURGE_S,
|
||||||
"oc_profile": self.oc_profile,
|
"oc_profile": self.oc_profile,
|
||||||
"counters": dict(self.stats),
|
"counters": dict(self.stats),
|
||||||
|
"yield_backoff": {m: round(max(t - time.time(), 0), 1)
|
||||||
|
for m, t in self._yield_backoff_until.items()
|
||||||
|
if t > time.time()},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user