Let a higher-priority job preempt lower-priority work in progress

Measured failure: a diffusion job took 46 s instead of 3 s, squeezed into 1.6 GB.
The LLM yielded correctly, an inference reloaded it two seconds later, and
plan_release then refused to touch it because it was "busy" -- so ComfyUI crawled
while Ollama held 13 GB for the whole run.

"Never interrupt busy work" looks like the safe rule and is not. Preempting a
lower-priority tenant is safe precisely because releasing is asynchronous: an Ollama
unload queues behind its running request and applies when that finishes, so nothing
is killed mid-flight. That is what makes fast handoff possible at all, and refusing
to do it defeats the purpose of the service.

A tenant may now be asked for memory if it is idle, whatever its rank, or if it is
busy and ranks strictly below the demander. Equal or higher priority is never
interrupted, so peers cannot fight. Idle tenants are still preferred over preempting
busy ones.

Verified against the real contention: with Ollama at 12.38 GB and 98% utilisation, a
diffusion job released it within three seconds and completed in 18 s rather than 46 s.

Tests updated to the corrected rule, and the fixture's priorities aligned with what
actually ships -- it still had the LLM outranking diffusion from before that was
swapped.

Tests: 246.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-09-07 16:57:08 -07:00
parent ca97f18be6
commit 48bd0096ff
2 changed files with 64 additions and 19 deletions

View File

@@ -188,9 +188,11 @@ class TestReleasePlanning:
"reclaimable": False},
{"name": "stt-relay", "priority": 70, "vram_gb": 0.8, "busy": False,
"reclaimable": False},
{"name": "ollama", "priority": 60, "vram_gb": 0.0, "busy": True,
# Shipped priorities: diffusion outranks the LLM, whose weights reload
# from page cache in seconds.
{"name": "comfyui", "priority": 60, "vram_gb": 7.0, "busy": False,
"reclaimable": True},
{"name": "comfyui", "priority": 50, "vram_gb": 7.0, "busy": False,
{"name": "ollama", "priority": 50, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
]
for s in base:
@@ -210,13 +212,44 @@ class TestReleasePlanning:
plan = T.plan_release("ollama", self._state(), free_gb=1.5, needed_gb=14.9)
assert plan["release"] == ["comfyui"]
def test_a_busy_tenant_is_never_a_victim(self):
def test_a_busy_tenant_ranking_above_the_demander_is_not_a_victim(self):
# comfyui outranks ollama, so ollama may not interrupt it.
state = self._state(comfyui={"busy": True})
plan = T.plan_release("ollama", state, free_gb=1.5, needed_gb=14.9)
assert plan["release"] == []
assert any(b["name"] == "comfyui" and b["why"] == "busy"
assert any(b["name"] == "comfyui" and "busy" in b["why"]
for b in plan["blockers"])
def test_a_higher_priority_demander_preempts_busy_lower_priority_work(self):
"""The measured regression that made this rule necessary.
Refusing to touch anything busy looks safe and is not. With the LLM protected as
"busy", a diffusion job ran 46 s instead of 3 s, squeezed into 1.6 GB, because
the LLM reloaded immediately after yielding and was then untouchable. Preempting
a lower-priority tenant is safe because releasing is asynchronous: an Ollama
unload queues behind its running request rather than killing it.
"""
state = [
{"name": "comfyui", "priority": 60, "vram_gb": 1.65, "busy": True,
"reclaimable": True},
{"name": "ollama", "priority": 50, "vram_gb": 13.03, "busy": True,
"reclaimable": True},
]
plan = T.plan_release("comfyui", state, free_gb=0.28, needed_gb=6.0)
assert plan["release"] == ["ollama"]
def test_an_idle_tenant_is_preferred_over_preempting_a_busy_one(self):
state = [
{"name": "d", "priority": 60, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "busy_low", "priority": 10, "vram_gb": 8.0, "busy": True,
"reclaimable": True},
{"name": "idle_high", "priority": 90, "vram_gb": 8.0, "busy": False,
"reclaimable": True},
]
plan = T.plan_release("d", state, free_gb=0.0, needed_gb=8.0)
assert plan["release"] == ["idle_high"]
def test_unreclaimable_tenants_are_named_as_blockers_not_ignored(self):
# The user needs to know a third-party process is what stands in the way.
plan = T.plan_release("ollama", self._state(), free_gb=0.0, needed_gb=15.5)
@@ -250,16 +283,18 @@ class TestReleasePlanning:
# The lowest-priority idle tenant gives up memory first.
assert plan["release"][0] == "low"
def test_busy_work_is_never_interrupted_whatever_the_priority(self):
def test_peers_cannot_interrupt_each_other(self):
# Equal priority is never preempted, so two tenants at the same rank cannot
# fight over the card.
state = [
{"name": "demander", "priority": 99, "vram_gb": 0.0, "busy": True,
{"name": "a", "priority": 50, "vram_gb": 0.0, "busy": True,
"reclaimable": True},
{"name": "worker", "priority": 1, "vram_gb": 8.0, "busy": True,
{"name": "b", "priority": 50, "vram_gb": 8.0, "busy": True,
"reclaimable": True},
]
plan = T.plan_release("demander", state, free_gb=0.0, needed_gb=8.0)
plan = T.plan_release("a", state, free_gb=0.0, needed_gb=8.0)
assert plan["release"] == []
assert plan["blockers"][0]["why"] == "busy"
assert "busy" in plan["blockers"][0]["why"]
def test_lowest_priority_is_released_first(self):
state = self._state() + [