Describe GPU tenants as data so any application can be arbitrated
The point of this service is fast handoff of one GPU between applications. It grew up
around the two on this box, and their names ended up compiled into process matching,
VRAM attribution, busy detection and release calls alike -- about 385 references
across five modules. That made it a script for Ollama and ComfyUI rather than a GPU
arbitrator.
tenants.py describes an application as data: how to recognise its processes, how to
tell whether it is genuinely working, how to ask it for VRAM back, and how much it
matters when two want the card. Ollama, ComfyUI and the desktop compositor ship as
defaults in tenants.json, so behaviour is unchanged, but the arbitration logic no
longer knows any particular name. Endpoints are generic: GET /api/tenants,
GET /api/tenants/{name}, POST /api/tenants/{name}/release -- the last being the
general form of both the Ollama soft-yield and the ComfyUI purge.
Verified by registering a third application on this machine with no code change: the
speech relay that had been showing up only as anonymous "unmanaged VRAM" is now named,
attributed, and probed by the VRAM it holds rather than by an API it does not have.
Because it declares no release strategy, a release request returns 409 explaining that
its memory cannot be reclaimed, instead of reporting a success that did nothing.
Busy probes deliberately cannot use GPU utilisation. It is shared by every tenant, so
it cannot attribute work to one of them -- the mistake that made a stale ComfyUI queue
entry undetectable earlier in this branch. A tenant's own VRAM is the signal.
Writing the tests exposed that the suite had become non-hermetic: classification is now
configuration, so a test asserting "a third-party process is unmanaged" started failing
the moment the speech relay was registered on this machine. An autouse fixture now
isolates every test from the operator's live tenants.json.
Tests: 231 (was 206).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -12,6 +12,7 @@ import websockets
|
||||
|
||||
import overclock_manager
|
||||
import ram_optimizer
|
||||
import tenants as tenants_mod
|
||||
import telemetry_store
|
||||
|
||||
try:
|
||||
@@ -198,6 +199,19 @@ _PID_KIND_CACHE: Dict[tuple, str] = {}
|
||||
_PID_KIND_CACHE_MAX = 512
|
||||
|
||||
|
||||
def _pid_key(pid: int) -> Optional[tuple]:
|
||||
try:
|
||||
return (pid, psutil.Process(pid).create_time())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# Tenant names as used by this module's buckets. The tenant registry is the source of
|
||||
# truth for *which* application a process belongs to; these two names are kept because
|
||||
# the REST payloads and the dashboard have used them since the beginning.
|
||||
_BUCKET_ALIASES = {"comfyui": "comfy"}
|
||||
|
||||
|
||||
def _pid_key(pid: int) -> Optional[tuple]:
|
||||
try:
|
||||
return (pid, psutil.Process(pid).create_time())
|
||||
@@ -214,26 +228,16 @@ DESKTOP_PROCESS_HINTS = (
|
||||
|
||||
|
||||
def _classify_pid(pid: int) -> str:
|
||||
"""Bucket a GPU process into ollama | comfy | desktop | unmanaged.
|
||||
"""Which tenant owns this GPU process.
|
||||
|
||||
The old version had one catch-all "other" bucket, which put a 3.9 MB compositor and
|
||||
an 842 MB long-running inference script in the same number. That matters: this
|
||||
service can reclaim VRAM from ComfyUI, but it cannot touch a third-party workload,
|
||||
and pretending otherwise makes it promise headroom it cannot deliver.
|
||||
The matching rules used to be substrings compiled into this function, which made the
|
||||
two applications on this box part of the arbitrator rather than input to it. They now
|
||||
come from the tenant registry, so a third application is a config entry.
|
||||
|
||||
"unmanaged" still means something specific and useful: VRAM held by something with no
|
||||
declared way to release it, and therefore headroom this service can never offer.
|
||||
"""
|
||||
try:
|
||||
proc = psutil.Process(pid)
|
||||
pname = proc.name().lower()
|
||||
cmdline = " ".join(proc.cmdline()).lower()
|
||||
except Exception:
|
||||
return "unmanaged"
|
||||
if "ollama" in pname or "llama-server" in cmdline:
|
||||
return "ollama"
|
||||
if "comfyui" in cmdline or "comfy" in cmdline or cmdline.rstrip().endswith("main.py"):
|
||||
return "comfy"
|
||||
if any(hint in pname or hint in cmdline for hint in DESKTOP_PROCESS_HINTS):
|
||||
return "desktop"
|
||||
return "unmanaged"
|
||||
return _BUCKET_ALIASES.get(tenants_mod.classify_pid(pid), tenants_mod.classify_pid(pid))
|
||||
|
||||
|
||||
def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
@@ -336,7 +340,10 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
"desktop_bytes": 0,
|
||||
"unmanaged_bytes": 0,
|
||||
"unmanaged": [],
|
||||
"processes": []
|
||||
"processes": [],
|
||||
# Generic attribution: one entry per tenant, so an application added to the
|
||||
# registry is reported without any change here.
|
||||
"by_tenant": {},
|
||||
}
|
||||
|
||||
try:
|
||||
@@ -376,6 +383,8 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
"vram_mb": round(used_mem / (1024**2), 1),
|
||||
})
|
||||
|
||||
proc_breakdown["by_tenant"][kind] = (
|
||||
proc_breakdown["by_tenant"].get(kind, 0) + used_mem)
|
||||
proc_breakdown["processes"].append({
|
||||
"pid": pid,
|
||||
"name": pname,
|
||||
@@ -431,6 +440,8 @@ def get_gpu_hardware_stats() -> Dict[str, Any]:
|
||||
# reclaimed, so it is permanently unavailable headroom.
|
||||
"unmanaged_gb": round(proc_breakdown["unmanaged_bytes"] / (1024**3), 2),
|
||||
"unmanaged": proc_breakdown["unmanaged"],
|
||||
"by_tenant_gb": {k: round(b / (1024**3), 2)
|
||||
for k, b in proc_breakdown["by_tenant"].items()},
|
||||
"free_mb": round(free_vram / (1024**2), 1),
|
||||
"free_gb": round(free_vram / (1024**3), 2),
|
||||
"processes": proc_breakdown["processes"],
|
||||
|
||||
Reference in New Issue
Block a user