Files
gpu-program-swapper/tests/test_vram_helpers.py
drjones 25b601e24a Fix two more stale-cache bugs; account for VRAM this service cannot reclaim
The stale-readback bug fixed in f5917a0 was a class, not an instance. Two more:

- ram_optimizer's residency report cached for 15s and was never invalidated when
  anything warmed a file, so warming a model and then looking at residency showed
  the state from before the warm. warm_file_to_ram now invalidates it.
- _PID_KIND_CACHE was keyed on pid alone and never expired. Linux recycles PIDs, so
  a stale entry could attribute a new process's VRAM to Ollama or ComfyUI -- inside
  the very snapshot the yield barrier trusts to decide whether VRAM was released.
  Now keyed by (pid, process start time) and bounded.

Unmanaged VRAM. Investigating a persistence-mode warning turned up a third GPU
consumer this service does not model: stt_relay.py, holding 842 MB for nearly three
days. It was bucketed as "system" alongside gnome-shell's 3.9 MB. That conflation
matters, because ComfyUI's memory can be reclaimed and a third party's cannot, and
the reclaim path assumed ComfyUI was always to blame for missing headroom.

Processes are now bucketed ollama | comfy | desktop | unmanaged. The breakdown
reports desktop_gb and unmanaged_gb separately and names the unmanaged processes;
when a reclaim-and-retry still fails, the error identifies them rather than
implying ComfyUI was at fault; and the dashboard shows the unreclaimable total, so
headroom the arbitrator can never give back is visible rather than inferred.

Checked and deliberately not changed: persistence mode reads Disabled, but
nvidia-persistenced is active and two clients hold the GPU open continuously, so
the driver never unloads. The nvidia-smi warning is legacy noise here and is not a
source of the profile drift.

Tests: 169 (was 164). The new ones cover the bucketing, and one existing test used
Xorg as its "unknown process" fixture -- correct before a display server had its own
bucket, wrong after.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-05 19:11:40 -07:00

140 lines
5.5 KiB
Python

"""Pure helpers in vram_arbitrator: NVML throttle-bit decoding and PID attribution.
Deliberately excludes instant_free_ollama_vram, the AutoArbitrator yield/purge paths and
the SSE broker — that contract is in flux.
"""
import vram_arbitrator as va
def test_decode_throttle_reasons_empty_when_no_bits_set():
assert va.decode_throttle_reasons(0) == []
def test_decode_throttle_reasons_maps_each_known_bit():
"""Every mask in the table must decode to exactly its own name in isolation."""
for mask, name in va.THROTTLE_REASONS.items():
assert va.decode_throttle_reasons(mask) == [name]
def test_decode_throttle_reasons_decodes_combined_bits():
"""Real NVML samples set several bits at once; all of them must come back."""
bits = 0x20 | 0x40 # sw_thermal_slowdown | hw_thermal_slowdown
assert set(va.decode_throttle_reasons(bits)) == {"sw_thermal_slowdown", "hw_thermal_slowdown"}
def test_decode_throttle_reasons_ignores_unknown_bits():
"""An undocumented bit from a future driver must not crash or invent a reason."""
assert va.decode_throttle_reasons(0x8000_0000) == []
def test_hard_throttle_names_match_thermal_governor_expectations():
"""thermal_governor escalates on a fixed set of reason strings produced here.
If a name is renamed in one module and not the other the governor silently stops
reacting to hardware slowdowns, so pin the shared vocabulary."""
import thermal_governor as tg
assert tg.HARD_THROTTLES <= set(va.THROTTLE_REASONS.values())
class _FakeProc:
def __init__(self, name, cmdline):
self._name = name
self._cmdline = cmdline
def name(self):
return self._name
def cmdline(self):
return self._cmdline
def _patch_proc(monkeypatch, proc):
monkeypatch.setattr(va.psutil, "Process", lambda pid: proc)
def test_classify_pid_detects_ollama_by_process_name(monkeypatch):
_patch_proc(monkeypatch, _FakeProc("ollama", ["/usr/local/bin/ollama", "serve"]))
assert va._classify_pid(1234) == "ollama"
def test_classify_pid_detects_ollama_runner_by_cmdline(monkeypatch):
"""Ollama's model runner is a separate llama-server process; its VRAM is Ollama's."""
_patch_proc(monkeypatch, _FakeProc("llama-server",
["/usr/lib/ollama/llama-server", "--model", "blob"]))
assert va._classify_pid(1234) == "ollama"
def test_classify_pid_detects_comfyui(monkeypatch):
_patch_proc(monkeypatch, _FakeProc("python3", ["python3", "/opt/ComfyUI/main.py", "--listen"]))
assert va._classify_pid(1234) == "comfy"
def test_classify_pid_unknown_process_is_unmanaged(monkeypatch):
# Xorg used to stand in for "unknown" here, but a display server is now its own
# bucket, so this needs a process that is genuinely neither ours nor the desktop's.
_patch_proc(monkeypatch, _FakeProc("trainer", ["/opt/ml/bin/trainer", "--epochs", "3"]))
# "unmanaged" rather than "other": a third-party GPU workload holds VRAM this
# service cannot reclaim, and must not be lumped in with the desktop compositor.
assert va._classify_pid(1234) == "unmanaged"
def test_classify_pid_display_server_is_desktop(monkeypatch):
_patch_proc(monkeypatch, _FakeProc("Xorg", ["/usr/lib/xorg/Xorg", ":8"]))
assert va._classify_pid(1234) == "desktop"
def test_classify_pid_returns_unmanaged_when_process_vanished(monkeypatch):
"""PIDs are read from NVML and can exit before psutil looks them up; that is normal
and must not raise inside the 20 ms VRAM poll loop."""
def _boom(pid):
raise va.psutil.NoSuchProcess(pid)
monkeypatch.setattr(va.psutil, "Process", _boom)
assert va._classify_pid(999999) == "unmanaged"
# --- process bucketing: desktop vs unmanaged ---------------------------------
#
# Real case from this machine: stt_relay.py held 842 MB of VRAM for nearly three days
# while gnome-shell held 3.9 MB. A single "other" bucket reported them as one number,
# which matters because ComfyUI's memory can be reclaimed and a third party's cannot.
class _FakeProc:
def __init__(self, name, cmdline):
self._name = name
self._cmdline = cmdline
def name(self):
return self._name
def cmdline(self):
return self._cmdline
def create_time(self):
return 1234.5
def _classify(monkeypatch, name, cmdline):
monkeypatch.setattr(va.psutil, "Process",
lambda pid: _FakeProc(name, cmdline))
return va._classify_pid(4321)
def test_desktop_compositors_are_their_own_bucket(monkeypatch):
assert _classify(monkeypatch, "gnome-shell", ["/usr/bin/gnome-shell"]) == "desktop"
assert _classify(monkeypatch, "Xorg", ["/usr/lib/xorg/Xorg", ":8"]) == "desktop"
def test_third_party_compute_is_unmanaged_not_desktop(monkeypatch):
kind = _classify(monkeypatch, "python",
["/home/u/robopest-venv/bin/python", "/home/u/stt_relay.py"])
assert kind == "unmanaged"
def test_ollama_and_comfy_still_win_over_the_catch_all(monkeypatch):
assert _classify(monkeypatch, "llama-server",
["/usr/local/lib/ollama/llama-server", "--model", "x"]) == "ollama"
assert _classify(monkeypatch, "python",
["/home/u/ComfyUI/venv/bin/python", "main.py"]) == "comfy"
def test_pid_cache_is_keyed_by_start_time_not_pid_alone():
# Linux recycles PIDs; a stale entry would attribute a new process's VRAM to Ollama
# inside the same snapshot the yield barrier trusts.
assert all(isinstance(k, tuple) and len(k) == 2 for k in va._PID_KIND_CACHE)