responsiveness: show model sizes, warm model on select, skip tools for non-tool models
This commit is contained in:
50
metatron.py
50
metatron.py
@@ -80,13 +80,18 @@ def post_stream(path, payload):
|
|||||||
|
|
||||||
def list_models():
|
def list_models():
|
||||||
d = get("/api/tags")
|
d = get("/api/tags")
|
||||||
return [m["name"] for m in d.get("models", [])]
|
out = []
|
||||||
|
for m in d.get("models", []):
|
||||||
|
ps = (m.get("details") or {}).get("parameter_size", "") or ""
|
||||||
|
out.append((m["name"], ps))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def pick_model(models):
|
def pick_model(models):
|
||||||
print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model\n")
|
print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model "
|
||||||
for i, name in enumerate(models, 1):
|
f"{C['y']}(smaller = faster){C['r']}\n")
|
||||||
print(f" {C['c']}[{i}]{C['r']} {name}")
|
for i, (name, ps) in enumerate(models, 1):
|
||||||
|
print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}")
|
||||||
print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n")
|
print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n")
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
@@ -95,12 +100,41 @@ def pick_model(models):
|
|||||||
sys.exit(0)
|
sys.exit(0)
|
||||||
idx = int(sel) - 1
|
idx = int(sel) - 1
|
||||||
if 0 <= idx < len(models):
|
if 0 <= idx < len(models):
|
||||||
return models[idx]
|
return models[idx][0]
|
||||||
except ValueError:
|
except ValueError:
|
||||||
pass
|
pass
|
||||||
print(f"{C['y']} pick a number 1-{len(models)}{C['r']}")
|
print(f"{C['y']} pick a number 1-{len(models)}{C['r']}")
|
||||||
|
|
||||||
|
|
||||||
|
def warm_model(model):
|
||||||
|
"""Load the model into VRAM so the first real message is fast (no cold start)."""
|
||||||
|
sys.stdout.write(f"{C['y']} warming {model} ...{C['r']}")
|
||||||
|
sys.stdout.flush()
|
||||||
|
payload = {"model": model, "messages": [{"role": "user", "content": "hi"}],
|
||||||
|
"stream": False, "think": False,
|
||||||
|
"options": {"num_predict": 1}}
|
||||||
|
try:
|
||||||
|
data = json.dumps(payload).encode()
|
||||||
|
req = urllib.request.Request(OLLAMA + "/api/chat", data=data,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
|
urllib.request.urlopen(req, timeout=180)
|
||||||
|
sys.stdout.write(f"\r{C['g']} ready.{C['r']} \n")
|
||||||
|
except Exception as e:
|
||||||
|
sys.stdout.write(f"\r{C['y']} warm failed ({e}){C['r']} \n")
|
||||||
|
|
||||||
|
|
||||||
|
def supports_tools(model):
|
||||||
|
"""True if this model has native function-calling (else omit `tools`)."""
|
||||||
|
try:
|
||||||
|
data = json.dumps({"model": model}).encode()
|
||||||
|
req = urllib.request.Request(OLLAMA + "/api/show", data=data,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
|
d = json.load(urllib.request.urlopen(req, timeout=15))
|
||||||
|
return "tools" in (d.get("capabilities") or [])
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def run_command(cmd):
|
def run_command(cmd):
|
||||||
try:
|
try:
|
||||||
r = subprocess.run(cmd, shell=True, capture_output=True, text=True,
|
r = subprocess.run(cmd, shell=True, capture_output=True, text=True,
|
||||||
@@ -129,6 +163,7 @@ def web_search(query):
|
|||||||
|
|
||||||
def chat(model):
|
def chat(model):
|
||||||
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
|
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
|
||||||
|
use_tools = supports_tools(model)
|
||||||
print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} "
|
print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} "
|
||||||
f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n")
|
f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n")
|
||||||
while True:
|
while True:
|
||||||
@@ -154,7 +189,9 @@ def chat(model):
|
|||||||
# agentic loop: let the model call run_command until it's satisfied
|
# agentic loop: let the model call run_command until it's satisfied
|
||||||
for _ in range(6):
|
for _ in range(6):
|
||||||
payload = {"model": model, "messages": messages, "stream": True,
|
payload = {"model": model, "messages": messages, "stream": True,
|
||||||
"think": False, "tools": TOOLS}
|
"think": False}
|
||||||
|
if use_tools:
|
||||||
|
payload["tools"] = TOOLS
|
||||||
buf = ""
|
buf = ""
|
||||||
tool_calls = []
|
tool_calls = []
|
||||||
try:
|
try:
|
||||||
@@ -210,6 +247,7 @@ def main():
|
|||||||
print(f"{C['y']} no models on nightmare{C['r']}")
|
print(f"{C['y']} no models on nightmare{C['r']}")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
model = pick_model(models)
|
model = pick_model(models)
|
||||||
|
warm_model(model)
|
||||||
chat(model)
|
chat(model)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user