responsiveness: show model sizes, warm model on select, skip tools for non-tool models
This commit is contained in:
50
metatron.py
50
metatron.py
@@ -80,13 +80,18 @@ def post_stream(path, payload):
|
||||
|
||||
def list_models():
|
||||
d = get("/api/tags")
|
||||
return [m["name"] for m in d.get("models", [])]
|
||||
out = []
|
||||
for m in d.get("models", []):
|
||||
ps = (m.get("details") or {}).get("parameter_size", "") or ""
|
||||
out.append((m["name"], ps))
|
||||
return out
|
||||
|
||||
|
||||
def pick_model(models):
|
||||
print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model\n")
|
||||
for i, name in enumerate(models, 1):
|
||||
print(f" {C['c']}[{i}]{C['r']} {name}")
|
||||
print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model "
|
||||
f"{C['y']}(smaller = faster){C['r']}\n")
|
||||
for i, (name, ps) in enumerate(models, 1):
|
||||
print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}")
|
||||
print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n")
|
||||
while True:
|
||||
try:
|
||||
@@ -95,12 +100,41 @@ def pick_model(models):
|
||||
sys.exit(0)
|
||||
idx = int(sel) - 1
|
||||
if 0 <= idx < len(models):
|
||||
return models[idx]
|
||||
return models[idx][0]
|
||||
except ValueError:
|
||||
pass
|
||||
print(f"{C['y']} pick a number 1-{len(models)}{C['r']}")
|
||||
|
||||
|
||||
def warm_model(model):
|
||||
"""Load the model into VRAM so the first real message is fast (no cold start)."""
|
||||
sys.stdout.write(f"{C['y']} warming {model} ...{C['r']}")
|
||||
sys.stdout.flush()
|
||||
payload = {"model": model, "messages": [{"role": "user", "content": "hi"}],
|
||||
"stream": False, "think": False,
|
||||
"options": {"num_predict": 1}}
|
||||
try:
|
||||
data = json.dumps(payload).encode()
|
||||
req = urllib.request.Request(OLLAMA + "/api/chat", data=data,
|
||||
headers={"Content-Type": "application/json"})
|
||||
urllib.request.urlopen(req, timeout=180)
|
||||
sys.stdout.write(f"\r{C['g']} ready.{C['r']} \n")
|
||||
except Exception as e:
|
||||
sys.stdout.write(f"\r{C['y']} warm failed ({e}){C['r']} \n")
|
||||
|
||||
|
||||
def supports_tools(model):
|
||||
"""True if this model has native function-calling (else omit `tools`)."""
|
||||
try:
|
||||
data = json.dumps({"model": model}).encode()
|
||||
req = urllib.request.Request(OLLAMA + "/api/show", data=data,
|
||||
headers={"Content-Type": "application/json"})
|
||||
d = json.load(urllib.request.urlopen(req, timeout=15))
|
||||
return "tools" in (d.get("capabilities") or [])
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def run_command(cmd):
|
||||
try:
|
||||
r = subprocess.run(cmd, shell=True, capture_output=True, text=True,
|
||||
@@ -129,6 +163,7 @@ def web_search(query):
|
||||
|
||||
def chat(model):
|
||||
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
|
||||
use_tools = supports_tools(model)
|
||||
print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} "
|
||||
f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n")
|
||||
while True:
|
||||
@@ -154,7 +189,9 @@ def chat(model):
|
||||
# agentic loop: let the model call run_command until it's satisfied
|
||||
for _ in range(6):
|
||||
payload = {"model": model, "messages": messages, "stream": True,
|
||||
"think": False, "tools": TOOLS}
|
||||
"think": False}
|
||||
if use_tools:
|
||||
payload["tools"] = TOOLS
|
||||
buf = ""
|
||||
tool_calls = []
|
||||
try:
|
||||
@@ -210,6 +247,7 @@ def main():
|
||||
print(f"{C['y']} no models on nightmare{C['r']}")
|
||||
sys.exit(1)
|
||||
model = pick_model(models)
|
||||
warm_model(model)
|
||||
chat(model)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user