improvements: ?fetch, /help, /save, remember-last-model, output truncation, keep_alive + fail-fast

This commit is contained in:
drjones
2026-09-18 18:32:04 -07:00
parent 5c1e487a3c
commit 5a97d04875

View File

@@ -2,22 +2,28 @@
"""METATRON — lightweight Ollama terminal harness.
Talks to Ollama on nightmare (hardcoded). Launched by typing `metatron`:
1. model selection screen (fetches live models from nightmare)
1. model selection screen (fetches live models + sizes from nightmare)
2. chat REPL with streaming
3. full system access: `!command` runs a shell command directly, and the
model can also run commands via the run_command tool (native function calling).
3. system access: `!command` + native run_command tool
4. internet: `?query` (search) / `?fetch <url>` (read a page) + web_search tool
Stdlib-only (urllib + subprocess). Hardcoded: Ollama URL, system prompt, tool.
Only the model list is dynamic.
Stdlib-only (urllib + subprocess). Hardcoded: Ollama URL, SearXNG URL, system
prompt, tools. Only the model list is dynamic.
"""
import json
import os
import re
import subprocess
import sys
import time
import urllib.parse
import urllib.request
from html import unescape
OLLAMA = "http://10.30.20.29:11434" # nightmare
SEARXNG = "http://10.30.20.35:6969" # self-hosted meta-search (CT 516)
HERE = os.path.dirname(os.path.abspath(__file__))
LAST_MODEL = os.path.join(HERE, ".last_model")
SYSTEM_PROMPT = (
"You are METATRON, a system-access AI harness running on a Windows Commando "
@@ -67,11 +73,10 @@ def get(path):
def post_stream(path, payload):
"""POST to Ollama /api/chat, yield NDJSON objects (streaming)."""
data = json.dumps(payload).encode()
req = urllib.request.Request(OLLAMA + path, data=data,
headers={"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=300) as r:
with urllib.request.urlopen(req, timeout=120) as r:
for line in r:
line = line.strip()
if line:
@@ -87,44 +92,64 @@ def list_models():
return out
def load_last():
try:
return open(LAST_MODEL).read().strip()
except Exception:
return None
def save_last(model):
try:
open(LAST_MODEL, "w").write(model)
except Exception:
pass
def pick_model(models):
last = load_last()
names = [n for n, _ in models]
print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model "
f"{C['y']}(smaller = faster){C['r']}\n")
for i, (name, ps) in enumerate(models, 1):
print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}")
print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n")
mark = f" {C['m']}*{C['r']}" if name == last else ""
print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}{mark}")
tail = f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}"
if last and last in names:
tail += f" {C['y']}(Enter = {last}){C['r']}"
print(tail + "\n")
while True:
try:
sel = input(f"{C['g']}model>{C['r']} ").strip()
if sel in ("", "0", "q", "quit", "exit"):
if sel == "" and last and last in names:
return last
if sel in ("0", "q", "quit", "exit"):
sys.exit(0)
idx = int(sel) - 1
if 0 <= idx < len(models):
return models[idx][0]
return names[idx]
except ValueError:
pass
print(f"{C['y']} pick a number 1-{len(models)}{C['r']}")
def warm_model(model):
"""Load the model into VRAM so the first real message is fast (no cold start)."""
sys.stdout.write(f"{C['y']} warming {model} ...{C['r']}")
sys.stdout.flush()
payload = {"model": model, "messages": [{"role": "user", "content": "hi"}],
"stream": False, "think": False,
"stream": False, "think": False, "keep_alive": "30m",
"options": {"num_predict": 1}}
try:
data = json.dumps(payload).encode()
req = urllib.request.Request(OLLAMA + "/api/chat", data=data,
headers={"Content-Type": "application/json"})
urllib.request.urlopen(req, timeout=180)
urllib.request.urlopen(req, timeout=120)
sys.stdout.write(f"\r{C['g']} ready.{C['r']} \n")
except Exception as e:
sys.stdout.write(f"\r{C['y']} warm failed ({e}){C['r']} \n")
def supports_tools(model):
"""True if this model has native function-calling (else omit `tools`)."""
try:
data = json.dumps({"model": model}).encode()
req = urllib.request.Request(OLLAMA + "/api/show", data=data,
@@ -161,11 +186,65 @@ def web_search(query):
return "\n".join(lines)
def fetch_page(url):
if not re.match(r"^https?://", url):
url = "https://" + url
try:
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
raw = urllib.request.urlopen(req, timeout=25).read().decode("utf-8", "ignore")
except Exception as e:
return f"fetch error: {e}"
t = re.sub(r"<script.*?</script>", " ", raw, flags=re.S)
t = re.sub(r"<style.*?</style>", " ", t, flags=re.S)
t = re.sub(r"<[^>]+>", " ", t)
t = unescape(t)
t = re.sub(r"\s+", " ", t)
return t.strip()[:2000]
def truncate(text, max_lines=50):
lines = text.split("\n")
if len(lines) > max_lines:
return "\n".join(lines[:max_lines]) + f"\n [... {len(lines) - max_lines} more lines]"
return text
def print_help():
print(f"""
{C['b']}commands{C['r']}
{C['c']}!<cmd>{C['r']} run a system command
{C['c']}?<query>{C['r']} search the web (SearXNG)
{C['c']}?fetch <url>{C['r']} read a page's text
{C['c']}/save{C['r']} save this session to markdown
{C['c']}/help{C['r']} this list
{C['c']}exit{C['r']} quit
""")
def save_transcript(messages, model):
fn = os.path.join(os.getcwd(),
f"metatron-{time.strftime('%Y%m%d-%H%M%S')}.md")
try:
with open(fn, "w", encoding="utf-8") as f:
f.write(f"# METATRON session — {model}\n\n")
for m in messages:
role, content = m.get("role"), m.get("content", "")
if role == "user":
f.write(f"**you:** {content}\n\n")
elif role == "assistant" and content:
f.write(f"**metatron:** {content}\n\n")
elif role == "tool":
f.write(f"```\n{content}\n```\n\n")
return fn
except Exception as e:
return f"save failed: {e}"
def chat(model):
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
use_tools = supports_tools(model)
print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} "
f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n")
f"{C['y']}(!cmd, ?search, ?fetch, /save, /help, exit){C['r']}\n")
while True:
try:
user = input(f"{C['g']}you>{C['r']} ").strip()
@@ -175,30 +254,51 @@ def chat(model):
continue
if user.lower() in ("exit", "quit", "/q", "/exit"):
break
if user in ("/help", "help", "?"):
print_help()
continue
if user == "/save":
fn = save_transcript(messages, model)
print(f" {C['g']}saved: {fn}{C['r']}\n")
continue
if user.startswith("!"):
print(f"{C['y']} $ {user[1:]}{C['r']}")
print(f" {run_command(user[1:])}\n")
cmd = user[1:].strip()
print(f"{C['y']} $ {cmd}{C['r']}")
print(f" {truncate(run_command(cmd))}\n")
continue
if user.startswith("?"):
q = user[1:].strip()
print(f"{C['c']} [search] {q}{C['r']}")
print(f" {web_search(q)}\n")
if q.startswith("fetch "):
url = q[6:].strip()
print(f"{C['c']} [fetch] {url}{C['r']}")
print(f" {fetch_page(url)}\n")
elif re.match(r"^https?://|^www\.", q):
print(f"{C['c']} [fetch] {q}{C['r']}")
print(f" {fetch_page(q)}\n")
else:
print(f"{C['c']} [search] {q}{C['r']}")
print(f" {web_search(q)}\n")
continue
messages.append({"role": "user", "content": user})
# agentic loop: let the model call run_command until it's satisfied
for _ in range(6):
payload = {"model": model, "messages": messages, "stream": True,
"think": False}
"think": False, "keep_alive": "30m"}
if use_tools:
payload["tools"] = TOOLS
buf = ""
tool_calls = []
got_content = False
sys.stdout.write(" ...")
sys.stdout.flush()
try:
for obj in post_stream("/api/chat", payload):
msg = obj.get("message", {})
piece = msg.get("content") or ""
if piece:
if not got_content:
sys.stdout.write("\r ")
got_content = True
buf += piece
sys.stdout.write(piece)
sys.stdout.flush()
@@ -207,6 +307,9 @@ def chat(model):
except Exception as e:
print(f"\n{C['y']} [ollama error: {e}]{C['r']}")
break
if not got_content:
sys.stdout.write("\r")
sys.stdout.flush()
if tool_calls:
messages.append({"role": "assistant", "content": buf,
@@ -222,7 +325,7 @@ def chat(model):
cmd = args.get("command", "")
print(f"\n{C['y']} $ {cmd}{C['r']}")
out = run_command(cmd)
print(f" {out}")
print(f" {truncate(out)}")
messages.append({"role": "tool", "content": out})
elif name == "web_search":
q = args.get("query", "")
@@ -230,7 +333,7 @@ def chat(model):
out = web_search(q)
print(f" {out[:600]}")
messages.append({"role": "tool", "content": out})
continue # re-send to model with tool results
continue
else:
messages.append({"role": "assistant", "content": buf})
break
@@ -247,6 +350,7 @@ def main():
print(f"{C['y']} no models on nightmare{C['r']}")
sys.exit(1)
model = pick_model(models)
save_last(model)
warm_model(model)
chat(model)