diff --git a/metatron.py b/metatron.py index dfc20f4..9a77c33 100644 --- a/metatron.py +++ b/metatron.py @@ -2,22 +2,28 @@ """METATRON — lightweight Ollama terminal harness. Talks to Ollama on nightmare (hardcoded). Launched by typing `metatron`: - 1. model selection screen (fetches live models from nightmare) + 1. model selection screen (fetches live models + sizes from nightmare) 2. chat REPL with streaming - 3. full system access: `!command` runs a shell command directly, and the - model can also run commands via the run_command tool (native function calling). + 3. system access: `!command` + native run_command tool + 4. internet: `?query` (search) / `?fetch ` (read a page) + web_search tool -Stdlib-only (urllib + subprocess). Hardcoded: Ollama URL, system prompt, tool. -Only the model list is dynamic. +Stdlib-only (urllib + subprocess). Hardcoded: Ollama URL, SearXNG URL, system +prompt, tools. Only the model list is dynamic. """ import json +import os +import re import subprocess import sys +import time import urllib.parse import urllib.request +from html import unescape OLLAMA = "http://10.30.20.29:11434" # nightmare SEARXNG = "http://10.30.20.35:6969" # self-hosted meta-search (CT 516) +HERE = os.path.dirname(os.path.abspath(__file__)) +LAST_MODEL = os.path.join(HERE, ".last_model") SYSTEM_PROMPT = ( "You are METATRON, a system-access AI harness running on a Windows Commando " @@ -67,11 +73,10 @@ def get(path): def post_stream(path, payload): - """POST to Ollama /api/chat, yield NDJSON objects (streaming).""" data = json.dumps(payload).encode() req = urllib.request.Request(OLLAMA + path, data=data, headers={"Content-Type": "application/json"}) - with urllib.request.urlopen(req, timeout=300) as r: + with urllib.request.urlopen(req, timeout=120) as r: for line in r: line = line.strip() if line: @@ -87,44 +92,64 @@ def list_models(): return out +def load_last(): + try: + return open(LAST_MODEL).read().strip() + except Exception: + return None + + +def save_last(model): + try: + open(LAST_MODEL, "w").write(model) + except Exception: + pass + + def pick_model(models): + last = load_last() + names = [n for n, _ in models] print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model " f"{C['y']}(smaller = faster){C['r']}\n") for i, (name, ps) in enumerate(models, 1): - print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}") - print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n") + mark = f" {C['m']}*{C['r']}" if name == last else "" + print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}{mark}") + tail = f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}" + if last and last in names: + tail += f" {C['y']}(Enter = {last}){C['r']}" + print(tail + "\n") while True: try: sel = input(f"{C['g']}model>{C['r']} ").strip() - if sel in ("", "0", "q", "quit", "exit"): + if sel == "" and last and last in names: + return last + if sel in ("0", "q", "quit", "exit"): sys.exit(0) idx = int(sel) - 1 if 0 <= idx < len(models): - return models[idx][0] + return names[idx] except ValueError: pass print(f"{C['y']} pick a number 1-{len(models)}{C['r']}") def warm_model(model): - """Load the model into VRAM so the first real message is fast (no cold start).""" sys.stdout.write(f"{C['y']} warming {model} ...{C['r']}") sys.stdout.flush() payload = {"model": model, "messages": [{"role": "user", "content": "hi"}], - "stream": False, "think": False, + "stream": False, "think": False, "keep_alive": "30m", "options": {"num_predict": 1}} try: data = json.dumps(payload).encode() req = urllib.request.Request(OLLAMA + "/api/chat", data=data, headers={"Content-Type": "application/json"}) - urllib.request.urlopen(req, timeout=180) + urllib.request.urlopen(req, timeout=120) sys.stdout.write(f"\r{C['g']} ready.{C['r']} \n") except Exception as e: sys.stdout.write(f"\r{C['y']} warm failed ({e}){C['r']} \n") def supports_tools(model): - """True if this model has native function-calling (else omit `tools`).""" try: data = json.dumps({"model": model}).encode() req = urllib.request.Request(OLLAMA + "/api/show", data=data, @@ -161,11 +186,65 @@ def web_search(query): return "\n".join(lines) +def fetch_page(url): + if not re.match(r"^https?://", url): + url = "https://" + url + try: + req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"}) + raw = urllib.request.urlopen(req, timeout=25).read().decode("utf-8", "ignore") + except Exception as e: + return f"fetch error: {e}" + t = re.sub(r"", " ", raw, flags=re.S) + t = re.sub(r"", " ", t, flags=re.S) + t = re.sub(r"<[^>]+>", " ", t) + t = unescape(t) + t = re.sub(r"\s+", " ", t) + return t.strip()[:2000] + + +def truncate(text, max_lines=50): + lines = text.split("\n") + if len(lines) > max_lines: + return "\n".join(lines[:max_lines]) + f"\n [... {len(lines) - max_lines} more lines]" + return text + + +def print_help(): + print(f""" + {C['b']}commands{C['r']} + {C['c']}!{C['r']} run a system command + {C['c']}?{C['r']} search the web (SearXNG) + {C['c']}?fetch {C['r']} read a page's text + {C['c']}/save{C['r']} save this session to markdown + {C['c']}/help{C['r']} this list + {C['c']}exit{C['r']} quit +""") + + +def save_transcript(messages, model): + fn = os.path.join(os.getcwd(), + f"metatron-{time.strftime('%Y%m%d-%H%M%S')}.md") + try: + with open(fn, "w", encoding="utf-8") as f: + f.write(f"# METATRON session — {model}\n\n") + for m in messages: + role, content = m.get("role"), m.get("content", "") + if role == "user": + f.write(f"**you:** {content}\n\n") + elif role == "assistant" and content: + f.write(f"**metatron:** {content}\n\n") + elif role == "tool": + f.write(f"```\n{content}\n```\n\n") + return fn + except Exception as e: + return f"save failed: {e}" + + def chat(model): messages = [{"role": "system", "content": SYSTEM_PROMPT}] use_tools = supports_tools(model) print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} " - f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n") + f"{C['y']}(!cmd, ?search, ?fetch, /save, /help, exit){C['r']}\n") while True: try: user = input(f"{C['g']}you>{C['r']} ").strip() @@ -175,30 +254,51 @@ def chat(model): continue if user.lower() in ("exit", "quit", "/q", "/exit"): break + if user in ("/help", "help", "?"): + print_help() + continue + if user == "/save": + fn = save_transcript(messages, model) + print(f" {C['g']}saved: {fn}{C['r']}\n") + continue if user.startswith("!"): - print(f"{C['y']} $ {user[1:]}{C['r']}") - print(f" {run_command(user[1:])}\n") + cmd = user[1:].strip() + print(f"{C['y']} $ {cmd}{C['r']}") + print(f" {truncate(run_command(cmd))}\n") continue if user.startswith("?"): q = user[1:].strip() - print(f"{C['c']} [search] {q}{C['r']}") - print(f" {web_search(q)}\n") + if q.startswith("fetch "): + url = q[6:].strip() + print(f"{C['c']} [fetch] {url}{C['r']}") + print(f" {fetch_page(url)}\n") + elif re.match(r"^https?://|^www\.", q): + print(f"{C['c']} [fetch] {q}{C['r']}") + print(f" {fetch_page(q)}\n") + else: + print(f"{C['c']} [search] {q}{C['r']}") + print(f" {web_search(q)}\n") continue messages.append({"role": "user", "content": user}) - # agentic loop: let the model call run_command until it's satisfied for _ in range(6): payload = {"model": model, "messages": messages, "stream": True, - "think": False} + "think": False, "keep_alive": "30m"} if use_tools: payload["tools"] = TOOLS buf = "" tool_calls = [] + got_content = False + sys.stdout.write(" ...") + sys.stdout.flush() try: for obj in post_stream("/api/chat", payload): msg = obj.get("message", {}) piece = msg.get("content") or "" if piece: + if not got_content: + sys.stdout.write("\r ") + got_content = True buf += piece sys.stdout.write(piece) sys.stdout.flush() @@ -207,6 +307,9 @@ def chat(model): except Exception as e: print(f"\n{C['y']} [ollama error: {e}]{C['r']}") break + if not got_content: + sys.stdout.write("\r") + sys.stdout.flush() if tool_calls: messages.append({"role": "assistant", "content": buf, @@ -222,7 +325,7 @@ def chat(model): cmd = args.get("command", "") print(f"\n{C['y']} $ {cmd}{C['r']}") out = run_command(cmd) - print(f" {out}") + print(f" {truncate(out)}") messages.append({"role": "tool", "content": out}) elif name == "web_search": q = args.get("query", "") @@ -230,7 +333,7 @@ def chat(model): out = web_search(q) print(f" {out[:600]}") messages.append({"role": "tool", "content": out}) - continue # re-send to model with tool results + continue else: messages.append({"role": "assistant", "content": buf}) break @@ -247,6 +350,7 @@ def main(): print(f"{C['y']} no models on nightmare{C['r']}") sys.exit(1) model = pick_model(models) + save_last(model) warm_model(model) chat(model)