#!/usr/bin/env python3 """METATRON — lightweight Ollama terminal harness. Talks to Ollama on nightmare (hardcoded). Launched by typing `metatron`: 1. model selection screen (fetches live models from nightmare) 2. chat REPL with streaming 3. full system access: `!command` runs a shell command directly, and the model can also run commands via the run_command tool (native function calling). Stdlib-only (urllib + subprocess). Hardcoded: Ollama URL, system prompt, tool. Only the model list is dynamic. """ import json import subprocess import sys import urllib.parse import urllib.request OLLAMA = "http://10.30.20.29:11434" # nightmare SEARXNG = "http://10.30.20.35:6969" # self-hosted meta-search (CT 516) SYSTEM_PROMPT = ( "You are METATRON, a system-access AI harness running on a Windows Commando " "VM. You have full system access through the run_command tool — execute shell " "commands when asked and report their output. Use web_search for any up-to-date " "or factual internet information. Be concise and direct." ) TOOLS = [{ "type": "function", "function": { "name": "run_command", "description": "Execute a system command (cmd/PowerShell) and return its output. " "Use for file ops, process management, recon, anything on the box.", "parameters": { "type": "object", "properties": { "command": {"type": "string", "description": "The command to run"}, }, "required": ["command"], }, }, }, { "type": "function", "function": { "name": "web_search", "description": "Search the internet and return the top results (title, URL, snippet). " "Use for current events, facts, anything you don't already know.", "parameters": { "type": "object", "properties": { "query": {"type": "string", "description": "The search query"}, }, "required": ["query"], }, }, }] C = {"r": "\033[0m", "b": "\033[1m", "c": "\033[36m", "g": "\033[32m", "y": "\033[33m", "m": "\033[35m"} def get(path): req = urllib.request.Request(OLLAMA + path) with urllib.request.urlopen(req, timeout=30) as r: return json.load(r) def post_stream(path, payload): """POST to Ollama /api/chat, yield NDJSON objects (streaming).""" data = json.dumps(payload).encode() req = urllib.request.Request(OLLAMA + path, data=data, headers={"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=300) as r: for line in r: line = line.strip() if line: yield json.loads(line) def list_models(): d = get("/api/tags") out = [] for m in d.get("models", []): ps = (m.get("details") or {}).get("parameter_size", "") or "" out.append((m["name"], ps)) return out def pick_model(models): print(f"\n{C['b']}{C['m']} METATRON{C['r']} — select a model " f"{C['y']}(smaller = faster){C['r']}\n") for i, (name, ps) in enumerate(models, 1): print(f" {C['c']}[{i}]{C['r']} {name} {C['g']}{ps}{C['r']}") print(f" {C['c']}[0]{C['r']} {C['y']}quit{C['r']}\n") while True: try: sel = input(f"{C['g']}model>{C['r']} ").strip() if sel in ("", "0", "q", "quit", "exit"): sys.exit(0) idx = int(sel) - 1 if 0 <= idx < len(models): return models[idx][0] except ValueError: pass print(f"{C['y']} pick a number 1-{len(models)}{C['r']}") def warm_model(model): """Load the model into VRAM so the first real message is fast (no cold start).""" sys.stdout.write(f"{C['y']} warming {model} ...{C['r']}") sys.stdout.flush() payload = {"model": model, "messages": [{"role": "user", "content": "hi"}], "stream": False, "think": False, "options": {"num_predict": 1}} try: data = json.dumps(payload).encode() req = urllib.request.Request(OLLAMA + "/api/chat", data=data, headers={"Content-Type": "application/json"}) urllib.request.urlopen(req, timeout=180) sys.stdout.write(f"\r{C['g']} ready.{C['r']} \n") except Exception as e: sys.stdout.write(f"\r{C['y']} warm failed ({e}){C['r']} \n") def supports_tools(model): """True if this model has native function-calling (else omit `tools`).""" try: data = json.dumps({"model": model}).encode() req = urllib.request.Request(OLLAMA + "/api/show", data=data, headers={"Content-Type": "application/json"}) d = json.load(urllib.request.urlopen(req, timeout=15)) return "tools" in (d.get("capabilities") or []) except Exception: return False def run_command(cmd): try: r = subprocess.run(cmd, shell=True, capture_output=True, text=True, timeout=120) out = (r.stdout or "") + (r.stderr or "") return out.strip() or "(no output)" except Exception as e: return f"error: {e}" def web_search(query): url = f"{SEARXNG}/search?q={urllib.parse.quote(query)}&format=json" try: d = json.load(urllib.request.urlopen(url, timeout=25)) except Exception as e: return f"search error: {e}" results = d.get("results", []) if not results: return "no results" lines = [] for r in results[:8]: lines.append(f"- {r.get('title', '')}\n {r.get('url', '')}\n" f" {(r.get('content') or '')[:220]}") return "\n".join(lines) def chat(model): messages = [{"role": "system", "content": SYSTEM_PROMPT}] use_tools = supports_tools(model) print(f"\n{C['b']} METATRON {C['c']}:: {model}{C['r']} " f"{C['y']}(!cmd = run, ?query = search, exit = quit){C['r']}\n") while True: try: user = input(f"{C['g']}you>{C['r']} ").strip() except (EOFError, KeyboardInterrupt): break if not user: continue if user.lower() in ("exit", "quit", "/q", "/exit"): break if user.startswith("!"): print(f"{C['y']} $ {user[1:]}{C['r']}") print(f" {run_command(user[1:])}\n") continue if user.startswith("?"): q = user[1:].strip() print(f"{C['c']} [search] {q}{C['r']}") print(f" {web_search(q)}\n") continue messages.append({"role": "user", "content": user}) # agentic loop: let the model call run_command until it's satisfied for _ in range(6): payload = {"model": model, "messages": messages, "stream": True, "think": False} if use_tools: payload["tools"] = TOOLS buf = "" tool_calls = [] try: for obj in post_stream("/api/chat", payload): msg = obj.get("message", {}) piece = msg.get("content") or "" if piece: buf += piece sys.stdout.write(piece) sys.stdout.flush() if obj.get("done") and msg.get("tool_calls"): tool_calls = msg["tool_calls"] except Exception as e: print(f"\n{C['y']} [ollama error: {e}]{C['r']}") break if tool_calls: messages.append({"role": "assistant", "content": buf, "tool_calls": tool_calls}) for tc in tool_calls: fn = tc.get("function", {}) name = fn.get("name") try: args = json.loads(fn.get("arguments", "{}")) except Exception: args = {} if name == "run_command": cmd = args.get("command", "") print(f"\n{C['y']} $ {cmd}{C['r']}") out = run_command(cmd) print(f" {out}") messages.append({"role": "tool", "content": out}) elif name == "web_search": q = args.get("query", "") print(f"\n{C['c']} [search] {q}{C['r']}") out = web_search(q) print(f" {out[:600]}") messages.append({"role": "tool", "content": out}) continue # re-send to model with tool results else: messages.append({"role": "assistant", "content": buf}) break print("\n") def main(): try: models = list_models() except Exception as e: print(f"{C['y']} can't reach Ollama at {OLLAMA}: {e}{C['r']}") sys.exit(1) if not models: print(f"{C['y']} no models on nightmare{C['r']}") sys.exit(1) model = pick_model(models) warm_model(model) chat(model) if __name__ == "__main__": main()