Files
astraea/polish.py
drjones 9f9d812320 Astraea v1.0 — multi-agent WA family-law assistant
Eight specialist agents over a 16-book verified WA law corpus (RAG with citations),
per-user document vault, WA court-form PDF auto-fill, comms missions with DV
safety guard, no-KYC auth, TTS. Self-hosted: Flask + SQLite + Ollama, stdlib-only RAG.

Includes README, LICENSE (MIT + not-legal-advice notice), DEPLOY runbook, .gitignore.
2026-09-07 18:49:32 -07:00

212 lines
8.6 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Astraea Polisher v2 — recursive qwen3.8fast self-improvement loop.
Goal: make the most visually appealing, premium, cinematic site possible.
Each iteration: read frontend -> qwen3.8fast critiques + rewrites toward GOAL
(contract = Jinja + chat API + external links) -> validate (JS/HTML id cross-check)
-> backup -> apply -> curl-verify -> auto-rollback on failure -> log. Retries a
failed generation up to 2x with targeted feedback.
"""
import json
import os
import re
import sys
import subprocess
import time
import urllib.request
from datetime import datetime
OLLAMA = "http://10.30.20.29:11434"
MODEL = "qwen3.8fast:latest"
TEMPLATE = "/opt/astraea/templates/index.html"
BACKUP_DIR = "/opt/astraea/polish-backups"
LOG_PATH = "/opt/astraea/polish-log.json"
ITERATIONS = int(sys.argv[1]) if len(sys.argv) > 1 else 6
GOAL = (
"Make this the most visually stunning, premium, cinematic, and emotionally "
"compelling family-law website on the internet. Obsess over typography "
"(scale, weight, letter-spacing), a refined dark color palette with a "
"confident accent, spacing and rhythm, motion and micro-interactions, depth "
"(gradients, glows, glass), and copy that is authoritative yet human. Target "
"the feel of a top-tier law firm's flagship site crossed with a futuristic AI "
"product — dark, elegant, trustworthy, memorable. Every iteration must be "
"strictly MORE appealing than the last."
)
CONTRACT = (
"HARD REQUIREMENTS — preserve these EXACTLY or the rewrite is invalid:\n"
"1. Keep the Jinja placeholders verbatim: `{{ agents | tojson }}` and "
"`{{ disclaimer | tojson }}`.\n"
"2. Keep `const AGENTS = {{ agents | tojson }};` and "
"`const DISCLAIMER = {{ disclaimer | tojson }};`.\n"
"3. Keep `fetch('/api/chat', ...)` and the request body {agent_id, message, history}.\n"
"4. Keep the Umami beacon `<script async src=\"https://analytics.thetempleofdoom.com/script.js\" "
"data-website-id=\"c8c2f69c-4448-40ab-9848-02179f10c001\"></script>`.\n"
"5. Keep the BMAC footer link to https://buymeacoffee.com/r26xrthzttg.\n"
"6. Keep every element id that the JavaScript references via getElementById / querySelector "
"matching the SAME id in the HTML (you may rename, but keep them consistent).\n"
"7. Output ONLY the complete valid HTML document (from <!DOCTYPE html> to </html>), "
"with no code fences and no commentary."
)
def _p(*a, **k):
print(*a, **k, flush=True)
def chat(system, user, max_tokens=14000, temperature=0.65):
payload = {
"model": MODEL,
"messages": [{"role": "system", "content": system},
{"role": "user", "content": user}],
"stream": False, "think": False,
"options": {"temperature": temperature, "num_predict": max_tokens, "num_ctx": 24000},
}
req = urllib.request.Request(f"{OLLAMA}/api/chat", data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"})
op = urllib.request.build_opener(urllib.request.ProxyHandler({}))
with op.open(req, timeout=360) as r:
d = json.loads(r.read().decode("utf-8"))
m = d.get("message", {})
return (m.get("content") or m.get("thinking") or "").strip()
def extract_html(text):
"""Pull the HTML document out of a model reply (strip fences + prose)."""
# strip a code fence if present (any language tag)
m = re.search(r"```\w*\s*\n?(.*?)```", text, re.DOTALL)
if m:
text = m.group(1)
# find the <html ...> ... </html> block (with or without DOCTYPE)
m = re.search(r"(<html[^>]*>.*?</html>)", text, re.DOTALL | re.IGNORECASE)
if m:
html = m.group(1)
if not html.lower().lstrip().startswith("<!doctype"):
html = "<!DOCTYPE html>\n" + html
return html.strip()
return ""
def validate(html):
"""Return (ok, [problems]). External contract + internal JS/HTML id consistency."""
problems = []
if "{{ agents | tojson }}" not in html:
problems.append("missing {{ agents | tojson }}")
if "{{ disclaimer | tojson }}" not in html:
problems.append("missing {{ disclaimer | tojson }}")
if "const AGENTS =" not in html:
problems.append("missing const AGENTS")
if "const DISCLAIMER =" not in html:
problems.append("missing const DISCLAIMER")
if "fetch('/api/chat'" not in html:
problems.append("missing fetch('/api/chat')")
if "c8c2f69c-4448-40ab-9848-02179f10c001" not in html:
problems.append("missing Umami id")
if "buymeacoffee.com/r26xrthzttg" not in html:
problems.append("missing BMAC link")
for tag in ("html", "body", "script", "style"):
if html.count(f"<{tag}") != html.count(f"</{tag}>"):
problems.append(f"unbalanced <{tag}> tags")
# JS/HTML id cross-check
scripts = "\n".join(re.findall(r"<script[^>]*>(.*?)</script>", html, re.DOTALL))
refs = set(re.findall(r"getElementById\(['\"]([^'\"]+)['\"]\)", scripts))
refs |= set(re.findall(r"querySelector\(['\"]#([^'\"]+)['\"]\)", scripts))
refs |= set(re.findall(r"querySelectorAll\(['\"]#([^'\"]+)['\"]\)", scripts))
for r in refs:
if not re.search(rf"id\s*=\s*[\"']{re.escape(r)}[\"']", html):
problems.append(f"JS refs #{r} but no matching id in HTML")
return (len(problems) == 0, problems)
def shell(cmd):
return subprocess.run(cmd, shell=True, capture_output=True, text=True)
def verify_live():
"""Check the served page returns 200 (stdlib urllib, no curl on the CT)."""
try:
req = urllib.request.Request("http://127.0.0.1:5000/")
op = urllib.request.build_opener(urllib.request.ProxyHandler({}))
with op.open(req, timeout=12) as r:
return r.status == 200
except Exception:
return False
def log_entry(entry):
log = []
if os.path.exists(LOG_PATH):
try:
log = json.load(open(LOG_PATH))
except Exception:
log = []
log.append(entry)
json.dump(log, open(LOG_PATH, "w"), indent=2)
def main():
os.makedirs(BACKUP_DIR, exist_ok=True)
_p(f"=== Astraea Polisher v2 — {ITERATIONS} iterations, {MODEL} ===")
_p(f"GOAL: {GOAL[:80]}...")
for it in range(1, ITERATIONS + 1):
_p(f"\n--- iteration {it}/{ITERATIONS} ---")
current = open(TEMPLATE, encoding="utf-8").read()
system = f"You are a world-class front-end design engineer. OBJECTIVE: {GOAL}\n\n{CONTRACT}"
applied = False
problems = []
for attempt in range(1, 4):
feedback = ""
if attempt > 1:
feedback = f"\n\nYour previous attempt was REJECTED for: {problems}. Fix ONLY those issues and resubmit the complete HTML."
user = (
"Here is the CURRENT full HTML/CSS/JS. Rewrite it into a COMPLETE, "
"significantly more appealing HTML document. Output ONLY the HTML — no "
"commentary, no explanation, no code fences."
f"{feedback}\n\n=== CURRENT HTML ===\n{current}"
)
try:
reply = chat(system, user)
except Exception as e:
_p(f" LLM error: {e}")
break
new_html = extract_html(reply)
if not new_html:
_p(f" attempt {attempt}: could not extract HTML (reply len {len(reply)})")
continue
ok, problems = validate(new_html)
if not ok:
_p(f" attempt {attempt}: validation failed ({len(problems)}): {problems[:4]}")
continue
# backup + apply
ts = datetime.now().strftime("%Y%m%d-%H%M%S")
backup = os.path.join(BACKUP_DIR, f"index.html.bak-{ts}")
open(backup, "w", encoding="utf-8").write(current)
open(TEMPLATE, "w", encoding="utf-8").write(new_html)
shell("systemctl restart astraea")
time.sleep(4)
if verify_live():
_p(f" ✓ applied ({len(new_html)-len(current):+d} chars) — {backup}")
log_entry({"ts": ts, "iter": it, "attempt": attempt,
"delta_chars": len(new_html) - len(current)})
applied = True
else:
open(TEMPLATE, "w", encoding="utf-8").write(current)
shell("systemctl restart astraea")
_p(" ✗ live verify failed — rolled back")
log_entry({"ts": ts, "iter": it, "rollback": True})
break
if not applied:
_p(" (no change this iteration)")
_p(f"\n=== done. log: {LOG_PATH} ===")
if __name__ == "__main__":
main()