Advanced mode: chat-driven form filling (fill_form tool), 10 official WA forms auto-grabbed + auto-mapped, doc vault, web search, per-user doc RAG

This commit is contained in:
drjones
2026-09-05 12:35:04 -07:00
parent b075699cb6
commit eadce36742
7 changed files with 863 additions and 14 deletions

View File

@@ -6,8 +6,8 @@ Each agent scopes a domain and pulls from its own set of 'books' (markdown sourc
# Backend Ollama (nightmare, RTX 4080 SUPER). Renumbers after reboot — .29 as of Sept 2026.
OLLAMA_URL = "http://10.30.20.29:11434"
EMBED_MODEL = "nomic-embed-text-v2-moe:latest" # semantic embeddings (verified on nightmare)
RAG_MODEL = "ornith-1.5:9b" # verified best legal-advice answer lane (benchmark Sept 2026)
GENERAL_MODEL = "granite4.2:latest" # fallback (reliable, grounded)
RAG_MODEL = "qwen3.8fast:latest" # tool-calling orchestrator (web search + comms + RAG answer)
GENERAL_MODEL = "ornith-1.5:9b" # fallback (reliable, grounded, also tool-capable)
BOOKS_DIR = "/opt/astraea/books"

309
app.py
View File

@@ -3,9 +3,13 @@
Multi-agent RAG + no-KYC profiles (About Me) + document prep + comms missions + TTS.
"""
import json
import os
import re
import threading
import time
import urllib.parse
import urllib.request
import uuid
from flask import Flask, jsonify, render_template, request, send_file, Response
@@ -13,7 +17,9 @@ import rag
import store
import comms
import documents
import forms
import tts
import userdocs
from agents import AGENTS, AGENT_BY_ID, OLLAMA_URL, RAG_MODEL, GENERAL_MODEL
app = Flask(__name__)
@@ -34,9 +40,76 @@ AVATARS = [
{"id": 11, "glyph": "\U0001F9ED", "c1": "#5eead4", "c2": "#0d9488", "label": "Compass"},
]
UPLOAD_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "uploads")
FILE_CATEGORIES = ["Court filings", "Correspondence", "Financial", "Agreements",
"Evidence", "Medical & safety", "Generated forms", "Other"]
def _docx_text(path):
import zipfile as _z
try:
with _z.ZipFile(path) as z:
xml = z.read("word/document.xml").decode("utf-8", "ignore")
out = []
for p in re.split(r"</w:p>", xml):
ts = re.findall(r"<w:t[^>]*>(.*?)</w:t>", p)
if ts:
out.append("".join(ts))
return "\n".join(out)
except Exception:
return ""
def _reindex_user(uid):
"""Rebuild a user's document vector index in a background thread."""
try:
userdocs.build(uid, store.list_files(uid), UPLOAD_DIR)
except Exception:
pass
def _autofill_values(uid, form_id):
"""Map the user's About Me + display name onto a form's fields via the LLM."""
spec = forms.FORM_NAMES.get(form_id)
fm = forms.FIELDMAP.get(form_id, [])
if not spec or not fm:
return {}
profile = store.get_profile(uid) if uid else {}
about = profile.get("about_me", "")
display = profile.get("display_name", "")
field_desc = ", ".join(f"{f['key']} ({f.get('label', '')})" for f in fm)
prompt = (
"You are filling a Washington State court form. Map the user's facts to the form fields.\n"
f"Form: {spec['name']}\n"
f"Fields: {field_desc}\n\n"
f"User's display name: {display or '(not set)'}\n"
f"User's information (About Me):\n{about or '(not provided)'}\n\n"
"Output ONLY valid JSON like {\"county\": \"...\", \"petitioner\": \"...\"}. "
"For any field you cannot determine, output an empty string \"\". "
"Petitioner is the user themself (use the display name); respondent is their spouse."
)
try:
resp = _ollama_chat(RAG_MODEL, [{"role": "user", "content": prompt}],
temperature=0.1, num_predict=800)
content = resp.get("message", {}).get("content", "")
m = re.search(r"\{.*\}", content, re.DOTALL)
vals = json.loads(m.group(0)) if m else {}
except Exception:
vals = {}
return {f["key"]: str(vals.get(f["key"], "") or "") for f in fm}
def _save_pdf_to_vault(uid, path, name):
data = open(path, "rb").read()
fn = f"gen_{uid}_{uuid.uuid4().hex[:8]}.pdf"
with open(os.path.join(UPLOAD_DIR, fn), "wb") as f:
f.write(data)
return store.add_file(uid, "Generated forms", fn, name + ".pdf",
"application/pdf", len(data))
# ── Ollama helpers ──
def _ollama_chat(model, messages, tools=None, num_ctx=16384, num_predict=1400,
def _ollama_chat(model, messages, tools=None, num_ctx=24000, num_predict=1400,
temperature=0.2, timeout=180):
payload = {"model": model, "messages": messages, "stream": False, "think": False,
"options": {"temperature": temperature, "num_predict": num_predict, "num_ctx": num_ctx}}
@@ -49,6 +122,23 @@ def _ollama_chat(model, messages, tools=None, num_ctx=16384, num_predict=1400,
return json.loads(r.read().decode("utf-8"))
def _web_search(query, limit=5):
"""Search the live web via local SearXNG (separate service, not the LLM)."""
try:
url = f"http://10.30.20.35:6969/search?q={urllib.parse.quote(query)}&format=json"
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Astraea)"})
op = urllib.request.build_opener(urllib.request.ProxyHandler({}))
with op.open(req, timeout=15) as r:
data = json.loads(r.read().decode("utf-8"))
out = []
for it in data.get("results", [])[:limit]:
out.append({"title": it.get("title", ""), "url": it.get("url", ""),
"snippet": (it.get("content") or "")[:400]})
return out
except Exception as e:
return [{"error": f"web search failed: {e}"}]
def _strip_sources(text):
text = re.split(r"\n\s*(?:Sources|References|Citations)\s*:\s*\n", text, flags=re.I)[0]
return text.strip()
@@ -88,6 +178,17 @@ TOOLS = [
"recipient": {"type": "string", "description": "recipient phone number in E.164"},
"message": {"type": "string", "description": "the message to speak"}},
"required": ["recipient", "message"]}}},
{"type": "function", "function": {
"name": "web_search", "description": "Search the live web for current facts, statutes, or answers not in your reference documents. Use when you don't know something or need up-to-date information.",
"parameters": {"type": "object", "properties": {
"query": {"type": "string", "description": "the search query"}},
"required": ["query"]}}},
{"type": "function", "function": {
"name": "fill_form", "description": "Fill an official Washington court form with the user's info and save the filled PDF to their case documents. Use when the user asks to fill out / prepare / complete paperwork. Form ids: fl200=Summons, fl201=Petition, fl001=Confidential Info, fl140=Parenting Plan, wscss=Child Support Worksheets, fl131=Financial Declaration, fl231=Findings & Conclusions, fl241=Final Divorce Order, fl211=Response, fl223=Motion for Temporary Order.",
"parameters": {"type": "object", "properties": {
"form_id": {"type": "string", "description": "the form id, e.g. fl201"},
"values": {"type": "object", "description": "optional explicit field values; omitted fields auto-fill from the user's profile"}},
"required": ["form_id"]}}},
]
@@ -108,6 +209,24 @@ def _execute_tool(name, args, settings, uid, display_name):
twiml_url = request.host_url.rstrip("/") + "/twilio/voice"
comms.make_call(settings, recipient, twiml_url)
return f"Call initiated to {recipient}."
if name == "web_search":
res = _web_search(args.get("query", ""), limit=5)
if not res:
return "No results found."
return json.dumps(res, ensure_ascii=False)
if name == "fill_form":
form_id = args.get("form_id", "")
if not forms.FORM_NAMES.get(form_id):
return "Unknown form. Use one of: " + ", ".join(forms.FORM_NAMES.keys())
values = args.get("values") or {}
auto = _autofill_values(uid, form_id)
merged = {k: str(values.get(k) or auto.get(k) or "") for k in auto}
path = forms.fill_form(form_id, merged)
if not path:
return "Could not generate the form."
_save_pdf_to_vault(uid, path, forms.FORM_NAMES[form_id]["name"])
return (f"Filled '{forms.FORM_NAMES[form_id]['name']}' and saved it to the user's "
f"Case Documents. Tell the user to open their profile → Case documents to download it.")
except Exception as e:
return f"ERROR sending {name}: {e}"
return "Unknown tool."
@@ -149,6 +268,16 @@ def _build_answer(agent, message, history, about_me, settings, uid, display_name
citations.append({"source": c["source"], "title": c["title"],
"score": round(score, 3), "snippet": c["text"][:260]})
# user's own uploaded documents (per-user vector index) — retrieve alongside case law
if uid:
try:
for d in userdocs.retrieve(uid, message, top_k=4):
context_blocks.append(f"[USER DOC — {d['source']}]\n{d['text']}")
citations.append({"source": d["source"], "title": "your document",
"score": d["score"], "snippet": d["text"][:260]})
except Exception:
pass
system = (
f"{agent['system']}\n\n"
"Rules:\n"
@@ -158,8 +287,12 @@ def _build_answer(agent, message, history, about_me, settings, uid, display_name
"- Ground every legal claim in the reference documents. Cite the RCW section AND the "
"controlling case law by name.\n"
"- Be plain-English, specific to Washington State.\n"
"- Items marked '[USER DOC]' are the user's own uploaded files — reference them by name "
"to ground your answer in their actual case.\n"
"- You may use the available tools (send_sms / send_email / make_call) ONLY when the "
"user explicitly asks you to contact someone on their behalf.\n"
"- If you don't know the answer, or need current/up-to-date facts not in the reference "
"documents, use the web_search tool and answer from those results.\n"
+ about_me_block(about_me)
+ safety_block(settings)
)
@@ -283,6 +416,180 @@ def api_profile():
return jsonify(prof)
@app.route("/api/profile/pic", methods=["POST"])
def api_profile_pic():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
f = request.files.get("file")
if not f or not f.filename:
return jsonify({"error": "no file"}), 400
data = f.read()
if len(data) > 5 * 1024 * 1024:
return jsonify({"error": "image too large (max 5 MB)"}), 400
ext = os.path.splitext(f.filename)[1].lower()
if ext not in (".png", ".jpg", ".jpeg", ".gif", ".webp"):
return jsonify({"error": "unsupported type — use PNG/JPG/GIF/WebP"}), 400
up = os.path.join(os.path.dirname(os.path.abspath(__file__)), "static", "uploads")
os.makedirs(up, exist_ok=True)
fn = f"u{uid}{ext}"
with open(os.path.join(up, fn), "wb") as out:
out.write(data)
prof = store.set_profile(uid, profile_pic=f"/static/uploads/{fn}")
return jsonify(prof)
# ── Document vault ──
@app.route("/api/files", methods=["GET", "POST"])
def api_files():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
if request.method == "GET":
return jsonify({"categories": FILE_CATEGORIES, "files": store.list_files(uid)})
cat = request.form.get("category", "Other")
if cat not in FILE_CATEGORIES:
cat = "Other"
f = request.files.get("file")
if not f or not f.filename:
return jsonify({"error": "no file"}), 400
data = f.read()
if len(data) > 25 * 1024 * 1024:
return jsonify({"error": "file too large (max 25 MB)"}), 400
os.makedirs(UPLOAD_DIR, exist_ok=True)
ext = os.path.splitext(f.filename)[1].lower()[:12]
fn = f"{uid}_{uuid.uuid4().hex[:12]}{ext}"
with open(os.path.join(UPLOAD_DIR, fn), "wb") as out:
out.write(data)
fid = store.add_file(uid, cat, fn, f.filename, f.mimetype or "", len(data))
threading.Thread(target=_reindex_user, args=(uid,), daemon=True).start()
return jsonify({"ok": True, "id": fid, "category": cat})
@app.route("/api/files/reindex", methods=["POST"])
def api_files_reindex():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
n = userdocs.build(uid, store.list_files(uid), UPLOAD_DIR)
return jsonify({"ok": True, "chunks": n})
@app.route("/api/files/status")
def api_files_status():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
return jsonify(userdocs.status(uid))
@app.route("/api/files/<int:fid>/raw")
def api_file_raw(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if not rec:
return jsonify({"error": "not found"}), 404
path = os.path.join(UPLOAD_DIR, rec["filename"])
if not os.path.exists(path):
return jsonify({"error": "missing"}), 404
return send_file(path, mimetype=rec["mime"] or "application/octet-stream",
as_attachment=False, download_name=rec["orig_name"])
@app.route("/api/files/<int:fid>/text")
def api_file_text(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if not rec:
return jsonify({"error": "not found"}), 404
path = os.path.join(UPLOAD_DIR, rec["filename"])
ext = os.path.splitext(rec["orig_name"])[1].lower()
if ext == ".docx":
return jsonify({"text": _docx_text(path), "mime": "text/plain"})
if ext in (".txt", ".md", ".csv", ".log", ".json", ".html", ".xml"):
try:
with open(path, "r", encoding="utf-8", errors="replace") as fh:
return jsonify({"text": fh.read()[:200000], "mime": "text/plain"})
except Exception:
return jsonify({"text": "", "mime": "text/plain"})
return jsonify({"text": "", "mime": rec["mime"] or ""})
@app.route("/api/files/<int:fid>", methods=["DELETE"])
def api_file_delete(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if rec:
p = os.path.join(UPLOAD_DIR, rec["filename"])
if os.path.exists(p):
try:
os.remove(p)
except Exception:
pass
store.delete_file(uid, fid)
return jsonify({"ok": True})
# ── Form filling (official WA court forms via pymupdf overlay) ──
@app.route("/api/forms")
def api_forms():
return jsonify(forms.list_forms())
@app.route("/api/forms/autofill", methods=["POST"])
def api_forms_autofill():
uid = _uid()
d = request.get_json(force=True, silent=True) or {}
form_id = d.get("form_id", "")
spec = forms.FORM_NAMES.get(form_id)
fm = forms.FIELDMAP.get(form_id, [])
if not spec:
return jsonify({"error": "unknown form"}), 400
profile = store.get_profile(uid) if uid else {}
about = profile.get("about_me", "")
display = profile.get("display_name", "")
field_desc = ", ".join(f"{f['key']} ({f.get('label','')})" for f in fm)
prompt = (
"You are filling a Washington State court form. Map the user's facts to the form fields.\n"
f"Form: {spec['name']}\n"
f"Fields: {field_desc}\n\n"
f"User's display name: {display or '(not set)'}\n"
f"User's information (About Me):\n{about or '(not provided)'}\n\n"
"Output ONLY valid JSON like {\"county\": \"...\", \"petitioner\": \"...\"}. "
"For any field you cannot determine, output an empty string. "
"Petitioner is the user themself (use the display name); respondent is their spouse."
)
try:
resp = _ollama_chat(RAG_MODEL, [{"role": "user", "content": prompt}],
temperature=0.1, num_predict=800)
content = resp.get("message", {}).get("content", "")
m = re.search(r"\{.*\}", content, re.DOTALL)
vals = json.loads(m.group(0)) if m else {}
except Exception:
vals = {}
known = {f["key"]: str(vals.get(f["key"], "") or "") for f in fm}
return jsonify({"form_id": form_id, "values": known})
@app.route("/api/forms/fill", methods=["POST"])
def api_forms_fill():
uid = _uid()
d = request.get_json(force=True, silent=True) or {}
form_id = d.get("form_id", "")
values = d.get("values", {})
path = forms.fill_form(form_id, values)
if not path:
return jsonify({"error": "could not fill form"}), 400
return send_file(path, mimetype="application/pdf", as_attachment=True,
download_name=f"{form_id}_filled.pdf")
@app.route("/api/settings", methods=["GET", "POST"])
def api_settings():
uid = _uid()

122
detect_fields.py Normal file
View File

@@ -0,0 +1,122 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Detect fillable blanks across WA court forms: underscore runs + the standard
caption (county/petitioner/respondent/case_no). Emits a field-map JSON."""
import json
import os
import re
import sys
import pymupdf
FORMS_DIR = "/opt/astraea/forms"
def lines_with_pos(page):
words = page.get_text("words")
groups = {}
for w in words:
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
key = int(y0 / 5) * 5
groups.setdefault(key, []).append((x0, x1, word))
out = []
for y in sorted(groups):
ws = sorted(groups[y], key=lambda t: t[0])
text = " ".join(w for _, _, w in ws)
out.append({"y": y, "x0": ws[0][0], "x1": ws[-1][1], "text": text, "words": ws})
return out
def detect_caption(lines):
"""Find the standard caption fields. Returns list of {key, label, page, x, y}."""
fields = []
for ln in lines:
t = ln["text"]
# "County of" -> blank right after 'of'
if re.search(r"\bCounty\b.*\bof\b\s*$", t) and "Superior" in t:
ofx = None
for x0, x1, w in ln["words"]:
if w == "of":
ofx = x1
if ofx:
fields.append({"key": "county", "label": "County", "x": ofx + 6, "y": ln["y"] + 9})
# Petitioner ... case): -> name after the colon
if "Petitioner" in t and ("case" in t or "case):" in t) and ":" in t:
colonx = None
for x0, x1, w in ln["words"]:
if w.endswith(":"):
colonx = x1
if colonx and colonx < 400:
fields.append({"key": "petitioner", "label": "Petitioner (your full name)", "x": colonx + 6, "y": ln["y"] + 9})
# Respondent ... partner): -> name after colon
if "Respondent" in t and ":" in t:
colonx = None
for x0, x1, w in ln["words"]:
if w.endswith(":"):
colonx = x1
if colonx and colonx < 400:
fields.append({"key": "respondent", "label": "Respondent (spouse's full name)", "x": colonx + 6, "y": ln["y"] + 9})
# Case No.
if "Case No." in t:
nox = None
for x0, x1, w in ln["words"]:
if w in ("No.", "No"):
nox = x1
if nox:
fields.append({"key": "case_no", "label": "Case number", "x": nox + 6, "y": ln["y"] + 9})
return fields
def detect_underscores(pno, page):
"""Find underscore runs (blank fields) and their preceding label."""
fields = []
words = page.get_text("words")
groups = {}
for w in words:
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
key = int(y0 / 5) * 5
groups.setdefault(key, []).append((x0, x1, y0, word))
for y in sorted(groups):
ws = sorted(groups[y], key=lambda t: t[0])
for i, (x0, x1, y0, word) in enumerate(ws):
if word.strip("_").strip() == "" and len(word.strip()) >= 4:
label = " ".join(w for _, _, _, w in ws[:i]).strip()
fields.append({"page": pno, "x": x0, "y": y + 9,
"label": label, "type": "text"})
return fields
def main():
all_fields = {}
for fn in sorted(os.listdir(FORMS_DIR)):
if not fn.endswith(".pdf"):
continue
fid = fn[:-4]
path = os.path.join(FORMS_DIR, fn)
try:
d = pymupdf.open(path)
except Exception as e:
print(f"{fid}: OPEN FAIL {e}", file=sys.stderr)
continue
fields = []
seen_keys = set()
for pno in range(min(len(d), 3)): # caption + first 2 pages
lines = lines_with_pos(d[pno])
if pno == 0:
for f in detect_caption(lines):
f["page"] = 0
if f["key"] not in seen_keys:
seen_keys.add(f["key"])
fields.append(f)
for f in detect_underscores(pno, d[pno]):
f["key"] = "f_" + str(len(fields))
fields.append(f)
d.close()
all_fields[fid] = fields
out = "/opt/astraea/forms/fieldmap.json"
with open(out, "w") as f:
json.dump(all_fields, f, indent=1)
print(f"saved {out}: { {k: len(v) for k, v in all_fields.items()} }")
if __name__ == "__main__":
main()

88
forms.py Normal file
View File

@@ -0,0 +1,88 @@
# -*- coding: utf-8 -*-
"""Astraea form filling — overlay the user's answers onto official WA court forms
with pymupdf (no PDF MCP exists; pymupdf is the editor). Field maps are
auto-detected from the flat forms and cached in forms/fieldmap.json.
"""
import json
import os
import uuid
import pymupdf
FORMS_DIR = "/opt/astraea/forms"
FIELDMAP_PATH = os.path.join(FORMS_DIR, "fieldmap.json")
FORM_NAMES = {
"fl200": {"name": "Summons (FL Divorce 200)", "file": "fl200.pdf",
"description": "Notifies your spouse the divorce case has started."},
"fl201": {"name": "Petition for Dissolution (FL Divorce 201)", "file": "fl201.pdf",
"description": "The main form that starts your divorce case."},
"fl001": {"name": "Confidential Information (FL All Family 001)", "file": "fl001.pdf",
"description": "Private contact/safety info for the court."},
"fl140": {"name": "Parenting Plan (FL All Family 140)", "file": "fl140.pdf",
"description": "Residential schedule and decision-making for your children."},
"wscss": {"name": "Child Support Worksheets (WSCSS)", "file": "wscss.pdf",
"description": "Calculates child support under RCW 26.19."},
"fl131": {"name": "Financial Declaration (FL All Family 131)", "file": "fl131.pdf",
"description": "Your income, assets, and debts."},
"fl231": {"name": "Findings & Conclusions (FL Divorce 231)", "file": "fl231.pdf",
"description": "The court's factual findings."},
"fl241": {"name": "Final Divorce Order (FL Divorce 241)", "file": "fl241.pdf",
"description": "The final decree ending your marriage."},
"fl211": {"name": "Response to Petition (FL Divorce 211)", "file": "fl211.pdf",
"description": "Your reply to your spouse's petition."},
"fl223": {"name": "Motion for Temporary Order (FL Divorce 223)", "file": "fl223.pdf",
"description": "Ask for temporary custody, support, or protection orders."},
}
def _load_fieldmap():
try:
with open(FIELDMAP_PATH) as f:
return json.load(f)
except Exception:
return {}
FIELDMAP = _load_fieldmap()
def list_forms():
out = []
for fid, spec in FORM_NAMES.items():
fm = FIELDMAP.get(fid, [])
fields = [{"key": f["key"], "label": f.get("label", f["key"])} for f in fm]
out.append({"id": fid, "name": spec["name"], "description": spec["description"],
"fields": fields})
return out
def fill_form(form_id, values):
"""Overlay `values` (dict field_key -> text) onto the form, return the filled PDF path."""
spec = FORM_NAMES.get(form_id)
fm = FIELDMAP.get(form_id, [])
if not spec:
return None
src = os.path.join(FORMS_DIR, spec["file"])
if not os.path.exists(src):
return None
fmap = {f["key"]: f for f in fm}
doc = pymupdf.open(src)
for key, text in values.items():
f = fmap.get(key)
if not f or not text:
continue
text = str(text).strip()
if not text:
continue
page = doc[f.get("page", 0)]
page.insert_text((f["x"], f["y"]), text, fontsize=f.get("size", 11),
fontname="helv", color=(0, 0, 0))
out = os.path.join("/tmp", f"filled_{form_id}_{uuid.uuid4().hex[:8]}.pdf")
doc.save(out)
doc.close()
return out
def available_forms():
return list(FORM_NAMES.keys())

View File

@@ -35,6 +35,7 @@ def init_db():
display_name TEXT DEFAULT '',
avatar INTEGER DEFAULT 0,
about_me TEXT DEFAULT '',
profile_pic TEXT DEFAULT '',
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
);
CREATE TABLE IF NOT EXISTS settings (
@@ -63,6 +64,16 @@ def init_db():
citations TEXT DEFAULT '',
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
);
CREATE TABLE IF NOT EXISTS files (
id INTEGER PRIMARY KEY AUTOINCREMENT,
user_id INTEGER NOT NULL,
category TEXT DEFAULT 'Other',
filename TEXT NOT NULL,
orig_name TEXT NOT NULL,
mime TEXT DEFAULT '',
size INTEGER DEFAULT 0,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
);
""")
c.commit()
# migrations for pre-existing DBs
@@ -74,6 +85,11 @@ def init_db():
c.commit()
except Exception:
pass
try:
c.execute("ALTER TABLE profiles ADD COLUMN profile_pic TEXT DEFAULT ''")
c.commit()
except Exception:
pass
c.close()
@@ -127,26 +143,27 @@ def get_profile(uid):
row = c.execute("SELECT * FROM profiles WHERE user_id=?", (uid,)).fetchone()
c.close()
if not row:
return {"display_name": "", "avatar": 0, "about_me": ""}
return {"display_name": "", "avatar": 0, "about_me": "", "profile_pic": ""}
return {"display_name": row["display_name"], "avatar": row["avatar"],
"about_me": row["about_me"]}
"about_me": row["about_me"], "profile_pic": row["profile_pic"] or ""}
def set_profile(uid, display_name=None, avatar=None, about_me=None):
def set_profile(uid, display_name=None, avatar=None, about_me=None, profile_pic=None):
cur = get_profile(uid)
dn = cur["display_name"] if display_name is None else display_name
av = cur["avatar"] if avatar is None else avatar
am = cur["about_me"] if about_me is None else about_me
pp = cur.get("profile_pic", "") if profile_pic is None else profile_pic
c = _conn()
c.execute("""INSERT INTO profiles (user_id, display_name, avatar, about_me, updated_at)
VALUES (?,?,?,?, CURRENT_TIMESTAMP)
c.execute("""INSERT INTO profiles (user_id, display_name, avatar, about_me, profile_pic, updated_at)
VALUES (?,?,?,?,?, CURRENT_TIMESTAMP)
ON CONFLICT(user_id) DO UPDATE SET
display_name=excluded.display_name, avatar=excluded.avatar,
about_me=excluded.about_me, updated_at=CURRENT_TIMESTAMP""",
(uid, dn, av, am))
about_me=excluded.about_me, profile_pic=excluded.profile_pic, updated_at=CURRENT_TIMESTAMP""",
(uid, dn, av, am, pp))
c.commit()
c.close()
return {"display_name": dn, "avatar": av, "about_me": am}
return {"display_name": dn, "avatar": av, "about_me": am, "profile_pic": pp}
def get_settings(uid):
@@ -203,3 +220,35 @@ def set_settings(uid, **fields):
c.commit()
c.close()
return get_settings(uid)
# ── Document vault ──
def add_file(uid, category, filename, orig_name, mime, size):
c = _conn()
c.execute("INSERT INTO files (user_id, category, filename, orig_name, mime, size) VALUES (?,?,?,?,?,?)",
(uid, category, filename, orig_name, mime, size))
c.commit()
fid = c.execute("SELECT last_insert_rowid()").fetchone()[0]
c.close()
return fid
def list_files(uid):
c = _conn()
rows = c.execute("SELECT * FROM files WHERE user_id=? ORDER BY created_at DESC, id DESC", (uid,)).fetchall()
c.close()
return [dict(r) for r in rows]
def get_file(uid, fid):
c = _conn()
row = c.execute("SELECT * FROM files WHERE id=? AND user_id=?", (fid, uid)).fetchone()
c.close()
return dict(row) if row else None
def delete_file(uid, fid):
c = _conn()
c.execute("DELETE FROM files WHERE id=? AND user_id=?", (fid, uid))
c.commit()
c.close()

View File

@@ -83,6 +83,18 @@
.avatar-grid { display:grid; grid-template-columns:repeat(6,1fr); gap:10px; margin-bottom:18px; }
.avatar-opt { aspect-ratio:1; border-radius:50%; display:grid; place-items:center; font-size:22px; color:#fff; cursor:pointer; border:2px solid transparent; }
.avatar-opt.sel { border-color:#fff; box-shadow:0 0 0 3px rgba(255,255,255,.2); }
.progbar { height:6px; background:var(--panel2); border-radius:3px; overflow:hidden; margin:14px 0 18px; }
.progbar > div { height:100%; background:linear-gradient(90deg,#6ea8ff,#a78bfa); border-radius:3px; transition:width .2s; }
.intake-q { font-size:16px; font-weight:600; margin-bottom:12px; line-height:1.4; }
.intake-nav { display:flex; justify-content:space-between; margin-top:14px; gap:10px; }
.intake-nav .btn { flex:1; }
.modal.wide { width:min(880px,94vw); max-height:90vh; overflow:auto; }
.file-list { max-height:240px; overflow:auto; display:flex; flex-direction:column; gap:6px; }
.file-item { display:flex; align-items:center; gap:8px; padding:8px 10px; background:var(--panel2); border-radius:8px; font-size:13px; }
.file-item .nm { flex:1; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; }
.file-item .cat { color:var(--muted); font-size:11px; }
.file-item button { background:none; border:1px solid var(--line); color:var(--text); border-radius:6px; padding:4px 9px; cursor:pointer; font-size:12px; }
.file-item button:hover { background:var(--line); }
.switch { display:flex; align-items:center; gap:10px; }
.switch input { width:auto; }
.grid2 { display:grid; grid-template-columns:1fr 1fr; gap:13px; }
@@ -121,6 +133,7 @@
<div class="messages" id="messages"></div>
<div class="tools-row">
<button class="btn" onclick="openDocs()">📄 Prep a document</button>
<button class="btn" onclick="openFill()">📝 Fill a form</button>
<button class="btn" onclick="openMission()">📨 Send on my behalf</button>
<button class="btn" onclick="clearConv()">🗑 Clear chat</button>
</div>
@@ -141,10 +154,45 @@
<div class="fld"><label>Display name</label><input id="pfName" placeholder="Your name"></div>
<div class="fld"><label>Display picture</label></div>
<div class="avatar-grid" id="avatarGrid"></div>
<div class="fld"><label>About Me — everything your attorney should know (your story, dates, names, what you're going through)</label><textarea id="pfAbout" placeholder="e.g. Married 2011, two children (ages 9 and 12), separated June 2025, own a home in King County, spouse filed a parenting plan…"></textarea></div>
<div class="fld"><label>Or upload your own photo</label><input type="file" id="pfPic" accept="image/*" onchange="uploadPic()"><span id="pfPicPreview"></span></div>
<div class="fld"><label>About Me — everything your attorney should know</label><button class="btn" style="width:100%;margin-bottom:8px" onclick="openIntake()">✍️ Guided intake — answer questions step by step</button><textarea id="pfAbout" placeholder="…or write it freely here. e.g. Married 2011, two children (ages 9 and 12), separated June 2025, own a home in King County, spouse filed a parenting plan…"></textarea></div>
<div class="fld"><label>📁 Case documents — upload & organize your paperwork</label>
<div style="display:flex;gap:8px;margin-bottom:8px"><select id="fileCat" style="width:auto"></select><input type="file" id="fileInput" multiple onchange="uploadFiles()" style="flex:1"></div>
<div id="fileList" class="file-list"><div class="sub">No documents yet — upload your petition, orders, agreements, and evidence.</div></div>
</div>
<button class="btn primary" style="width:100%" onclick="saveProfile()">Save profile</button>
</div></div>
<!-- Guided intake modal -->
<div class="modal-bg" id="intakeModal"><div class="modal">
<button class="close" onclick="closeModal('intakeModal')">×</button>
<h2>Guided intake</h2><div class="sub">Answer a few questions — your attorney uses these as context for every answer.</div>
<div class="progbar"><div id="intakeBar"></div></div>
<div class="intake-q" id="intakeQ"></div>
<input id="intakeA" placeholder="">
<div class="intake-nav">
<button class="btn" id="intakeBack" onclick="intakePrev()">← Back</button>
<button class="btn primary" id="intakeNext" onclick="intakeNext()">Next →</button>
</div>
</div></div>
<!-- Document viewer modal -->
<div class="modal-bg" id="viewerModal"><div class="modal wide">
<button class="close" onclick="closeModal('viewerModal')">×</button>
<h2 id="viewerTitle"></h2>
<div id="viewerBody" style="margin-top:12px"></div>
</div></div>
<!-- Fill a form modal -->
<div class="modal-bg" id="fillModal"><div class="modal">
<button class="close" onclick="closeModal('fillModal')">×</button>
<h2>Fill a court form</h2><div class="sub">Your answers are typed onto the official WA form PDF, ready to print and file.</div>
<div class="fld"><label>Form</label><select id="fillFormSelect"></select></div>
<button class="btn" style="width:100%;margin-bottom:12px" onclick="autofillForm()">✨ Auto-fill from my info</button>
<div id="fillFields"></div>
<button class="btn primary" style="width:100%" onclick="generateForm()">⬇ Generate filled PDF</button>
</div></div>
<!-- Settings modal -->
<div class="modal-bg" id="settingsModal"><div class="modal">
<button class="close" onclick="closeModal('settingsModal')">×</button>
@@ -284,8 +332,126 @@
async function loadAvatars(){ const r=await fetch('/api/avatars'); avatars=await r.json(); const g=document.getElementById('avatarGrid'); g.innerHTML=''; avatars.forEach(a=>{ const d=document.createElement('div'); d.className='avatar-opt'; d.style.background=`linear-gradient(135deg,${a.c1},${a.c2})`; d.textContent=a.glyph; d.title=a.label; d.onclick=()=>{ myProfile.avatar=a.id; renderAvatarSel(); }; g.appendChild(d); }); }
function renderAvatarSel(){ document.querySelectorAll('.avatar-opt').forEach((d,i)=>d.classList.toggle('sel',i===myProfile.avatar)); }
function renderMyAvatar(){ const a=avatars[myProfile.avatar]||avatars[0]; const el=document.getElementById('myAvatar'); if(a){ el.style.background=`linear-gradient(135deg,${a.c1},${a.c2})`; el.textContent=a.glyph; } }
function openProfile(){ document.getElementById('pfName').value=myProfile.display_name||''; document.getElementById('pfAbout').value=myProfile.about_me||''; renderAvatarSel(); document.getElementById('profileModal').classList.add('open'); }
function renderMyAvatar(){ const el=document.getElementById('myAvatar'); if(myProfile.profile_pic){ el.textContent=''; el.style.background=`url('${myProfile.profile_pic}') center/cover`; } else { const a=avatars[myProfile.avatar]||avatars[0]; if(a){ el.style.background=`linear-gradient(135deg,${a.c1},${a.c2})`; el.textContent=a.glyph; } } }
async function uploadPic(){
const f=document.getElementById('pfPic').files[0];
if(!f) return;
const fd=new FormData(); fd.append('file',f);
const pv=document.getElementById('pfPicPreview'); pv.textContent='Uploading…';
try{ const r=await fetch('/api/profile/pic',{method:'POST',headers:authHeaders(),body:fd}); const d=await r.json();
if(d.error){ pv.textContent=d.error; pv.style.color='#f87171'; return; }
myProfile=d; renderMyAvatar(); pv.textContent='✓ uploaded'; pv.style.color='#4ade80';
}catch(e){ pv.textContent='Upload failed'; pv.style.color='#f87171'; }
}
const INTAKE=[
{label:"My name", q:"What's your full name?", p:"Jane Marie Doe"},
{label:"Spouse's name", q:"What's your spouse's full name?", p:"John Robert Doe"},
{label:"Married", q:"When and where did you get married?", p:"June 15, 2011, Seattle, WA"},
{label:"Separated", q:"When did you separate?", p:"June 2025"},
{label:"Children", q:"How many children do you share, and their names and ages?", p:"Two — Emma (12) and Liam (9)"},
{label:"Current custody", q:"Where do the children live now, and what's the current schedule?", p:"They live with me; dad has them every other weekend"},
{label:"Home", q:"Do you own a home or other real estate? Address, when bought, whose name is on the title/mortgage?", p:"House at 123 Main St, bought 2012, both names on title"},
{label:"Assets", q:"What are your main assets? (bank accounts, retirement, vehicles, business, investments)", p:"Joint savings ~$20k, my 401k ~$85k, two cars"},
{label:"Debts", q:"What are your main debts? (mortgage, credit cards, loans)", p:"Mortgage $280k remaining, $9k credit card, $12k car loan"},
{label:"Income", q:"What's your employment and income? And your spouse's?", p:"I make $72k/yr as a nurse; spouse makes $95k as an engineer"},
{label:"Case status", q:"Has either of you filed for divorce yet, and what does each side want?", p:"Spouse filed; wants full custody and to keep the house"},
{label:"Safety", q:"Is there any domestic violence, abuse, or safety concern? Any protection orders?", p:"No / Yes — details"},
{label:"Goals", q:"What outcome matters most to you? What are you most worried about?", p:"Keeping the kids with me and keeping the house"},
];
let intakeIdx=0; const intakeAnswers={};
function openIntake(){ intakeIdx=0; for(const k in intakeAnswers) delete intakeAnswers[k]; closeModal('profileModal'); renderIntake(); document.getElementById('intakeModal').classList.add('open'); }
function renderIntake(){ const q=INTAKE[intakeIdx]; document.getElementById('intakeQ').textContent=(intakeIdx+1)+'. '+q.q; const a=document.getElementById('intakeA'); a.value=intakeAnswers[intakeIdx]||''; a.placeholder=q.p; document.getElementById('intakeBar').style.width=Math.round((intakeIdx/(INTAKE.length-1))*100)+'%'; document.getElementById('intakeBack').style.visibility=intakeIdx===0?'hidden':'visible'; document.getElementById('intakeNext').textContent=intakeIdx===INTAKE.length-1?'Finish ✓':'Next →'; }
function intakeNext(){ intakeAnswers[intakeIdx]=document.getElementById('intakeA').value.trim(); if(intakeIdx<INTAKE.length-1){ intakeIdx++; renderIntake(); } else { finishIntake(); } }
function intakePrev(){ intakeAnswers[intakeIdx]=document.getElementById('intakeA').value.trim(); if(intakeIdx>0){ intakeIdx--; renderIntake(); } }
function finishIntake(){
const parts=[]; INTAKE.forEach((q,i)=>{ const v=(intakeAnswers[i]||'').trim(); if(v) parts.push(q.label+': '+v); });
document.getElementById('pfAbout').value=parts.join('\n');
closeModal('intakeModal'); document.getElementById('profileModal').classList.add('open');
}
let myFiles=[], myCats=['Other'];
function escapeHtml(s){ return (s||'').replace(/[&<>"]/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;'}[c])); }
function fmtSize(n){ return n>1048576?(n/1048576).toFixed(1)+' MB':n>1024?(n/1024).toFixed(0)+' KB':(n||0)+' B'; }
async function loadFiles(){
if(!token) return;
try{ const r=await fetch('/api/files',{headers:authHeaders()}); const d=await r.json(); myFiles=d.files||[]; myCats=d.categories||['Other'];
const sel=document.getElementById('fileCat'); if(sel){ sel.innerHTML=''; myCats.forEach(c=>{ const o=document.createElement('option'); o.value=c; o.textContent=c; sel.appendChild(o); }); }
renderFiles();
}catch(e){}
}
function renderFiles(){
const el=document.getElementById('fileList'); if(!el) return;
if(!myFiles.length){ el.innerHTML='<div class="sub">No documents yet — upload your petition, orders, agreements, and evidence.</div>'; return; }
const groups={}; myFiles.forEach(f=>{ (groups[f.category]=groups[f.category]||[]).push(f); });
el.innerHTML='';
Object.keys(groups).forEach(cat=>{
const h=document.createElement('div'); h.style.cssText='color:var(--gold);font-size:12px;font-weight:700;margin-top:10px'; h.textContent=cat+' ('+groups[cat].length+')'; el.appendChild(h);
groups[cat].forEach(f=>{
const it=document.createElement('div'); it.className='file-item';
const nm=document.createElement('span'); nm.className='nm'; nm.textContent=f.orig_name; it.appendChild(nm);
const sz=document.createElement('span'); sz.className='cat'; sz.textContent=fmtSize(f.size); it.appendChild(sz);
const v=document.createElement('button'); v.textContent='View'; v.onclick=()=>viewFile(f.id); it.appendChild(v);
const d=document.createElement('button'); d.textContent='✕'; d.onclick=()=>delFile(f.id); it.appendChild(d);
el.appendChild(it);
});
});
}
async function uploadFiles(){
const fs=document.getElementById('fileInput').files; if(!fs.length) return;
const cat=document.getElementById('fileCat').value;
for(const f of fs){ const fd=new FormData(); fd.append('file',f); fd.append('category',cat); try{ await fetch('/api/files',{method:'POST',headers:authHeaders(),body:fd}); }catch(e){} }
document.getElementById('fileInput').value='';
await loadFiles();
}
async function delFile(id){ try{ await fetch('/api/files/'+id,{method:'DELETE',headers:authHeaders()}); }catch(e){} await loadFiles(); }
async function viewFile(id){
const rec=myFiles.find(f=>f.id===id); if(!rec) return;
const ext=(rec.orig_name.split('.').pop()||'').toLowerCase();
document.getElementById('viewerTitle').textContent=rec.orig_name;
const body=document.getElementById('viewerBody'); body.innerHTML='<div class="sub">Loading…</div>';
document.getElementById('viewerModal').classList.add('open');
if(['png','jpg','jpeg','gif','webp','svg'].includes(ext)){
const r=await fetch('/api/files/'+id+'/raw',{headers:authHeaders()}); const b=await r.blob(); const u=URL.createObjectURL(b);
body.innerHTML='<img src="'+u+'" style="max-width:100%;border-radius:8px">';
} else if(ext==='pdf'){
const r=await fetch('/api/files/'+id+'/raw',{headers:authHeaders()}); const b=await r.blob(); const u=URL.createObjectURL(b);
body.innerHTML='<iframe src="'+u+'" style="width:100%;height:70vh;border:0;border-radius:8px"></iframe>';
} else {
let txt=''; try{ const r=await fetch('/api/files/'+id+'/text',{headers:authHeaders()}); const d=await r.json(); txt=d.text||''; }catch(e){}
body.innerHTML='<pre style="white-space:pre-wrap;font-family:inherit;font-size:13px;line-height:1.55">'+escapeHtml(txt||'(no text preview — this file type downloads instead)')+'</pre>';
}
}
function openProfile(){ document.getElementById('pfName').value=myProfile.display_name||''; document.getElementById('pfAbout').value=myProfile.about_me||''; renderAvatarSel(); document.getElementById('profileModal').classList.add('open'); loadFiles(); }
let allForms=[];
async function openFill(){
try{ const r=await fetch('/api/forms'); allForms=await r.json();
const sel=document.getElementById('fillFormSelect'); sel.innerHTML='';
allForms.forEach(f=>{ const o=document.createElement('option'); o.value=f.id; o.textContent=f.name; sel.appendChild(o); });
document.getElementById('fillFields').innerHTML='<div class="sub">Click "Auto-fill from my info" to pre-fill the fields.</div>';
document.getElementById('fillModal').classList.add('open');
}catch(e){}
}
async function autofillForm(){
const fid=document.getElementById('fillFormSelect').value;
const el=document.getElementById('fillFields'); el.innerHTML='<div class="sub">Filling…</div>';
const r=await fetch('/api/forms/autofill',{method:'POST',headers:{'Content-Type':'application/json',...authHeaders()},body:JSON.stringify({form_id:fid})});
const d=await r.json();
const f=allForms.find(x=>x.id===fid);
el.innerHTML='';
(f.fields||[]).forEach(fd=>{
const row=document.createElement('div'); row.style.cssText='margin-bottom:8px';
const lb=document.createElement('label'); lb.style.cssText='font-size:12px;color:var(--muted)'; lb.textContent=fd.label; row.appendChild(lb);
const inp=document.createElement('input'); inp.dataset.key=fd.key; inp.value=(d.values||{})[fd.key]||''; row.appendChild(inp);
el.appendChild(row);
});
}
async function generateForm(){
const fid=document.getElementById('fillFormSelect').value;
const values={}; document.querySelectorAll('#fillFields input').forEach(i=>values[i.dataset.key]=i.value);
const r=await fetch('/api/forms/fill',{method:'POST',headers:{'Content-Type':'application/json',...authHeaders()},body:JSON.stringify({form_id:fid,values})});
if(!r.ok){ alert('Could not generate form'); return; }
const blob=await r.blob(); const url=URL.createObjectURL(blob);
const a=document.createElement('a'); a.href=url; a.download=fid+'_filled.pdf'; document.body.appendChild(a); a.click(); a.remove();
}
async function saveProfile(){
const body={display_name:document.getElementById('pfName').value, about_me:document.getElementById('pfAbout').value, avatar:myProfile.avatar};
const r=await fetch('/api/profile',{method:'POST',headers:{'Content-Type':'application/json',...authHeaders()},body:JSON.stringify(body)});

117
userdocs.py Normal file
View File

@@ -0,0 +1,117 @@
# -*- coding: utf-8 -*-
"""Per-user document index — ingest uploaded files into a vector array so the
attorney can reference the user's actual documents alongside the case law.
Reuses rag.py's embedder + chunker + cosine. stdlib + optional pymupdf (PDF).
"""
import json
import os
import re
from rag import embed, _split_long, _cosine, MAX_CHUNK
INDEX_DIR = "/opt/astraea/indexes"
def _path(uid):
return os.path.join(INDEX_DIR, f"user_{uid}.json")
def _docx_text(path):
import zipfile
try:
with zipfile.ZipFile(path) as z:
xml = z.read("word/document.xml").decode("utf-8", "ignore")
out = []
for p in re.split(r"</w:p>", xml):
ts = re.findall(r"<w:t[^>]*>(.*?)</w:t>", p)
if ts:
out.append("".join(ts))
return "\n".join(out)
except Exception:
return ""
def _pdf_text(path):
try:
import fitz # pymupdf
doc = fitz.open(path)
parts = []
for page in doc:
parts.append(page.get_text())
doc.close()
return "\n".join(parts)
except Exception:
return ""
def extract_text(orig_name, path):
ext = os.path.splitext(orig_name)[1].lower()
if ext == ".docx":
return _docx_text(path)
if ext == ".pdf":
return _pdf_text(path)
if ext in (".txt", ".md", ".csv", ".log", ".json", ".html", ".xml", ".rtf"):
try:
with open(path, "r", encoding="utf-8", errors="replace") as f:
return f.read()
except Exception:
return ""
return ""
def build(uid, files, upload_dir):
"""Embed every file's text into a per-user vector index. Returns chunk count."""
chunks = []
for rec in files:
path = os.path.join(upload_dir, rec["filename"])
if not os.path.exists(path):
continue
text = extract_text(rec.get("orig_name", rec["filename"]), path)
if not text or not text.strip():
continue
for piece in _split_long(text, MAX_CHUNK):
chunks.append({"source": rec.get("orig_name", rec["filename"]), "text": piece, "vector": None})
for c in chunks:
try:
c["vector"] = embed(c["text"])
except Exception:
c["vector"] = None
chunks = [c for c in chunks if c["vector"]]
os.makedirs(INDEX_DIR, exist_ok=True)
try:
with open(_path(uid), "w") as f:
json.dump({"chunks": chunks}, f)
except Exception:
pass
return len(chunks)
def retrieve(uid, query, top_k=4):
if not os.path.exists(_path(uid)):
return []
try:
with open(_path(uid)) as f:
data = json.load(f)
except Exception:
return []
qvec = embed(query)
scored = []
for c in data.get("chunks", []):
v = c.get("vector")
if not v:
continue
scored.append((_cosine(qvec, v), c))
scored.sort(key=lambda x: x[0], reverse=True)
return [{"source": c["source"], "text": c["text"], "score": round(s, 3)}
for s, c in scored[:top_k] if s > 0.15]
def status(uid):
if not os.path.exists(_path(uid)):
return {"indexed": False, "chunks": 0}
try:
with open(_path(uid)) as f:
data = json.load(f)
return {"indexed": True, "chunks": len(data.get("chunks", []))}
except Exception:
return {"indexed": False, "chunks": 0}