Advanced mode: chat-driven form filling (fill_form tool), 10 official WA forms auto-grabbed + auto-mapped, doc vault, web search, per-user doc RAG

This commit is contained in:
drjones
2026-09-05 12:35:04 -07:00
parent b075699cb6
commit eadce36742
7 changed files with 863 additions and 14 deletions

309
app.py
View File

@@ -3,9 +3,13 @@
Multi-agent RAG + no-KYC profiles (About Me) + document prep + comms missions + TTS.
"""
import json
import os
import re
import threading
import time
import urllib.parse
import urllib.request
import uuid
from flask import Flask, jsonify, render_template, request, send_file, Response
@@ -13,7 +17,9 @@ import rag
import store
import comms
import documents
import forms
import tts
import userdocs
from agents import AGENTS, AGENT_BY_ID, OLLAMA_URL, RAG_MODEL, GENERAL_MODEL
app = Flask(__name__)
@@ -34,9 +40,76 @@ AVATARS = [
{"id": 11, "glyph": "\U0001F9ED", "c1": "#5eead4", "c2": "#0d9488", "label": "Compass"},
]
UPLOAD_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "uploads")
FILE_CATEGORIES = ["Court filings", "Correspondence", "Financial", "Agreements",
"Evidence", "Medical & safety", "Generated forms", "Other"]
def _docx_text(path):
import zipfile as _z
try:
with _z.ZipFile(path) as z:
xml = z.read("word/document.xml").decode("utf-8", "ignore")
out = []
for p in re.split(r"</w:p>", xml):
ts = re.findall(r"<w:t[^>]*>(.*?)</w:t>", p)
if ts:
out.append("".join(ts))
return "\n".join(out)
except Exception:
return ""
def _reindex_user(uid):
"""Rebuild a user's document vector index in a background thread."""
try:
userdocs.build(uid, store.list_files(uid), UPLOAD_DIR)
except Exception:
pass
def _autofill_values(uid, form_id):
"""Map the user's About Me + display name onto a form's fields via the LLM."""
spec = forms.FORM_NAMES.get(form_id)
fm = forms.FIELDMAP.get(form_id, [])
if not spec or not fm:
return {}
profile = store.get_profile(uid) if uid else {}
about = profile.get("about_me", "")
display = profile.get("display_name", "")
field_desc = ", ".join(f"{f['key']} ({f.get('label', '')})" for f in fm)
prompt = (
"You are filling a Washington State court form. Map the user's facts to the form fields.\n"
f"Form: {spec['name']}\n"
f"Fields: {field_desc}\n\n"
f"User's display name: {display or '(not set)'}\n"
f"User's information (About Me):\n{about or '(not provided)'}\n\n"
"Output ONLY valid JSON like {\"county\": \"...\", \"petitioner\": \"...\"}. "
"For any field you cannot determine, output an empty string \"\". "
"Petitioner is the user themself (use the display name); respondent is their spouse."
)
try:
resp = _ollama_chat(RAG_MODEL, [{"role": "user", "content": prompt}],
temperature=0.1, num_predict=800)
content = resp.get("message", {}).get("content", "")
m = re.search(r"\{.*\}", content, re.DOTALL)
vals = json.loads(m.group(0)) if m else {}
except Exception:
vals = {}
return {f["key"]: str(vals.get(f["key"], "") or "") for f in fm}
def _save_pdf_to_vault(uid, path, name):
data = open(path, "rb").read()
fn = f"gen_{uid}_{uuid.uuid4().hex[:8]}.pdf"
with open(os.path.join(UPLOAD_DIR, fn), "wb") as f:
f.write(data)
return store.add_file(uid, "Generated forms", fn, name + ".pdf",
"application/pdf", len(data))
# ── Ollama helpers ──
def _ollama_chat(model, messages, tools=None, num_ctx=16384, num_predict=1400,
def _ollama_chat(model, messages, tools=None, num_ctx=24000, num_predict=1400,
temperature=0.2, timeout=180):
payload = {"model": model, "messages": messages, "stream": False, "think": False,
"options": {"temperature": temperature, "num_predict": num_predict, "num_ctx": num_ctx}}
@@ -49,6 +122,23 @@ def _ollama_chat(model, messages, tools=None, num_ctx=16384, num_predict=1400,
return json.loads(r.read().decode("utf-8"))
def _web_search(query, limit=5):
"""Search the live web via local SearXNG (separate service, not the LLM)."""
try:
url = f"http://10.30.20.35:6969/search?q={urllib.parse.quote(query)}&format=json"
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Astraea)"})
op = urllib.request.build_opener(urllib.request.ProxyHandler({}))
with op.open(req, timeout=15) as r:
data = json.loads(r.read().decode("utf-8"))
out = []
for it in data.get("results", [])[:limit]:
out.append({"title": it.get("title", ""), "url": it.get("url", ""),
"snippet": (it.get("content") or "")[:400]})
return out
except Exception as e:
return [{"error": f"web search failed: {e}"}]
def _strip_sources(text):
text = re.split(r"\n\s*(?:Sources|References|Citations)\s*:\s*\n", text, flags=re.I)[0]
return text.strip()
@@ -88,6 +178,17 @@ TOOLS = [
"recipient": {"type": "string", "description": "recipient phone number in E.164"},
"message": {"type": "string", "description": "the message to speak"}},
"required": ["recipient", "message"]}}},
{"type": "function", "function": {
"name": "web_search", "description": "Search the live web for current facts, statutes, or answers not in your reference documents. Use when you don't know something or need up-to-date information.",
"parameters": {"type": "object", "properties": {
"query": {"type": "string", "description": "the search query"}},
"required": ["query"]}}},
{"type": "function", "function": {
"name": "fill_form", "description": "Fill an official Washington court form with the user's info and save the filled PDF to their case documents. Use when the user asks to fill out / prepare / complete paperwork. Form ids: fl200=Summons, fl201=Petition, fl001=Confidential Info, fl140=Parenting Plan, wscss=Child Support Worksheets, fl131=Financial Declaration, fl231=Findings & Conclusions, fl241=Final Divorce Order, fl211=Response, fl223=Motion for Temporary Order.",
"parameters": {"type": "object", "properties": {
"form_id": {"type": "string", "description": "the form id, e.g. fl201"},
"values": {"type": "object", "description": "optional explicit field values; omitted fields auto-fill from the user's profile"}},
"required": ["form_id"]}}},
]
@@ -108,6 +209,24 @@ def _execute_tool(name, args, settings, uid, display_name):
twiml_url = request.host_url.rstrip("/") + "/twilio/voice"
comms.make_call(settings, recipient, twiml_url)
return f"Call initiated to {recipient}."
if name == "web_search":
res = _web_search(args.get("query", ""), limit=5)
if not res:
return "No results found."
return json.dumps(res, ensure_ascii=False)
if name == "fill_form":
form_id = args.get("form_id", "")
if not forms.FORM_NAMES.get(form_id):
return "Unknown form. Use one of: " + ", ".join(forms.FORM_NAMES.keys())
values = args.get("values") or {}
auto = _autofill_values(uid, form_id)
merged = {k: str(values.get(k) or auto.get(k) or "") for k in auto}
path = forms.fill_form(form_id, merged)
if not path:
return "Could not generate the form."
_save_pdf_to_vault(uid, path, forms.FORM_NAMES[form_id]["name"])
return (f"Filled '{forms.FORM_NAMES[form_id]['name']}' and saved it to the user's "
f"Case Documents. Tell the user to open their profile → Case documents to download it.")
except Exception as e:
return f"ERROR sending {name}: {e}"
return "Unknown tool."
@@ -149,6 +268,16 @@ def _build_answer(agent, message, history, about_me, settings, uid, display_name
citations.append({"source": c["source"], "title": c["title"],
"score": round(score, 3), "snippet": c["text"][:260]})
# user's own uploaded documents (per-user vector index) — retrieve alongside case law
if uid:
try:
for d in userdocs.retrieve(uid, message, top_k=4):
context_blocks.append(f"[USER DOC — {d['source']}]\n{d['text']}")
citations.append({"source": d["source"], "title": "your document",
"score": d["score"], "snippet": d["text"][:260]})
except Exception:
pass
system = (
f"{agent['system']}\n\n"
"Rules:\n"
@@ -158,8 +287,12 @@ def _build_answer(agent, message, history, about_me, settings, uid, display_name
"- Ground every legal claim in the reference documents. Cite the RCW section AND the "
"controlling case law by name.\n"
"- Be plain-English, specific to Washington State.\n"
"- Items marked '[USER DOC]' are the user's own uploaded files — reference them by name "
"to ground your answer in their actual case.\n"
"- You may use the available tools (send_sms / send_email / make_call) ONLY when the "
"user explicitly asks you to contact someone on their behalf.\n"
"- If you don't know the answer, or need current/up-to-date facts not in the reference "
"documents, use the web_search tool and answer from those results.\n"
+ about_me_block(about_me)
+ safety_block(settings)
)
@@ -283,6 +416,180 @@ def api_profile():
return jsonify(prof)
@app.route("/api/profile/pic", methods=["POST"])
def api_profile_pic():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
f = request.files.get("file")
if not f or not f.filename:
return jsonify({"error": "no file"}), 400
data = f.read()
if len(data) > 5 * 1024 * 1024:
return jsonify({"error": "image too large (max 5 MB)"}), 400
ext = os.path.splitext(f.filename)[1].lower()
if ext not in (".png", ".jpg", ".jpeg", ".gif", ".webp"):
return jsonify({"error": "unsupported type — use PNG/JPG/GIF/WebP"}), 400
up = os.path.join(os.path.dirname(os.path.abspath(__file__)), "static", "uploads")
os.makedirs(up, exist_ok=True)
fn = f"u{uid}{ext}"
with open(os.path.join(up, fn), "wb") as out:
out.write(data)
prof = store.set_profile(uid, profile_pic=f"/static/uploads/{fn}")
return jsonify(prof)
# ── Document vault ──
@app.route("/api/files", methods=["GET", "POST"])
def api_files():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
if request.method == "GET":
return jsonify({"categories": FILE_CATEGORIES, "files": store.list_files(uid)})
cat = request.form.get("category", "Other")
if cat not in FILE_CATEGORIES:
cat = "Other"
f = request.files.get("file")
if not f or not f.filename:
return jsonify({"error": "no file"}), 400
data = f.read()
if len(data) > 25 * 1024 * 1024:
return jsonify({"error": "file too large (max 25 MB)"}), 400
os.makedirs(UPLOAD_DIR, exist_ok=True)
ext = os.path.splitext(f.filename)[1].lower()[:12]
fn = f"{uid}_{uuid.uuid4().hex[:12]}{ext}"
with open(os.path.join(UPLOAD_DIR, fn), "wb") as out:
out.write(data)
fid = store.add_file(uid, cat, fn, f.filename, f.mimetype or "", len(data))
threading.Thread(target=_reindex_user, args=(uid,), daemon=True).start()
return jsonify({"ok": True, "id": fid, "category": cat})
@app.route("/api/files/reindex", methods=["POST"])
def api_files_reindex():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
n = userdocs.build(uid, store.list_files(uid), UPLOAD_DIR)
return jsonify({"ok": True, "chunks": n})
@app.route("/api/files/status")
def api_files_status():
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
return jsonify(userdocs.status(uid))
@app.route("/api/files/<int:fid>/raw")
def api_file_raw(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if not rec:
return jsonify({"error": "not found"}), 404
path = os.path.join(UPLOAD_DIR, rec["filename"])
if not os.path.exists(path):
return jsonify({"error": "missing"}), 404
return send_file(path, mimetype=rec["mime"] or "application/octet-stream",
as_attachment=False, download_name=rec["orig_name"])
@app.route("/api/files/<int:fid>/text")
def api_file_text(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if not rec:
return jsonify({"error": "not found"}), 404
path = os.path.join(UPLOAD_DIR, rec["filename"])
ext = os.path.splitext(rec["orig_name"])[1].lower()
if ext == ".docx":
return jsonify({"text": _docx_text(path), "mime": "text/plain"})
if ext in (".txt", ".md", ".csv", ".log", ".json", ".html", ".xml"):
try:
with open(path, "r", encoding="utf-8", errors="replace") as fh:
return jsonify({"text": fh.read()[:200000], "mime": "text/plain"})
except Exception:
return jsonify({"text": "", "mime": "text/plain"})
return jsonify({"text": "", "mime": rec["mime"] or ""})
@app.route("/api/files/<int:fid>", methods=["DELETE"])
def api_file_delete(fid):
uid = _uid()
if not uid:
return jsonify({"error": "auth required"}), 401
rec = store.get_file(uid, fid)
if rec:
p = os.path.join(UPLOAD_DIR, rec["filename"])
if os.path.exists(p):
try:
os.remove(p)
except Exception:
pass
store.delete_file(uid, fid)
return jsonify({"ok": True})
# ── Form filling (official WA court forms via pymupdf overlay) ──
@app.route("/api/forms")
def api_forms():
return jsonify(forms.list_forms())
@app.route("/api/forms/autofill", methods=["POST"])
def api_forms_autofill():
uid = _uid()
d = request.get_json(force=True, silent=True) or {}
form_id = d.get("form_id", "")
spec = forms.FORM_NAMES.get(form_id)
fm = forms.FIELDMAP.get(form_id, [])
if not spec:
return jsonify({"error": "unknown form"}), 400
profile = store.get_profile(uid) if uid else {}
about = profile.get("about_me", "")
display = profile.get("display_name", "")
field_desc = ", ".join(f"{f['key']} ({f.get('label','')})" for f in fm)
prompt = (
"You are filling a Washington State court form. Map the user's facts to the form fields.\n"
f"Form: {spec['name']}\n"
f"Fields: {field_desc}\n\n"
f"User's display name: {display or '(not set)'}\n"
f"User's information (About Me):\n{about or '(not provided)'}\n\n"
"Output ONLY valid JSON like {\"county\": \"...\", \"petitioner\": \"...\"}. "
"For any field you cannot determine, output an empty string. "
"Petitioner is the user themself (use the display name); respondent is their spouse."
)
try:
resp = _ollama_chat(RAG_MODEL, [{"role": "user", "content": prompt}],
temperature=0.1, num_predict=800)
content = resp.get("message", {}).get("content", "")
m = re.search(r"\{.*\}", content, re.DOTALL)
vals = json.loads(m.group(0)) if m else {}
except Exception:
vals = {}
known = {f["key"]: str(vals.get(f["key"], "") or "") for f in fm}
return jsonify({"form_id": form_id, "values": known})
@app.route("/api/forms/fill", methods=["POST"])
def api_forms_fill():
uid = _uid()
d = request.get_json(force=True, silent=True) or {}
form_id = d.get("form_id", "")
values = d.get("values", {})
path = forms.fill_form(form_id, values)
if not path:
return jsonify({"error": "could not fill form"}), 400
return send_file(path, mimetype="application/pdf", as_attachment=True,
download_name=f"{form_id}_filled.pdf")
@app.route("/api/settings", methods=["GET", "POST"])
def api_settings():
uid = _uid()