Astraea v1.0 — multi-agent WA family-law assistant
Eight specialist agents over a 16-book verified WA law corpus (RAG with citations), per-user document vault, WA court-form PDF auto-fill, comms missions with DV safety guard, no-KYC auth, TTS. Self-hosted: Flask + SQLite + Ollama, stdlib-only RAG. Includes README, LICENSE (MIT + not-legal-advice notice), DEPLOY runbook, .gitignore.
This commit is contained in:
149
detect_fields.py
Normal file
149
detect_fields.py
Normal file
@@ -0,0 +1,149 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Detect fillable blanks across WA court forms: underscore runs + the standard
|
||||
caption (county/petitioner/respondent/case_no). Emits a field-map JSON."""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import pymupdf
|
||||
|
||||
FORMS_DIR = "/opt/astraea/forms"
|
||||
|
||||
def lines_with_pos(page):
|
||||
words = page.get_text("words")
|
||||
groups = {}
|
||||
for w in words:
|
||||
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
||||
key = int(y0 / 5) * 5
|
||||
groups.setdefault(key, []).append((x0, x1, word))
|
||||
out = []
|
||||
for y in sorted(groups):
|
||||
ws = sorted(groups[y], key=lambda t: t[0])
|
||||
text = " ".join(w for _, _, w in ws)
|
||||
out.append({"y": y, "x0": ws[0][0], "x1": ws[-1][1], "text": text, "words": ws})
|
||||
return out
|
||||
|
||||
|
||||
def detect_caption(lines):
|
||||
"""Find the standard caption fields. Returns list of {key, label, page, x, y}."""
|
||||
fields = []
|
||||
for ln in lines:
|
||||
t = ln["text"]
|
||||
# "County of" -> blank right after 'of'
|
||||
if re.search(r"\bCounty\b.*\bof\b\s*$", t) and "Superior" in t:
|
||||
ofx = None
|
||||
for x0, x1, w in ln["words"]:
|
||||
if w == "of":
|
||||
ofx = x1
|
||||
if ofx:
|
||||
fields.append({"key": "county", "label": "County", "x": ofx + 6, "y": ln["y"] + 9})
|
||||
# Petitioner ... case): -> name after the colon
|
||||
if "Petitioner" in t and ("case" in t or "case):" in t) and ":" in t:
|
||||
colonx = None
|
||||
for x0, x1, w in ln["words"]:
|
||||
if w.endswith(":"):
|
||||
colonx = x1
|
||||
if colonx and colonx < 400:
|
||||
fields.append({"key": "petitioner", "label": "Petitioner (your full name)", "x": colonx + 6, "y": ln["y"] + 9})
|
||||
# Respondent ... partner): -> name after colon
|
||||
if "Respondent" in t and ":" in t:
|
||||
colonx = None
|
||||
for x0, x1, w in ln["words"]:
|
||||
if w.endswith(":"):
|
||||
colonx = x1
|
||||
if colonx and colonx < 400:
|
||||
fields.append({"key": "respondent", "label": "Respondent (spouse's full name)", "x": colonx + 6, "y": ln["y"] + 9})
|
||||
# Case No.
|
||||
if "Case No." in t:
|
||||
nox = None
|
||||
for x0, x1, w in ln["words"]:
|
||||
if w in ("No.", "No"):
|
||||
nox = x1
|
||||
if nox:
|
||||
fields.append({"key": "case_no", "label": "Case number", "x": nox + 6, "y": ln["y"] + 9})
|
||||
return fields
|
||||
|
||||
|
||||
def detect_underscores(pno, page):
|
||||
"""Find underscore runs (blank fields) and their preceding label."""
|
||||
fields = []
|
||||
words = page.get_text("words")
|
||||
groups = {}
|
||||
for w in words:
|
||||
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
||||
key = int(y0 / 5) * 5
|
||||
groups.setdefault(key, []).append((x0, x1, y0, word))
|
||||
for y in sorted(groups):
|
||||
ws = sorted(groups[y], key=lambda t: t[0])
|
||||
for i, (x0, x1, y0, word) in enumerate(ws):
|
||||
if word.strip("_").strip() == "" and len(word.strip()) >= 4:
|
||||
label = " ".join(w for _, _, _, w in ws[:i]).strip()
|
||||
fields.append({"page": pno, "x": x0, "y": y + 9,
|
||||
"label": label, "type": "text"})
|
||||
return fields
|
||||
|
||||
|
||||
def detect_checkboxes(pno, page):
|
||||
"""Find checkbox glyphs (U+F0A8) and their labels. Type 'checkbox'."""
|
||||
fields = []
|
||||
words = page.get_text("words")
|
||||
groups = {}
|
||||
for w in words:
|
||||
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
||||
key = int(y0 / 5) * 5
|
||||
groups.setdefault(key, []).append((x0, x1, y0, word))
|
||||
for y in sorted(groups):
|
||||
ws = sorted(groups[y], key=lambda t: t[0])
|
||||
for i, (x0, x1, y0, word) in enumerate(ws):
|
||||
if "\uf0a8" in word:
|
||||
label_words = []
|
||||
for wx0, wx1, _, w in ws[i + 1:]:
|
||||
if "\uf0a8" in w:
|
||||
break
|
||||
label_words.append(w)
|
||||
label = " ".join(label_words).strip()
|
||||
fields.append({"page": pno, "x": x0, "y": y0, "type": "checkbox",
|
||||
"label": label})
|
||||
return fields
|
||||
|
||||
|
||||
def main():
|
||||
all_fields = {}
|
||||
for fn in sorted(os.listdir(FORMS_DIR)):
|
||||
if not fn.endswith(".pdf"):
|
||||
continue
|
||||
fid = fn[:-4]
|
||||
path = os.path.join(FORMS_DIR, fn)
|
||||
try:
|
||||
d = pymupdf.open(path)
|
||||
except Exception as e:
|
||||
print(f"{fid}: OPEN FAIL {e}", file=sys.stderr)
|
||||
continue
|
||||
fields = []
|
||||
seen_keys = set()
|
||||
for pno in range(min(len(d), 12)): # caption + all pages
|
||||
if pno == 0:
|
||||
lines = lines_with_pos(d[pno])
|
||||
for f in detect_caption(lines):
|
||||
f["page"] = 0
|
||||
if f["key"] not in seen_keys:
|
||||
seen_keys.add(f["key"])
|
||||
fields.append(f)
|
||||
for f in detect_underscores(pno, d[pno]):
|
||||
f["key"] = "f_" + str(len(fields))
|
||||
fields.append(f)
|
||||
for f in detect_checkboxes(pno, d[pno]):
|
||||
f["key"] = "cb_" + str(len(fields))
|
||||
fields.append(f)
|
||||
d.close()
|
||||
all_fields[fid] = fields
|
||||
out = "/opt/astraea/forms/fieldmap.json"
|
||||
with open(out, "w") as f:
|
||||
json.dump(all_fields, f, indent=1)
|
||||
print(f"saved {out}: { {k: len(v) for k, v in all_fields.items()} }")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user