Eight specialist agents over a 16-book verified WA law corpus (RAG with citations), per-user document vault, WA court-form PDF auto-fill, comms missions with DV safety guard, no-KYC auth, TTS. Self-hosted: Flask + SQLite + Ollama, stdlib-only RAG. Includes README, LICENSE (MIT + not-legal-advice notice), DEPLOY runbook, .gitignore.
150 lines
5.4 KiB
Python
150 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
"""Detect fillable blanks across WA court forms: underscore runs + the standard
|
|
caption (county/petitioner/respondent/case_no). Emits a field-map JSON."""
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
import pymupdf
|
|
|
|
FORMS_DIR = "/opt/astraea/forms"
|
|
|
|
def lines_with_pos(page):
|
|
words = page.get_text("words")
|
|
groups = {}
|
|
for w in words:
|
|
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
|
key = int(y0 / 5) * 5
|
|
groups.setdefault(key, []).append((x0, x1, word))
|
|
out = []
|
|
for y in sorted(groups):
|
|
ws = sorted(groups[y], key=lambda t: t[0])
|
|
text = " ".join(w for _, _, w in ws)
|
|
out.append({"y": y, "x0": ws[0][0], "x1": ws[-1][1], "text": text, "words": ws})
|
|
return out
|
|
|
|
|
|
def detect_caption(lines):
|
|
"""Find the standard caption fields. Returns list of {key, label, page, x, y}."""
|
|
fields = []
|
|
for ln in lines:
|
|
t = ln["text"]
|
|
# "County of" -> blank right after 'of'
|
|
if re.search(r"\bCounty\b.*\bof\b\s*$", t) and "Superior" in t:
|
|
ofx = None
|
|
for x0, x1, w in ln["words"]:
|
|
if w == "of":
|
|
ofx = x1
|
|
if ofx:
|
|
fields.append({"key": "county", "label": "County", "x": ofx + 6, "y": ln["y"] + 9})
|
|
# Petitioner ... case): -> name after the colon
|
|
if "Petitioner" in t and ("case" in t or "case):" in t) and ":" in t:
|
|
colonx = None
|
|
for x0, x1, w in ln["words"]:
|
|
if w.endswith(":"):
|
|
colonx = x1
|
|
if colonx and colonx < 400:
|
|
fields.append({"key": "petitioner", "label": "Petitioner (your full name)", "x": colonx + 6, "y": ln["y"] + 9})
|
|
# Respondent ... partner): -> name after colon
|
|
if "Respondent" in t and ":" in t:
|
|
colonx = None
|
|
for x0, x1, w in ln["words"]:
|
|
if w.endswith(":"):
|
|
colonx = x1
|
|
if colonx and colonx < 400:
|
|
fields.append({"key": "respondent", "label": "Respondent (spouse's full name)", "x": colonx + 6, "y": ln["y"] + 9})
|
|
# Case No.
|
|
if "Case No." in t:
|
|
nox = None
|
|
for x0, x1, w in ln["words"]:
|
|
if w in ("No.", "No"):
|
|
nox = x1
|
|
if nox:
|
|
fields.append({"key": "case_no", "label": "Case number", "x": nox + 6, "y": ln["y"] + 9})
|
|
return fields
|
|
|
|
|
|
def detect_underscores(pno, page):
|
|
"""Find underscore runs (blank fields) and their preceding label."""
|
|
fields = []
|
|
words = page.get_text("words")
|
|
groups = {}
|
|
for w in words:
|
|
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
|
key = int(y0 / 5) * 5
|
|
groups.setdefault(key, []).append((x0, x1, y0, word))
|
|
for y in sorted(groups):
|
|
ws = sorted(groups[y], key=lambda t: t[0])
|
|
for i, (x0, x1, y0, word) in enumerate(ws):
|
|
if word.strip("_").strip() == "" and len(word.strip()) >= 4:
|
|
label = " ".join(w for _, _, _, w in ws[:i]).strip()
|
|
fields.append({"page": pno, "x": x0, "y": y + 9,
|
|
"label": label, "type": "text"})
|
|
return fields
|
|
|
|
|
|
def detect_checkboxes(pno, page):
|
|
"""Find checkbox glyphs (U+F0A8) and their labels. Type 'checkbox'."""
|
|
fields = []
|
|
words = page.get_text("words")
|
|
groups = {}
|
|
for w in words:
|
|
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
|
|
key = int(y0 / 5) * 5
|
|
groups.setdefault(key, []).append((x0, x1, y0, word))
|
|
for y in sorted(groups):
|
|
ws = sorted(groups[y], key=lambda t: t[0])
|
|
for i, (x0, x1, y0, word) in enumerate(ws):
|
|
if "\uf0a8" in word:
|
|
label_words = []
|
|
for wx0, wx1, _, w in ws[i + 1:]:
|
|
if "\uf0a8" in w:
|
|
break
|
|
label_words.append(w)
|
|
label = " ".join(label_words).strip()
|
|
fields.append({"page": pno, "x": x0, "y": y0, "type": "checkbox",
|
|
"label": label})
|
|
return fields
|
|
|
|
|
|
def main():
|
|
all_fields = {}
|
|
for fn in sorted(os.listdir(FORMS_DIR)):
|
|
if not fn.endswith(".pdf"):
|
|
continue
|
|
fid = fn[:-4]
|
|
path = os.path.join(FORMS_DIR, fn)
|
|
try:
|
|
d = pymupdf.open(path)
|
|
except Exception as e:
|
|
print(f"{fid}: OPEN FAIL {e}", file=sys.stderr)
|
|
continue
|
|
fields = []
|
|
seen_keys = set()
|
|
for pno in range(min(len(d), 12)): # caption + all pages
|
|
if pno == 0:
|
|
lines = lines_with_pos(d[pno])
|
|
for f in detect_caption(lines):
|
|
f["page"] = 0
|
|
if f["key"] not in seen_keys:
|
|
seen_keys.add(f["key"])
|
|
fields.append(f)
|
|
for f in detect_underscores(pno, d[pno]):
|
|
f["key"] = "f_" + str(len(fields))
|
|
fields.append(f)
|
|
for f in detect_checkboxes(pno, d[pno]):
|
|
f["key"] = "cb_" + str(len(fields))
|
|
fields.append(f)
|
|
d.close()
|
|
all_fields[fid] = fields
|
|
out = "/opt/astraea/forms/fieldmap.json"
|
|
with open(out, "w") as f:
|
|
json.dump(all_fields, f, indent=1)
|
|
print(f"saved {out}: { {k: len(v) for k, v in all_fields.items()} }")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|