Files
astraea/detect_fields.py

123 lines
4.3 KiB
Python

#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Detect fillable blanks across WA court forms: underscore runs + the standard
caption (county/petitioner/respondent/case_no). Emits a field-map JSON."""
import json
import os
import re
import sys
import pymupdf
FORMS_DIR = "/opt/astraea/forms"
def lines_with_pos(page):
words = page.get_text("words")
groups = {}
for w in words:
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
key = int(y0 / 5) * 5
groups.setdefault(key, []).append((x0, x1, word))
out = []
for y in sorted(groups):
ws = sorted(groups[y], key=lambda t: t[0])
text = " ".join(w for _, _, w in ws)
out.append({"y": y, "x0": ws[0][0], "x1": ws[-1][1], "text": text, "words": ws})
return out
def detect_caption(lines):
"""Find the standard caption fields. Returns list of {key, label, page, x, y}."""
fields = []
for ln in lines:
t = ln["text"]
# "County of" -> blank right after 'of'
if re.search(r"\bCounty\b.*\bof\b\s*$", t) and "Superior" in t:
ofx = None
for x0, x1, w in ln["words"]:
if w == "of":
ofx = x1
if ofx:
fields.append({"key": "county", "label": "County", "x": ofx + 6, "y": ln["y"] + 9})
# Petitioner ... case): -> name after the colon
if "Petitioner" in t and ("case" in t or "case):" in t) and ":" in t:
colonx = None
for x0, x1, w in ln["words"]:
if w.endswith(":"):
colonx = x1
if colonx and colonx < 400:
fields.append({"key": "petitioner", "label": "Petitioner (your full name)", "x": colonx + 6, "y": ln["y"] + 9})
# Respondent ... partner): -> name after colon
if "Respondent" in t and ":" in t:
colonx = None
for x0, x1, w in ln["words"]:
if w.endswith(":"):
colonx = x1
if colonx and colonx < 400:
fields.append({"key": "respondent", "label": "Respondent (spouse's full name)", "x": colonx + 6, "y": ln["y"] + 9})
# Case No.
if "Case No." in t:
nox = None
for x0, x1, w in ln["words"]:
if w in ("No.", "No"):
nox = x1
if nox:
fields.append({"key": "case_no", "label": "Case number", "x": nox + 6, "y": ln["y"] + 9})
return fields
def detect_underscores(pno, page):
"""Find underscore runs (blank fields) and their preceding label."""
fields = []
words = page.get_text("words")
groups = {}
for w in words:
x0, y0, x1, y1, word = w[0], w[1], w[2], w[3], w[4]
key = int(y0 / 5) * 5
groups.setdefault(key, []).append((x0, x1, y0, word))
for y in sorted(groups):
ws = sorted(groups[y], key=lambda t: t[0])
for i, (x0, x1, y0, word) in enumerate(ws):
if word.strip("_").strip() == "" and len(word.strip()) >= 4:
label = " ".join(w for _, _, _, w in ws[:i]).strip()
fields.append({"page": pno, "x": x0, "y": y + 9,
"label": label, "type": "text"})
return fields
def main():
all_fields = {}
for fn in sorted(os.listdir(FORMS_DIR)):
if not fn.endswith(".pdf"):
continue
fid = fn[:-4]
path = os.path.join(FORMS_DIR, fn)
try:
d = pymupdf.open(path)
except Exception as e:
print(f"{fid}: OPEN FAIL {e}", file=sys.stderr)
continue
fields = []
seen_keys = set()
for pno in range(min(len(d), 3)): # caption + first 2 pages
lines = lines_with_pos(d[pno])
if pno == 0:
for f in detect_caption(lines):
f["page"] = 0
if f["key"] not in seen_keys:
seen_keys.add(f["key"])
fields.append(f)
for f in detect_underscores(pno, d[pno]):
f["key"] = "f_" + str(len(fields))
fields.append(f)
d.close()
all_fields[fid] = fields
out = "/opt/astraea/forms/fieldmap.json"
with open(out, "w") as f:
json.dump(all_fields, f, indent=1)
print(f"saved {out}: { {k: len(v) for k, v in all_fields.items()} }")
if __name__ == "__main__":
main()