From 241b1dfef824471e4a5e9e74884eedce2bef9828 Mon Sep 17 00:00:00 2001 From: drjones Date: Mon, 5 Oct 2026 04:14:15 -0700 Subject: [PATCH] =?UTF-8?q?v3:=20relationship=20engine=20=E2=80=94=20rare-?= =?UTF-8?q?token=20clusters=20keep=20related=20files=20together=20(Project?= =?UTF-8?q?s//),=20live=20nav=20status=20chip,=20LAN=20bindin?= =?UTF-8?q?g,=20progress=20feedback?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- app.py | 94 +++++++++++++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 84 insertions(+), 10 deletions(-) diff --git a/app.py b/app.py index d6d6545..2856a61 100644 --- a/app.py +++ b/app.py @@ -68,7 +68,7 @@ def init_db(): CREATE TABLE IF NOT EXISTS files( id INTEGER PRIMARY KEY, path TEXT UNIQUE, name TEXT, ext TEXT, size INTEGER, mtime REAL, sha256 TEXT, preview TEXT, file_class TEXT, subject TEXT, importance TEXT, action TEXT, - confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT); + confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT, topic TEXT, role TEXT); CREATE TABLE IF NOT EXISTS operations( id INTEGER PRIMARY KEY, ts REAL, session TEXT, file_id INTEGER, op TEXT, src TEXT, dst TEXT, sha256 TEXT, undone INTEGER DEFAULT 0); @@ -130,6 +130,58 @@ SKIP_DIRS = {".Trashes", ".Spotlight-V100", ".TemporaryItems", ".fseventsd", "Sy "$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"} JUNK = (".DS_Store",) +# ---------- CLUSTERING (relationship engine) ---------- +STOP = set("""copy final new the and for with from test old bak backup untitled document file img + screenshot picture photo log notes data v1 v2 v3 final1 2 1 3 zip rar""".split()) + +def tokens(name): + stem = os.path.splitext(os.path.basename(name))[0].lower() + words = re.split(r"[^a-z0-9]+", stem) + return [w for w in words if len(w) > 2 and w not in STOP and not w.isdigit()] + +def cluster_files(min_members=2): + """Group files sharing rare name/preview tokens -> topic clusters (files that belong together).""" + c = db() + rows = c.execute("SELECT id,name,path,preview,file_class FROM files WHERE status NOT IN ('moved','held')").fetchall() + tok_map = {} + for r in rows: + seen = set(tokens(r["name"])) + if r["preview"]: + for t in tokens(r["preview"][:300]): + seen.add(t) + for t in seen: + tok_map.setdefault(t, set()).add(r["id"]) + shared = {t: ids for t, ids in tok_map.items() if len(ids) >= min_members} + parent = {r["id"]: r["id"] for r in rows} + def find(x): + while parent[x] != x: + parent[x] = parent[parent[x]]; x = parent[x] + return x + for t, ids in shared.items(): + ids = list(ids) + for x in ids[1:]: + ra, rb = find(ids[0]), find(x) + if ra != rb: parent[ra] = rb + groups = {} + for r in rows: + groups.setdefault(find(r["id"]), []).append(r) + clusters = [] + for members in groups.values(): + if len(members) < min_members: + continue + mids = {m["id"] for m in members} + toks = [(t, len(ids & mids)) for t, ids in shared.items() if ids & mids] + best = max(toks, key=lambda tv: (tv[1], len(tv[0])))[0] if toks else "cluster" + topic = best.replace("_", " ").strip().title() + for m in members: + role = {"source_code": "Code", "dataset": "Data", "archive": "Archives", "financial": "Docs", + "document": "Docs", "image": "Media", "video": "Media", "audio": "Media", + "ebook": "Docs", "software": "Installers", "personal": "Docs"}.get(m["file_class"], "Misc") + c.execute("UPDATE files SET topic=?, role=? WHERE id=?", (topic, role, m["id"])) + clusters.append({"topic": topic, "n": len(members), "members": [m["path"] for m in members]}) + c.commit(); c.close() + return sorted(clusters, key=lambda x: -x["n"]) + def scan(root, max_files=5000, min_size=0, max_size=0): init_db() root = os.path.abspath(os.path.expanduser(root)) @@ -405,7 +457,7 @@ def build_plan(tx, min_conf): plans = [] for row in rows: if row["action"] not in ("organize", "archive"): continue - dst_dir = propose_destination(row, tx) + dst_dir = dest_for(row, tx) name = row["rename"] if row["rename"] else row["name"] base, ext = os.path.splitext(name) if ext.lower() != row["ext"] and row["ext"]: @@ -416,10 +468,16 @@ def build_plan(tx, min_conf): plans.append({"id": row["id"], "path": row["path"], "name": row["name"], "dst": dst, "cls": row["file_class"], "subj": row["subject"], "imp": row["importance"], "conf": row["confidence"], "verdict": verdict, "why": why, "dup": is_dup, - "size": row["size"]}) + "size": row["size"], "topic": row["topic"], "role": row["role"]}) return plans # ---------- ORGANIZER (apply / undo) ---------- +def dest_for(row, tx): + """Topic clusters stay together: Projects///. Everything else: taxonomy map.""" + if row["topic"]: + return os.path.join(tx["organized_root"], "Projects", row["topic"], row["role"] or "Misc") + return propose_destination(row, tx) + def apply_moves(ids, dry_run=True): tx = load_taxonomy() c = db() @@ -428,7 +486,7 @@ def apply_moves(ids, dry_run=True): for fid in ids: row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone() if not row or row["status"] not in ("classified", "approved"): skipped += 1; continue - dst_dir = propose_destination(row, tx) + dst_dir = dest_for(row, tx) name = row["rename"] or row["name"] base, ext = os.path.splitext(name) if ext.lower() != row["ext"] and row["ext"]: @@ -566,7 +624,7 @@ def shell(body, active="home"): nav += f"{label}" return ("" "LIBRA — File Librarian" - "" + "" "
" + body + "
" "
" "") @app.route("/") @@ -581,7 +646,7 @@ def home(): refresh_counts() err = f"

Last error

{STATE.get('last_error','')}
" if STATE.get("last_error") else "" scan_status = "
scan running… refresh page" if STATE["scan_running"] else "" - cls_status = f"
classifying {STATE['classify_done']}/{STATE['classify_total']}…" if STATE["classify_running"] else "" + cls_status = f"
classifying {STATE['classify_done']}/{STATE['classify_total']}…" if STATE["classify_running"] else "
idle" body = f"""

File Librarian

Scanner → Extractor → Nimble → Safety → You approve → Reversible move. Nothing moves without your click.

@@ -599,7 +664,10 @@ def home():
-

2 · Hash + Classify (Nimble on nightmare)

{cls_status} +

2 · Group what belongs together

+
+

Files sharing names/preview content get grouped into Projects/<Topic>/Docs·Code·Data·Media — moved together as one unit.

+

3 · Hash + Classify (Nimble on nightmare)

{cls_status}
@@ -676,16 +744,17 @@ def plan_page(): tx = load_taxonomy() min_conf = float(request.args.get("min_conf", 0.6)) plans = build_plan(tx, min_conf) - plans.sort(key=lambda p: (-p["conf"], p["verdict"] != "APPROVE")) + plans.sort(key=lambda p: (p.get("topic") or "zzz", -p["conf"], p["verdict"] != "APPROVE")) cards = "" approved_default = 0 for p in plans[:600]: checked = "checked" if p["verdict"] == "APPROVE" else "" if p["verdict"] == "APPROVE": approved_default += 1 dupnote = "DUPLICATE GROUP" if p["dup"] else "" + topicnote = f"◆ {p['topic']} · {p['role']}" if p.get("topic") else "" cards += (f"
" f"
{p['path']}
↓
{p['dst']}
" - f"
{p['cls']}{p['subj']} · {p['imp']}" + f"
{topicnote}{p['cls']}{p['subj']} · {p['imp']}" f"conf {int(p['conf']*100)}%" f"{p['verdict']}" + (f"({p['why']})" if p["why"] else "") + dupnote + @@ -726,6 +795,11 @@ def api_apply(): done, skipped, session = apply_moves(ids, dry_run=bool(d.get("dry_run", True))) return {"done": done, "skipped": skipped, "session": session} +@app.route("/cluster", methods=["POST"]) +def cluster_post(): + cluster_files() + return redirect("/plan") + @app.route("/dupes") def dupes_page(): c = db() @@ -829,4 +903,4 @@ def settings_save(): if __name__ == "__main__": init_db() refresh_counts() - app.run(host="127.0.0.1", port=5731, debug=False) + app.run(host="0.0.0.0", port=5731, debug=False)