v3: relationship engine — rare-token clusters keep related files together (Projects/<Topic>/<role>), live nav status chip, LAN binding, progress feedback

This commit is contained in:
drjones
2026-10-05 04:14:15 -07:00
parent 220cd6cdba
commit 241b1dfef8

94
app.py
View File

@@ -68,7 +68,7 @@ def init_db():
CREATE TABLE IF NOT EXISTS files( CREATE TABLE IF NOT EXISTS files(
id INTEGER PRIMARY KEY, path TEXT UNIQUE, name TEXT, ext TEXT, size INTEGER, mtime REAL, id INTEGER PRIMARY KEY, path TEXT UNIQUE, name TEXT, ext TEXT, size INTEGER, mtime REAL,
sha256 TEXT, preview TEXT, file_class TEXT, subject TEXT, importance TEXT, action TEXT, sha256 TEXT, preview TEXT, file_class TEXT, subject TEXT, importance TEXT, action TEXT,
confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT); confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT, topic TEXT, role TEXT);
CREATE TABLE IF NOT EXISTS operations( CREATE TABLE IF NOT EXISTS operations(
id INTEGER PRIMARY KEY, ts REAL, session TEXT, file_id INTEGER, op TEXT, id INTEGER PRIMARY KEY, ts REAL, session TEXT, file_id INTEGER, op TEXT,
src TEXT, dst TEXT, sha256 TEXT, undone INTEGER DEFAULT 0); src TEXT, dst TEXT, sha256 TEXT, undone INTEGER DEFAULT 0);
@@ -130,6 +130,58 @@ SKIP_DIRS = {".Trashes", ".Spotlight-V100", ".TemporaryItems", ".fseventsd", "Sy
"$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"} "$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"}
JUNK = (".DS_Store",) JUNK = (".DS_Store",)
# ---------- CLUSTERING (relationship engine) ----------
STOP = set("""copy final new the and for with from test old bak backup untitled document file img
screenshot picture photo log notes data v1 v2 v3 final1 2 1 3 zip rar""".split())
def tokens(name):
stem = os.path.splitext(os.path.basename(name))[0].lower()
words = re.split(r"[^a-z0-9]+", stem)
return [w for w in words if len(w) > 2 and w not in STOP and not w.isdigit()]
def cluster_files(min_members=2):
"""Group files sharing rare name/preview tokens -> topic clusters (files that belong together)."""
c = db()
rows = c.execute("SELECT id,name,path,preview,file_class FROM files WHERE status NOT IN ('moved','held')").fetchall()
tok_map = {}
for r in rows:
seen = set(tokens(r["name"]))
if r["preview"]:
for t in tokens(r["preview"][:300]):
seen.add(t)
for t in seen:
tok_map.setdefault(t, set()).add(r["id"])
shared = {t: ids for t, ids in tok_map.items() if len(ids) >= min_members}
parent = {r["id"]: r["id"] for r in rows}
def find(x):
while parent[x] != x:
parent[x] = parent[parent[x]]; x = parent[x]
return x
for t, ids in shared.items():
ids = list(ids)
for x in ids[1:]:
ra, rb = find(ids[0]), find(x)
if ra != rb: parent[ra] = rb
groups = {}
for r in rows:
groups.setdefault(find(r["id"]), []).append(r)
clusters = []
for members in groups.values():
if len(members) < min_members:
continue
mids = {m["id"] for m in members}
toks = [(t, len(ids & mids)) for t, ids in shared.items() if ids & mids]
best = max(toks, key=lambda tv: (tv[1], len(tv[0])))[0] if toks else "cluster"
topic = best.replace("_", " ").strip().title()
for m in members:
role = {"source_code": "Code", "dataset": "Data", "archive": "Archives", "financial": "Docs",
"document": "Docs", "image": "Media", "video": "Media", "audio": "Media",
"ebook": "Docs", "software": "Installers", "personal": "Docs"}.get(m["file_class"], "Misc")
c.execute("UPDATE files SET topic=?, role=? WHERE id=?", (topic, role, m["id"]))
clusters.append({"topic": topic, "n": len(members), "members": [m["path"] for m in members]})
c.commit(); c.close()
return sorted(clusters, key=lambda x: -x["n"])
def scan(root, max_files=5000, min_size=0, max_size=0): def scan(root, max_files=5000, min_size=0, max_size=0):
init_db() init_db()
root = os.path.abspath(os.path.expanduser(root)) root = os.path.abspath(os.path.expanduser(root))
@@ -405,7 +457,7 @@ def build_plan(tx, min_conf):
plans = [] plans = []
for row in rows: for row in rows:
if row["action"] not in ("organize", "archive"): continue if row["action"] not in ("organize", "archive"): continue
dst_dir = propose_destination(row, tx) dst_dir = dest_for(row, tx)
name = row["rename"] if row["rename"] else row["name"] name = row["rename"] if row["rename"] else row["name"]
base, ext = os.path.splitext(name) base, ext = os.path.splitext(name)
if ext.lower() != row["ext"] and row["ext"]: if ext.lower() != row["ext"] and row["ext"]:
@@ -416,10 +468,16 @@ def build_plan(tx, min_conf):
plans.append({"id": row["id"], "path": row["path"], "name": row["name"], "dst": dst, plans.append({"id": row["id"], "path": row["path"], "name": row["name"], "dst": dst,
"cls": row["file_class"], "subj": row["subject"], "imp": row["importance"], "cls": row["file_class"], "subj": row["subject"], "imp": row["importance"],
"conf": row["confidence"], "verdict": verdict, "why": why, "dup": is_dup, "conf": row["confidence"], "verdict": verdict, "why": why, "dup": is_dup,
"size": row["size"]}) "size": row["size"], "topic": row["topic"], "role": row["role"]})
return plans return plans
# ---------- ORGANIZER (apply / undo) ---------- # ---------- ORGANIZER (apply / undo) ----------
def dest_for(row, tx):
"""Topic clusters stay together: Projects/<Topic>/<role>/. Everything else: taxonomy map."""
if row["topic"]:
return os.path.join(tx["organized_root"], "Projects", row["topic"], row["role"] or "Misc")
return propose_destination(row, tx)
def apply_moves(ids, dry_run=True): def apply_moves(ids, dry_run=True):
tx = load_taxonomy() tx = load_taxonomy()
c = db() c = db()
@@ -428,7 +486,7 @@ def apply_moves(ids, dry_run=True):
for fid in ids: for fid in ids:
row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone() row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone()
if not row or row["status"] not in ("classified", "approved"): skipped += 1; continue if not row or row["status"] not in ("classified", "approved"): skipped += 1; continue
dst_dir = propose_destination(row, tx) dst_dir = dest_for(row, tx)
name = row["rename"] or row["name"] name = row["rename"] or row["name"]
base, ext = os.path.splitext(name) base, ext = os.path.splitext(name)
if ext.lower() != row["ext"] and row["ext"]: if ext.lower() != row["ext"] and row["ext"]:
@@ -566,7 +624,7 @@ def shell(body, active="home"):
nav += f"<a class='{'on' if k==active else ''}' href='/{k}'>{label}</a>" nav += f"<a class='{'on' if k==active else ''}' href='/{k}'>{label}</a>"
return ("<!doctype html><html><head><meta charset='utf-8'><meta name='viewport' content='width=device-width,initial-scale=1'>" return ("<!doctype html><html><head><meta charset='utf-8'><meta name='viewport' content='width=device-width,initial-scale=1'>"
"<title>LIBRA — File Librarian</title><style>" + BASE_CSS + "</style></head><body>" "<title>LIBRA — File Librarian</title><style>" + BASE_CSS + "</style></head><body>"
"<nav><div class='in'><span class='brand'>♎ LIBRA</span>" + nav + "</div></nav>" "<nav><div class='in'><span class='brand'>♎ LIBRA</span><span id='nav-status' class='chip' style='display:none;background:rgba(53,208,186,.18);color:#6fe8d6'></span>" + nav + "</div></nav>"
"<div class='wrap'>" + body + "</div>" "<div class='wrap'>" + body + "</div>"
"<div class='toast' id='toast'></div>" "<div class='toast' id='toast'></div>"
"<script>window.LIBRA=" + json.dumps(STATE) + ";\n" "<script>window.LIBRA=" + json.dumps(STATE) + ";\n"
@@ -574,6 +632,13 @@ def shell(body, active="home"):
"var el=function(id){return document.getElementById(id)};" "var el=function(id){return document.getElementById(id)};"
"if(el('st-files')){el('st-files').textContent=s.files;el('st-classified').textContent=s.classified;" "if(el('st-files')){el('st-files').textContent=s.files;el('st-classified').textContent=s.classified;"
"el('st-dupes').textContent=s.dupes;el('st-pending').textContent=s.pending;el('st-moved').textContent=s.moved}" "el('st-dupes').textContent=s.dupes;el('st-pending').textContent=s.pending;el('st-moved').textContent=s.moved}"
"var busy=s.scan_running||s.classify_running;"
"var msg='';if(s.scan_running){msg='scanning…'}else if(s.classify_running){msg='Nimble classifying '+s.classify_done+'/'+s.classify_total+'…';"
"if(el('pbar')){el('pbar').style.width=(100*s.classify_done/Math.max(1,s.classify_total))+'%'}}"
"else if(el('pbar')){el('pbar').style.width=(s.classify_total?100*s.classify_done/Math.max(1,s.classify_total):0)+'%';"
"if(s.classify_done>0){msg='classified '+s.classify_done+'/'+s.classify_total}}"
"var ch=el('nav-status');if(ch){ch.textContent=msg;ch.style.display=msg?'inline-block':'none'}"
"if(busy&&el('pmsg')){el('pmsg').textContent=msg}"
"window.LIBRA=s;setTimeout(pulse,2000)}).catch(function(){setTimeout(pulse,4000)})})();</script></body></html>") "window.LIBRA=s;setTimeout(pulse,2000)}).catch(function(){setTimeout(pulse,4000)})})();</script></body></html>")
@app.route("/") @app.route("/")
@@ -581,7 +646,7 @@ def home():
refresh_counts() refresh_counts()
err = f"<div class='panel' style='border-color:rgba(255,107,107,.4)'><h2 style='color:var(--bad)'>Last error</h2><pre style='white-space:pre-wrap;font-size:11px'>{STATE.get('last_error','')}</pre></div>" if STATE.get("last_error") else "" err = f"<div class='panel' style='border-color:rgba(255,107,107,.4)'><h2 style='color:var(--bad)'>Last error</h2><pre style='white-space:pre-wrap;font-size:11px'>{STATE.get('last_error','')}</pre></div>" if STATE.get("last_error") else ""
scan_status = "<div class='bar'><i style='width:100%'></i></div><span class='muted'>scan running… refresh page</span>" if STATE["scan_running"] else "" scan_status = "<div class='bar'><i style='width:100%'></i></div><span class='muted'>scan running… refresh page</span>" if STATE["scan_running"] else ""
cls_status = f"<div class='bar'><i style='width:{int(100*STATE['classify_done']/max(1,STATE['classify_total']))}%'></i></div><span class='muted'>classifying {STATE['classify_done']}/{STATE['classify_total']}…</span>" if STATE["classify_running"] else "" cls_status = f"<div class='bar'><i id='pbar' style='width:{int(100*STATE['classify_done']/max(1,STATE['classify_total']))}%'></i></div><span class='muted'>classifying {STATE['classify_done']}/{STATE['classify_total']}…</span>" if STATE["classify_running"] else "<div class='bar'><i id='pbar' style='width:0%'></i></div><span class='muted' id='pmsg'>idle</span>"
body = f""" body = f"""
<h1 class='grad'>File Librarian</h1><p class='muted'>Scanner → Extractor → Nimble → Safety → You approve → Reversible move. Nothing moves without your click.</p> <h1 class='grad'>File Librarian</h1><p class='muted'>Scanner → Extractor → Nimble → Safety → You approve → Reversible move. Nothing moves without your click.</p>
<div class='grid'> <div class='grid'>
@@ -599,7 +664,10 @@ def home():
<div><label>Max size (bytes, 0=no cap)</label><input type='number' name='max_size' value='0'></div> <div><label>Max size (bytes, 0=no cap)</label><input type='number' name='max_size' value='0'></div>
<div><button type='submit'>SCAN →</button></div> <div><button type='submit'>SCAN →</button></div>
</div></form></div> </div></form></div>
<div class='panel'><h2>2 · Hash + Classify (Nimble on nightmare)</h2>{cls_status} <div class='panel'><h2>2 · Group what belongs together</h2>
<form method='post' action='/cluster'><button class='ghost'>CLUSTER RELATED FILES →</button></form>
<p class='muted'>Files sharing names/preview content get grouped into Projects/&lt;Topic&gt;/Docs·Code·Data·Media — moved together as one unit.</p></div>
<div class='panel'><h2>3 · Hash + Classify (Nimble on nightmare)</h2>{cls_status}
<form method='post' action='/hash'><div class='row'> <form method='post' action='/hash'><div class='row'>
<div><label>Hash batch limit</label><input type='number' name='limit' value='500'></div> <div><label>Hash batch limit</label><input type='number' name='limit' value='500'></div>
<div><label>Max MB to hash per file</label><input type='number' name='max_mb' value='64'></div> <div><label>Max MB to hash per file</label><input type='number' name='max_mb' value='64'></div>
@@ -676,16 +744,17 @@ def plan_page():
tx = load_taxonomy() tx = load_taxonomy()
min_conf = float(request.args.get("min_conf", 0.6)) min_conf = float(request.args.get("min_conf", 0.6))
plans = build_plan(tx, min_conf) plans = build_plan(tx, min_conf)
plans.sort(key=lambda p: (-p["conf"], p["verdict"] != "APPROVE")) plans.sort(key=lambda p: (p.get("topic") or "zzz", -p["conf"], p["verdict"] != "APPROVE"))
cards = "" cards = ""
approved_default = 0 approved_default = 0
for p in plans[:600]: for p in plans[:600]:
checked = "checked" if p["verdict"] == "APPROVE" else "" checked = "checked" if p["verdict"] == "APPROVE" else ""
if p["verdict"] == "APPROVE": approved_default += 1 if p["verdict"] == "APPROVE": approved_default += 1
dupnote = "<span class='chip' style='background:rgba(255,180,84,.16);color:#ffcf8f'>DUPLICATE GROUP</span>" if p["dup"] else "" dupnote = "<span class='chip' style='background:rgba(255,180,84,.16);color:#ffcf8f'>DUPLICATE GROUP</span>" if p["dup"] else ""
topicnote = f"<span class='chip' style='background:rgba(124,108,255,.2);color:#c9bfff'>◆ {p['topic']} · {p['role']}</span>" if p.get("topic") else ""
cards += (f"<div class='plan'>" cards += (f"<div class='plan'>"
f"<div class='src'>{p['path']}</div><div class='arrow'>↓</div><div class='dst'>{p['dst']}</div>" f"<div class='src'>{p['path']}</div><div class='arrow'>↓</div><div class='dst'>{p['dst']}</div>"
f"<div class='meta'><span class='chip c-{p['cls']}'>{p['cls']}</span><span class='muted'>{p['subj']} · {p['imp']}</span>" f"<div class='meta'>{topicnote}<span class='chip c-{p['cls']}'>{p['cls']}</span><span class='muted'>{p['subj']} · {p['imp']}</span>"
f"<span class='muted'>conf {int(p['conf']*100)}%</span>" f"<span class='muted'>conf {int(p['conf']*100)}%</span>"
f"<span class='v-{p['verdict']}' style='font-weight:700'>{p['verdict']}</span>" f"<span class='v-{p['verdict']}' style='font-weight:700'>{p['verdict']}</span>"
+ (f"<span class='muted'>({p['why']})</span>" if p["why"] else "") + dupnote + + (f"<span class='muted'>({p['why']})</span>" if p["why"] else "") + dupnote +
@@ -726,6 +795,11 @@ def api_apply():
done, skipped, session = apply_moves(ids, dry_run=bool(d.get("dry_run", True))) done, skipped, session = apply_moves(ids, dry_run=bool(d.get("dry_run", True)))
return {"done": done, "skipped": skipped, "session": session} return {"done": done, "skipped": skipped, "session": session}
@app.route("/cluster", methods=["POST"])
def cluster_post():
cluster_files()
return redirect("/plan")
@app.route("/dupes") @app.route("/dupes")
def dupes_page(): def dupes_page():
c = db() c = db()
@@ -829,4 +903,4 @@ def settings_save():
if __name__ == "__main__": if __name__ == "__main__":
init_db() init_db()
refresh_counts() refresh_counts()
app.run(host="127.0.0.1", port=5731, debug=False) app.run(host="0.0.0.0", port=5731, debug=False)