v3: relationship engine — rare-token clusters keep related files together (Projects/<Topic>/<role>), live nav status chip, LAN binding, progress feedback
This commit is contained in:
94
app.py
94
app.py
@@ -68,7 +68,7 @@ def init_db():
|
|||||||
CREATE TABLE IF NOT EXISTS files(
|
CREATE TABLE IF NOT EXISTS files(
|
||||||
id INTEGER PRIMARY KEY, path TEXT UNIQUE, name TEXT, ext TEXT, size INTEGER, mtime REAL,
|
id INTEGER PRIMARY KEY, path TEXT UNIQUE, name TEXT, ext TEXT, size INTEGER, mtime REAL,
|
||||||
sha256 TEXT, preview TEXT, file_class TEXT, subject TEXT, importance TEXT, action TEXT,
|
sha256 TEXT, preview TEXT, file_class TEXT, subject TEXT, importance TEXT, action TEXT,
|
||||||
confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT);
|
confidence REAL, rename TEXT, status TEXT DEFAULT 'scanned', reason TEXT, topic TEXT, role TEXT);
|
||||||
CREATE TABLE IF NOT EXISTS operations(
|
CREATE TABLE IF NOT EXISTS operations(
|
||||||
id INTEGER PRIMARY KEY, ts REAL, session TEXT, file_id INTEGER, op TEXT,
|
id INTEGER PRIMARY KEY, ts REAL, session TEXT, file_id INTEGER, op TEXT,
|
||||||
src TEXT, dst TEXT, sha256 TEXT, undone INTEGER DEFAULT 0);
|
src TEXT, dst TEXT, sha256 TEXT, undone INTEGER DEFAULT 0);
|
||||||
@@ -130,6 +130,58 @@ SKIP_DIRS = {".Trashes", ".Spotlight-V100", ".TemporaryItems", ".fseventsd", "Sy
|
|||||||
"$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"}
|
"$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"}
|
||||||
JUNK = (".DS_Store",)
|
JUNK = (".DS_Store",)
|
||||||
|
|
||||||
|
# ---------- CLUSTERING (relationship engine) ----------
|
||||||
|
STOP = set("""copy final new the and for with from test old bak backup untitled document file img
|
||||||
|
screenshot picture photo log notes data v1 v2 v3 final1 2 1 3 zip rar""".split())
|
||||||
|
|
||||||
|
def tokens(name):
|
||||||
|
stem = os.path.splitext(os.path.basename(name))[0].lower()
|
||||||
|
words = re.split(r"[^a-z0-9]+", stem)
|
||||||
|
return [w for w in words if len(w) > 2 and w not in STOP and not w.isdigit()]
|
||||||
|
|
||||||
|
def cluster_files(min_members=2):
|
||||||
|
"""Group files sharing rare name/preview tokens -> topic clusters (files that belong together)."""
|
||||||
|
c = db()
|
||||||
|
rows = c.execute("SELECT id,name,path,preview,file_class FROM files WHERE status NOT IN ('moved','held')").fetchall()
|
||||||
|
tok_map = {}
|
||||||
|
for r in rows:
|
||||||
|
seen = set(tokens(r["name"]))
|
||||||
|
if r["preview"]:
|
||||||
|
for t in tokens(r["preview"][:300]):
|
||||||
|
seen.add(t)
|
||||||
|
for t in seen:
|
||||||
|
tok_map.setdefault(t, set()).add(r["id"])
|
||||||
|
shared = {t: ids for t, ids in tok_map.items() if len(ids) >= min_members}
|
||||||
|
parent = {r["id"]: r["id"] for r in rows}
|
||||||
|
def find(x):
|
||||||
|
while parent[x] != x:
|
||||||
|
parent[x] = parent[parent[x]]; x = parent[x]
|
||||||
|
return x
|
||||||
|
for t, ids in shared.items():
|
||||||
|
ids = list(ids)
|
||||||
|
for x in ids[1:]:
|
||||||
|
ra, rb = find(ids[0]), find(x)
|
||||||
|
if ra != rb: parent[ra] = rb
|
||||||
|
groups = {}
|
||||||
|
for r in rows:
|
||||||
|
groups.setdefault(find(r["id"]), []).append(r)
|
||||||
|
clusters = []
|
||||||
|
for members in groups.values():
|
||||||
|
if len(members) < min_members:
|
||||||
|
continue
|
||||||
|
mids = {m["id"] for m in members}
|
||||||
|
toks = [(t, len(ids & mids)) for t, ids in shared.items() if ids & mids]
|
||||||
|
best = max(toks, key=lambda tv: (tv[1], len(tv[0])))[0] if toks else "cluster"
|
||||||
|
topic = best.replace("_", " ").strip().title()
|
||||||
|
for m in members:
|
||||||
|
role = {"source_code": "Code", "dataset": "Data", "archive": "Archives", "financial": "Docs",
|
||||||
|
"document": "Docs", "image": "Media", "video": "Media", "audio": "Media",
|
||||||
|
"ebook": "Docs", "software": "Installers", "personal": "Docs"}.get(m["file_class"], "Misc")
|
||||||
|
c.execute("UPDATE files SET topic=?, role=? WHERE id=?", (topic, role, m["id"]))
|
||||||
|
clusters.append({"topic": topic, "n": len(members), "members": [m["path"] for m in members]})
|
||||||
|
c.commit(); c.close()
|
||||||
|
return sorted(clusters, key=lambda x: -x["n"])
|
||||||
|
|
||||||
def scan(root, max_files=5000, min_size=0, max_size=0):
|
def scan(root, max_files=5000, min_size=0, max_size=0):
|
||||||
init_db()
|
init_db()
|
||||||
root = os.path.abspath(os.path.expanduser(root))
|
root = os.path.abspath(os.path.expanduser(root))
|
||||||
@@ -405,7 +457,7 @@ def build_plan(tx, min_conf):
|
|||||||
plans = []
|
plans = []
|
||||||
for row in rows:
|
for row in rows:
|
||||||
if row["action"] not in ("organize", "archive"): continue
|
if row["action"] not in ("organize", "archive"): continue
|
||||||
dst_dir = propose_destination(row, tx)
|
dst_dir = dest_for(row, tx)
|
||||||
name = row["rename"] if row["rename"] else row["name"]
|
name = row["rename"] if row["rename"] else row["name"]
|
||||||
base, ext = os.path.splitext(name)
|
base, ext = os.path.splitext(name)
|
||||||
if ext.lower() != row["ext"] and row["ext"]:
|
if ext.lower() != row["ext"] and row["ext"]:
|
||||||
@@ -416,10 +468,16 @@ def build_plan(tx, min_conf):
|
|||||||
plans.append({"id": row["id"], "path": row["path"], "name": row["name"], "dst": dst,
|
plans.append({"id": row["id"], "path": row["path"], "name": row["name"], "dst": dst,
|
||||||
"cls": row["file_class"], "subj": row["subject"], "imp": row["importance"],
|
"cls": row["file_class"], "subj": row["subject"], "imp": row["importance"],
|
||||||
"conf": row["confidence"], "verdict": verdict, "why": why, "dup": is_dup,
|
"conf": row["confidence"], "verdict": verdict, "why": why, "dup": is_dup,
|
||||||
"size": row["size"]})
|
"size": row["size"], "topic": row["topic"], "role": row["role"]})
|
||||||
return plans
|
return plans
|
||||||
|
|
||||||
# ---------- ORGANIZER (apply / undo) ----------
|
# ---------- ORGANIZER (apply / undo) ----------
|
||||||
|
def dest_for(row, tx):
|
||||||
|
"""Topic clusters stay together: Projects/<Topic>/<role>/. Everything else: taxonomy map."""
|
||||||
|
if row["topic"]:
|
||||||
|
return os.path.join(tx["organized_root"], "Projects", row["topic"], row["role"] or "Misc")
|
||||||
|
return propose_destination(row, tx)
|
||||||
|
|
||||||
def apply_moves(ids, dry_run=True):
|
def apply_moves(ids, dry_run=True):
|
||||||
tx = load_taxonomy()
|
tx = load_taxonomy()
|
||||||
c = db()
|
c = db()
|
||||||
@@ -428,7 +486,7 @@ def apply_moves(ids, dry_run=True):
|
|||||||
for fid in ids:
|
for fid in ids:
|
||||||
row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone()
|
row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone()
|
||||||
if not row or row["status"] not in ("classified", "approved"): skipped += 1; continue
|
if not row or row["status"] not in ("classified", "approved"): skipped += 1; continue
|
||||||
dst_dir = propose_destination(row, tx)
|
dst_dir = dest_for(row, tx)
|
||||||
name = row["rename"] or row["name"]
|
name = row["rename"] or row["name"]
|
||||||
base, ext = os.path.splitext(name)
|
base, ext = os.path.splitext(name)
|
||||||
if ext.lower() != row["ext"] and row["ext"]:
|
if ext.lower() != row["ext"] and row["ext"]:
|
||||||
@@ -566,7 +624,7 @@ def shell(body, active="home"):
|
|||||||
nav += f"<a class='{'on' if k==active else ''}' href='/{k}'>{label}</a>"
|
nav += f"<a class='{'on' if k==active else ''}' href='/{k}'>{label}</a>"
|
||||||
return ("<!doctype html><html><head><meta charset='utf-8'><meta name='viewport' content='width=device-width,initial-scale=1'>"
|
return ("<!doctype html><html><head><meta charset='utf-8'><meta name='viewport' content='width=device-width,initial-scale=1'>"
|
||||||
"<title>LIBRA — File Librarian</title><style>" + BASE_CSS + "</style></head><body>"
|
"<title>LIBRA — File Librarian</title><style>" + BASE_CSS + "</style></head><body>"
|
||||||
"<nav><div class='in'><span class='brand'>♎ LIBRA</span>" + nav + "</div></nav>"
|
"<nav><div class='in'><span class='brand'>♎ LIBRA</span><span id='nav-status' class='chip' style='display:none;background:rgba(53,208,186,.18);color:#6fe8d6'></span>" + nav + "</div></nav>"
|
||||||
"<div class='wrap'>" + body + "</div>"
|
"<div class='wrap'>" + body + "</div>"
|
||||||
"<div class='toast' id='toast'></div>"
|
"<div class='toast' id='toast'></div>"
|
||||||
"<script>window.LIBRA=" + json.dumps(STATE) + ";\n"
|
"<script>window.LIBRA=" + json.dumps(STATE) + ";\n"
|
||||||
@@ -574,6 +632,13 @@ def shell(body, active="home"):
|
|||||||
"var el=function(id){return document.getElementById(id)};"
|
"var el=function(id){return document.getElementById(id)};"
|
||||||
"if(el('st-files')){el('st-files').textContent=s.files;el('st-classified').textContent=s.classified;"
|
"if(el('st-files')){el('st-files').textContent=s.files;el('st-classified').textContent=s.classified;"
|
||||||
"el('st-dupes').textContent=s.dupes;el('st-pending').textContent=s.pending;el('st-moved').textContent=s.moved}"
|
"el('st-dupes').textContent=s.dupes;el('st-pending').textContent=s.pending;el('st-moved').textContent=s.moved}"
|
||||||
|
"var busy=s.scan_running||s.classify_running;"
|
||||||
|
"var msg='';if(s.scan_running){msg='scanning…'}else if(s.classify_running){msg='Nimble classifying '+s.classify_done+'/'+s.classify_total+'…';"
|
||||||
|
"if(el('pbar')){el('pbar').style.width=(100*s.classify_done/Math.max(1,s.classify_total))+'%'}}"
|
||||||
|
"else if(el('pbar')){el('pbar').style.width=(s.classify_total?100*s.classify_done/Math.max(1,s.classify_total):0)+'%';"
|
||||||
|
"if(s.classify_done>0){msg='classified '+s.classify_done+'/'+s.classify_total}}"
|
||||||
|
"var ch=el('nav-status');if(ch){ch.textContent=msg;ch.style.display=msg?'inline-block':'none'}"
|
||||||
|
"if(busy&&el('pmsg')){el('pmsg').textContent=msg}"
|
||||||
"window.LIBRA=s;setTimeout(pulse,2000)}).catch(function(){setTimeout(pulse,4000)})})();</script></body></html>")
|
"window.LIBRA=s;setTimeout(pulse,2000)}).catch(function(){setTimeout(pulse,4000)})})();</script></body></html>")
|
||||||
|
|
||||||
@app.route("/")
|
@app.route("/")
|
||||||
@@ -581,7 +646,7 @@ def home():
|
|||||||
refresh_counts()
|
refresh_counts()
|
||||||
err = f"<div class='panel' style='border-color:rgba(255,107,107,.4)'><h2 style='color:var(--bad)'>Last error</h2><pre style='white-space:pre-wrap;font-size:11px'>{STATE.get('last_error','')}</pre></div>" if STATE.get("last_error") else ""
|
err = f"<div class='panel' style='border-color:rgba(255,107,107,.4)'><h2 style='color:var(--bad)'>Last error</h2><pre style='white-space:pre-wrap;font-size:11px'>{STATE.get('last_error','')}</pre></div>" if STATE.get("last_error") else ""
|
||||||
scan_status = "<div class='bar'><i style='width:100%'></i></div><span class='muted'>scan running… refresh page</span>" if STATE["scan_running"] else ""
|
scan_status = "<div class='bar'><i style='width:100%'></i></div><span class='muted'>scan running… refresh page</span>" if STATE["scan_running"] else ""
|
||||||
cls_status = f"<div class='bar'><i style='width:{int(100*STATE['classify_done']/max(1,STATE['classify_total']))}%'></i></div><span class='muted'>classifying {STATE['classify_done']}/{STATE['classify_total']}…</span>" if STATE["classify_running"] else ""
|
cls_status = f"<div class='bar'><i id='pbar' style='width:{int(100*STATE['classify_done']/max(1,STATE['classify_total']))}%'></i></div><span class='muted'>classifying {STATE['classify_done']}/{STATE['classify_total']}…</span>" if STATE["classify_running"] else "<div class='bar'><i id='pbar' style='width:0%'></i></div><span class='muted' id='pmsg'>idle</span>"
|
||||||
body = f"""
|
body = f"""
|
||||||
<h1 class='grad'>File Librarian</h1><p class='muted'>Scanner → Extractor → Nimble → Safety → You approve → Reversible move. Nothing moves without your click.</p>
|
<h1 class='grad'>File Librarian</h1><p class='muted'>Scanner → Extractor → Nimble → Safety → You approve → Reversible move. Nothing moves without your click.</p>
|
||||||
<div class='grid'>
|
<div class='grid'>
|
||||||
@@ -599,7 +664,10 @@ def home():
|
|||||||
<div><label>Max size (bytes, 0=no cap)</label><input type='number' name='max_size' value='0'></div>
|
<div><label>Max size (bytes, 0=no cap)</label><input type='number' name='max_size' value='0'></div>
|
||||||
<div><button type='submit'>SCAN →</button></div>
|
<div><button type='submit'>SCAN →</button></div>
|
||||||
</div></form></div>
|
</div></form></div>
|
||||||
<div class='panel'><h2>2 · Hash + Classify (Nimble on nightmare)</h2>{cls_status}
|
<div class='panel'><h2>2 · Group what belongs together</h2>
|
||||||
|
<form method='post' action='/cluster'><button class='ghost'>CLUSTER RELATED FILES →</button></form>
|
||||||
|
<p class='muted'>Files sharing names/preview content get grouped into Projects/<Topic>/Docs·Code·Data·Media — moved together as one unit.</p></div>
|
||||||
|
<div class='panel'><h2>3 · Hash + Classify (Nimble on nightmare)</h2>{cls_status}
|
||||||
<form method='post' action='/hash'><div class='row'>
|
<form method='post' action='/hash'><div class='row'>
|
||||||
<div><label>Hash batch limit</label><input type='number' name='limit' value='500'></div>
|
<div><label>Hash batch limit</label><input type='number' name='limit' value='500'></div>
|
||||||
<div><label>Max MB to hash per file</label><input type='number' name='max_mb' value='64'></div>
|
<div><label>Max MB to hash per file</label><input type='number' name='max_mb' value='64'></div>
|
||||||
@@ -676,16 +744,17 @@ def plan_page():
|
|||||||
tx = load_taxonomy()
|
tx = load_taxonomy()
|
||||||
min_conf = float(request.args.get("min_conf", 0.6))
|
min_conf = float(request.args.get("min_conf", 0.6))
|
||||||
plans = build_plan(tx, min_conf)
|
plans = build_plan(tx, min_conf)
|
||||||
plans.sort(key=lambda p: (-p["conf"], p["verdict"] != "APPROVE"))
|
plans.sort(key=lambda p: (p.get("topic") or "zzz", -p["conf"], p["verdict"] != "APPROVE"))
|
||||||
cards = ""
|
cards = ""
|
||||||
approved_default = 0
|
approved_default = 0
|
||||||
for p in plans[:600]:
|
for p in plans[:600]:
|
||||||
checked = "checked" if p["verdict"] == "APPROVE" else ""
|
checked = "checked" if p["verdict"] == "APPROVE" else ""
|
||||||
if p["verdict"] == "APPROVE": approved_default += 1
|
if p["verdict"] == "APPROVE": approved_default += 1
|
||||||
dupnote = "<span class='chip' style='background:rgba(255,180,84,.16);color:#ffcf8f'>DUPLICATE GROUP</span>" if p["dup"] else ""
|
dupnote = "<span class='chip' style='background:rgba(255,180,84,.16);color:#ffcf8f'>DUPLICATE GROUP</span>" if p["dup"] else ""
|
||||||
|
topicnote = f"<span class='chip' style='background:rgba(124,108,255,.2);color:#c9bfff'>◆ {p['topic']} · {p['role']}</span>" if p.get("topic") else ""
|
||||||
cards += (f"<div class='plan'>"
|
cards += (f"<div class='plan'>"
|
||||||
f"<div class='src'>{p['path']}</div><div class='arrow'>↓</div><div class='dst'>{p['dst']}</div>"
|
f"<div class='src'>{p['path']}</div><div class='arrow'>↓</div><div class='dst'>{p['dst']}</div>"
|
||||||
f"<div class='meta'><span class='chip c-{p['cls']}'>{p['cls']}</span><span class='muted'>{p['subj']} · {p['imp']}</span>"
|
f"<div class='meta'>{topicnote}<span class='chip c-{p['cls']}'>{p['cls']}</span><span class='muted'>{p['subj']} · {p['imp']}</span>"
|
||||||
f"<span class='muted'>conf {int(p['conf']*100)}%</span>"
|
f"<span class='muted'>conf {int(p['conf']*100)}%</span>"
|
||||||
f"<span class='v-{p['verdict']}' style='font-weight:700'>{p['verdict']}</span>"
|
f"<span class='v-{p['verdict']}' style='font-weight:700'>{p['verdict']}</span>"
|
||||||
+ (f"<span class='muted'>({p['why']})</span>" if p["why"] else "") + dupnote +
|
+ (f"<span class='muted'>({p['why']})</span>" if p["why"] else "") + dupnote +
|
||||||
@@ -726,6 +795,11 @@ def api_apply():
|
|||||||
done, skipped, session = apply_moves(ids, dry_run=bool(d.get("dry_run", True)))
|
done, skipped, session = apply_moves(ids, dry_run=bool(d.get("dry_run", True)))
|
||||||
return {"done": done, "skipped": skipped, "session": session}
|
return {"done": done, "skipped": skipped, "session": session}
|
||||||
|
|
||||||
|
@app.route("/cluster", methods=["POST"])
|
||||||
|
def cluster_post():
|
||||||
|
cluster_files()
|
||||||
|
return redirect("/plan")
|
||||||
|
|
||||||
@app.route("/dupes")
|
@app.route("/dupes")
|
||||||
def dupes_page():
|
def dupes_page():
|
||||||
c = db()
|
c = db()
|
||||||
@@ -829,4 +903,4 @@ def settings_save():
|
|||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
init_db()
|
init_db()
|
||||||
refresh_counts()
|
refresh_counts()
|
||||||
app.run(host="127.0.0.1", port=5731, debug=False)
|
app.run(host="0.0.0.0", port=5731, debug=False)
|
||||||
|
|||||||
Reference in New Issue
Block a user