diff --git a/app.py b/app.py index 4bba324..d6d6545 100644 --- a/app.py +++ b/app.py @@ -77,12 +77,61 @@ CREATE TABLE IF NOT EXISTS sessions( """) c.commit(); c.close() +# deterministic fast-path: obvious extensions never need the LLM (instant + accurate) +QUICK_RULES = { + ".pdf": ("document", None, None, 0.75), ".doc": ("document", None, None, 0.75), ".docx": ("document", None, None, 0.75), + ".xlsx": ("document", "finance", None, 0.8), ".xls": ("document", "finance", None, 0.8), ".pptx": ("document", "work", None, 0.8), + ".txt": ("document", None, None, 0.6), ".md": ("document", None, None, 0.6), ".csv": ("dataset", None, None, 0.8), + ".jpg": ("image", "photography", None, 0.9), ".jpeg": ("image", "photography", None, 0.9), ".png": ("image", None, None, 0.85), + ".heic": ("image", "photography", None, 0.9), ".gif": ("image", None, None, 0.85), ".webp": ("image", None, None, 0.85), + ".psd": ("image", None, None, 0.8), + ".mp4": ("video", None, None, 0.9), ".mov": ("video", None, None, 0.9), ".mkv": ("video", None, None, 0.9), ".avi": ("video", None, None, 0.9), + ".mp3": ("audio", None, None, 0.9), ".wav": ("audio", None, None, 0.9), ".flac": ("audio", None, None, 0.9), + ".zip": ("archive", None, None, 0.85), ".rar": ("archive", None, None, 0.85), ".7z": ("archive", None, None, 0.85), + ".tar": ("archive", None, None, 0.85), ".gz": ("archive", None, None, 0.85), ".iso": ("archive", None, None, 0.8), + ".py": ("source_code", "programming", None, 0.92), ".js": ("source_code", "programming", None, 0.92), ".ts": ("source_code", "programming", None, 0.92), + ".sh": ("source_code", "linux", None, 0.9), ".go": ("source_code", "programming", None, 0.92), ".rs": ("source_code", "programming", None, 0.92), + ".c": ("source_code", "programming", None, 0.9), ".cpp": ("source_code", "programming", None, 0.9), ".java": ("source_code", "programming", None, 0.92), + ".exe": ("software", None, None, 0.9), ".msi": ("software", None, None, 0.9), ".dmg": ("software", None, None, 0.9), ".app": ("software", None, None, 0.9), + ".deb": ("software", "linux", None, 0.9), ".rpm": ("software", "linux", None, 0.9), ".apk": ("software", None, None, 0.9), + ".ova": ("dataset", None, None, 0.7), ".vmdk": ("dataset", None, None, 0.7), ".sqlite": ("dataset", None, None, 0.8), ".db": ("dataset", None, None, 0.8), + ".epub": ("ebook", None, None, 0.95), ".mobi": ("ebook", None, None, 0.95), ".azw3": ("ebook", None, None, 0.95), ".djvu": ("ebook", None, None, 0.9), +} +# exts that LOOK like ebooks/docs but need content sniffing (pdf can be either) — handled by classifier + +# content signals — deterministic, beat quick rules when the content screams (receipt, fullz, hacking…) +SIGNALS = [ + (r"receipt|invoice|order confirmation|statement|payment|charged \$|billing", "financial", "finance", "important", "organize", 0.97), + (r"cvv|fullz|ssn|card number|carding|dump[s]? ", "personal", "unknown", "critical", "quarantine", 0.95), + (r"hacking|exploit|keylogger|rat |botnet|malware", "document", "cybersecurity", "important", "quarantine", 0.93), + (r"chapter one|novel|fiction|once upon", "document", "personal", "important", "organize", 0.9), +] + +def sniff_content(p, ext, size): + if ext not in (".txt", ".md", ".csv", ".json", ".log", ".pdf", ".html", ".cfg", ".ini") or size > 5_000_000: + return None + try: + if ext == ".pdf": + from pypdf import PdfReader + txt = " ".join((pg.extract_text() or "") for pg in PdfReader(p).pages[:2]) + else: + txt = open(p, "r", errors="replace").read(4000) + except Exception: + return None + low = txt.lower() + for rx, fc, sub, imp, act, conf in SIGNALS: + if re.search(rx, low): + return {"file_class": fc, "subject": sub, "importance": imp, "action": act, + "confidence": conf, "rename": None, "preview": re.sub(r"\s+", " ", txt)[:600]} + return None + # ---------- SCANNER ---------- SKIP_DIRS = {".Trashes", ".Spotlight-V100", ".TemporaryItems", ".fseventsd", "System Volume Information", "$RECYCLE.BIN", "node_modules", "__pycache__", ".git", "Library", ".Trash"} JUNK = (".DS_Store",) def scan(root, max_files=5000, min_size=0, max_size=0): + init_db() root = os.path.abspath(os.path.expanduser(root)) c = db() n = 0 @@ -99,6 +148,22 @@ def scan(root, max_files=5000, min_size=0, max_size=0): if st.st_size < min_size or (max_size and st.st_size > max_size): continue try: + qr = QUICK_RULES.get(os.path.splitext(f)[1].lower()) + if qr: + fc, sub, imp, conf = qr + sig = sniff_content(p, os.path.splitext(f)[1].lower(), st.st_size) + if sig: + fc, sub, imp, conf = sig["file_class"], sig["subject"], sig["importance"], sig["confidence"] + c.execute("""INSERT OR IGNORE INTO files(path,name,ext,size,mtime,file_class,subject,importance,confidence,action,preview,status) + VALUES(?,?,?,?,?,?,?,?,?,?,?, 'classified')""", + (p, f, os.path.splitext(f)[1].lower(), st.st_size, st.st_mtime, fc, sub, imp, conf, sig["action"], sig["preview"])) + n += 1 + continue + c.execute("""INSERT OR IGNORE INTO files(path,name,ext,size,mtime,file_class,subject,importance,confidence,action,status) + VALUES(?,?,?,?,?,?,?,?,?,'organize','classified')""", + (p, f, os.path.splitext(f)[1].lower(), st.st_size, st.st_mtime, fc, sub, imp or 'normal', conf)) + n += 1 + continue c.execute("INSERT OR IGNORE INTO files(path,name,ext,size,mtime,status) VALUES(?,?,?,?,?, 'scanned')", (p, f, os.path.splitext(f)[1].lower(), st.st_size, st.st_mtime)) except sqlite3.Error: @@ -124,15 +189,18 @@ def scan_thread(root, max_files, min_size, max_size): refresh_counts() def refresh_counts(): - c = db() - for k, q in [("files", "SELECT COUNT(*) FROM files"), + try: + c = db() + for k, q in [("files", "SELECT COUNT(*) FROM files"), ("scanned", "SELECT COUNT(*) FROM files WHERE file_class IS NULL"), ("classified", "SELECT COUNT(*) FROM files WHERE file_class IS NOT NULL"), ("dupes", "SELECT COUNT(*) FROM files WHERE sha256 IS NOT NULL AND sha256 IN (SELECT sha256 FROM files WHERE sha256 IS NOT NULL GROUP BY sha256 HAVING COUNT(*)>1)"), ("pending", "SELECT COUNT(*) FROM files WHERE status='approved'"), ("moved", "SELECT COUNT(*) FROM files WHERE status='moved'")]: - STATE[k] = c.execute(q).fetchone()[0] - c.close() + STATE[k] = c.execute(q).fetchone()[0] + c.close() + except Exception: + pass # ---------- EXTRACTOR ---------- def extract_preview(p, ext, size): @@ -198,7 +266,16 @@ ENUMS = { "action": ["keep","organize","archive","quarantine","review"], } def classify_one(row): - prompt = ("Classify this file. Reply ONLY minified JSON with keys file_class, subject, importance, action, confidence, rename. No other text.\n" + prompt = ("You are a file-organizer classifier. Reply ONLY minified JSON: " + "{\"file_class\":\"..\",\"subject\":\"..\",\"importance\":\"..\",\"action\":\"..\",\"confidence\":0.0,\"rename\":\"..\"}\n" + "Examples:\n" + "Input: invoice_stripe_march.pdf (PDF, preview: 'Stripe payment receipt $120.00 subscription') -> " + "{\"file_class\":\"financial\",\"subject\":\"finance\",\"importance\":\"important\",\"action\":\"organize\",\"confidence\":0.97,\"rename\":\"2026-03_stripe_invoice.pdf\"}\n" + "Input: novel_draft_v2.docx (DOCX, preview: 'Chapter One. The lighthouse keeper woke...') -> " + "{\"file_class\":\"document\",\"subject\":\"personal\",\"importance\":\"important\",\"action\":\"organize\",\"confidence\":0.9,\"rename\":\"novel_draft_v2.docx\"}\n" + "Input: dumps_fullz_usa.txt (TXT, preview: 'CC number exp cvv ssn address...') -> " + "{\"file_class\":\"personal\",\"subject\":\"unknown\",\"importance\":\"critical\",\"action\":\"quarantine\",\"confidence\":0.95,\"rename\":\"dumps_fullz_usa.txt\"}\n" + "Now classify:\n" f"File: {row['name']} ({row['ext'] or 'no ext'}, {row['size']} bytes)\n" + (f"Content preview: {row['preview'][:400]}\n" if row['preview'] else "") + "file_class in " + str(ENUMS["file_class"]) + "; subject in " + str(ENUMS["subject"]) @@ -206,9 +283,10 @@ def classify_one(row): + "; confidence 0.0-1.0; rename = clean proposed filename or same") import urllib.request body = json.dumps({"model": MODEL, "prompt": prompt, "stream": False, "think": False, - "options": {"temperature": 0.1, "num_predict": 400}}).encode() + "keep_alive": -1, + "options": {"temperature": 0.05, "num_predict": 220}}).encode() req = urllib.request.Request(OLLAMA + "/api/generate", data=body, headers={"Content-Type": "application/json"}) - with urllib.request.urlopen(req, timeout=120) as r: + with urllib.request.urlopen(req, timeout=60) as r: d = json.loads(r.read()) txt = d.get("response", "") m = re.search(r"\{.*\}", txt, re.S) @@ -230,22 +308,51 @@ def classify_one(row): out["rename"] = re.sub(r"[^A-Za-z0-9 ._-]", "", rn)[:120] or row["name"] return out +def classify_one_file(fid): + c = db() + row = c.execute("SELECT * FROM files WHERE id=?", (fid,)).fetchone() + if not row or (row["file_class"] and row["subject"]): + c.close(); return None + if row["preview"] is None: + prev = extract_preview(row["path"], row["ext"], row["size"]) + c.execute("UPDATE files SET preview=? WHERE id=?", (prev, fid)) + row = dict(row); row["preview"] = prev + try: + res = classify_one(row) + except Exception: + res = None + if not res: + # deterministic fallback — NEVER leave NULLs or stall the pipeline + qr = QUICK_RULES.get(row["ext"]) or ("unknown", "unknown", "normal", 0.3) + res = {"file_class": qr[0], "subject": qr[1] or "unknown", "importance": qr[2] or "normal", + "action": "review", "confidence": qr[3], "rename": row["name"]} + c.execute("UPDATE files SET file_class=?,subject=?,importance=?,action=?,confidence=?,rename=?,status='classified' WHERE id=?", + (res["file_class"], res["subject"], res["importance"], res["action"], res["confidence"], res["rename"], fid)) + c.commit() + c.close() + return res + def classify_thread(limit): + from concurrent.futures import ThreadPoolExecutor, as_completed try: with LOCK: STATE["classify_running"] = True c = db() - rows = c.execute("SELECT * FROM files WHERE file_class IS NULL LIMIT ?", (limit,)).fetchall() - STATE["classify_total"] = len(rows); STATE["classify_done"] = 0 - for row in rows: - res = classify_one(row) - if res: - if row["preview"] is None: - prev = extract_preview(row["path"], row["ext"], row["size"]) - c.execute("UPDATE files SET preview=? WHERE id=?", (prev, row["id"])) - c.execute("UPDATE files SET file_class=?,subject=?,importance=?,action=?,confidence=?,rename=?,status='classified' WHERE id=?", - (res["file_class"], res["subject"], res["importance"], res["action"], res["confidence"], res["rename"], row["id"])) - STATE["classify_done"] += 1 - c.commit(); c.close() + fids = [r["id"] for r in c.execute("SELECT id FROM files WHERE file_class IS NULL LIMIT ?", (limit,)).fetchall()] + c.close() + STATE["classify_total"] = len(fids); STATE["classify_done"] = 0 + # warm the model once (cold load can take 40s; warm calls ~4s) + try: + import urllib.request + body = json.dumps({"model": MODEL, "prompt": "warmup", "stream": False, "think": False, + "keep_alive": -1, "options": {"num_predict": 1}}).encode() + urllib.request.urlopen(urllib.request.Request(OLLAMA + "/api/generate", data=body, + headers={"Content-Type": "application/json"}), timeout=90).read() + except Exception: + pass + with ThreadPoolExecutor(max_workers=4) as ex: + futs = {ex.submit(classify_one_file, fid): fid for fid in fids} + for f in as_completed(futs): + STATE["classify_done"] += 1 except Exception: STATE["last_error"] = traceback.format_exc()[-800:] finally: @@ -462,7 +569,12 @@ def shell(body, active="home"): "" "