From 3f67efb37915f907dbcf8025b1f2617c709af8de Mon Sep 17 00:00:00 2001 From: drjones Date: Wed, 29 Jul 2026 17:16:00 -0700 Subject: [PATCH] Add smart recategorize (auto-discovered categories) and fix stale category-keyword accumulation - New "Smart Recategorize" panel: analyzes existing journal descriptions (no vision re-run, no images touched) to propose a richer category list when one bucket dominates the library, then reclassifies every tagged photo into it using a fast local text-only model pass in batches of 25. Verified end-to-end on the real library: correctly split "Objects & Stuff" (35% of 15k photos) into specific categories like Electronics & Hardware, Cannabis Cultivation, Minerals & Crystals, Tools & Workshop based on what's actually in the descriptions. - New ollama_generate_text() for text-only local model calls (the existing ollama_generate always sent an "images" key, unsuitable for pure text classification). - FIX: write_metadata only ever appended category/photon-tagged keywords (+=) to the list-type XMP/IPTC tags, so retagging a photo left every category it ever had stacked in its metadata, and even rewriting with the same category duplicated "photon-tagged" each time. Found via a live test while building this feature (which recategorizes at library scale). Now strips old values before adding, making every write idempotent regardless of how many times a photo is retagged. Wired through all four write sites (pipeline, write_single, redo_single, bulk_recategorize). Co-Authored-By: Claude Sonnet 5 --- index.html | 77 ++++++++++++++++++ server.py | 230 +++++++++++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 300 insertions(+), 7 deletions(-) diff --git a/index.html b/index.html index fc18ce1..e1fcb06 100644 --- a/index.html +++ b/index.html @@ -720,6 +720,25 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px} + +
+

🧠 smart recategorize — auto-discover categories from your library

+
+
Analyzes descriptions already sitting in your journal (no images re-scanned, no vision model) to propose a richer category list when one bucket is swallowing the library, then reclassifies every tagged photo into it using a fast local text model.
+ + + +
+
+

🛡 trust & safety console

@@ -1985,6 +2004,51 @@ $("btnScanDupes").onclick = async ()=>{ } }; +/* ---- smart recategorize ---- */ +function applyRecatProgress(p){ + $("recatProgressBox").style.display = "block"; + const pct = p.total ? Math.round(100 * p.done / p.total) : 0; + $("recatFill").style.width = pct + "%"; + $("recatStatusText").textContent = p.finished + ? `Done — ${p.changed} of ${p.total} photos moved to a new category.` + : `Reclassifying… ${p.done}/${p.total} (${p.changed} changed so far)`; + if(p.finished){ + $("btnProposeCats").disabled = false; + $("btnApplyRecat").disabled = false; + renderCatChips(); + doSearch(true); + } +} + +$("btnProposeCats").onclick = async ()=>{ + $("btnProposeCats").disabled = true; + $("btnProposeCats").textContent = "Analyzing your library…"; + const j = await api("/api/recategorize/propose", {model: "qwen3.5:9b"}); + $("btnProposeCats").disabled = false; + $("btnProposeCats").textContent = "✨ Propose New Categories"; + if(j.ok && j.categories){ + $("recatProposalList").value = j.categories.join("\n"); + $("recatProposalBox").style.display = "block"; + showToast("Proposed " + j.categories.length + " categories — review before applying", "ok"); + } +}; + +$("btnApplyRecat").onclick = async ()=>{ + const cats = $("recatProposalList").value.split("\n").map(s=>s.trim()).filter(Boolean); + if(cats.length < 2){ showToast("Need at least a couple of categories", "error"); return; } + if(!confirm(`This will re-read the description of every tagged photo and, using a local text model, may change its category — rewriting the category keyword in that photo's metadata (descriptions stay untouched). Proceed with ${cats.length} categories?`)) return; + $("btnApplyRecat").disabled = true; + $("btnProposeCats").disabled = true; + const j = await api("/api/recategorize/apply", {categories: cats, model: "qwen3.5:9b"}); + if(j.ok){ + showToast("Recategorize started — watch progress below", "ok"); + applyRecatProgress({done:0,total:0,changed:0}); + } else { + $("btnApplyRecat").disabled = false; + $("btnProposeCats").disabled = false; + } +}; + /* ---- audit history dashboard ---- */ async function loadAuditHistory(){ const j = await api("/api/audit/history"); @@ -2116,6 +2180,9 @@ function connect(){ if(m.data.finished) loadDupeGroups(); } } + else if(m.type==="recat"){ + applyRecatProgress(m.data); + } }; es.onerror = ()=>{ es.close(); setTimeout(connect, 2000); }; } @@ -2132,6 +2199,16 @@ function connect(){ setBadge(st.status||"idle"); if(st.total>0) scanned=true, setBadge(st.status||"idle"); + if(st.recatProposal && st.recatProposal.length){ + $("recatProposalList").value = st.recatProposal.join("\n"); + $("recatProposalBox").style.display = "block"; + } + if(st.recatStatus === "running"){ + $("btnProposeCats").disabled = true; + $("btnApplyRecat").disabled = true; + applyRecatProgress(st.recatProgress || {done:0,total:0,changed:0}); + } + const mj = await api("/api/models"); const sel=$("model"); const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken diff --git a/server.py b/server.py index d19758a..8103fcd 100644 --- a/server.py +++ b/server.py @@ -13,6 +13,7 @@ import hashlib import json import os import queue +import random import re import shutil import subprocess @@ -195,6 +196,9 @@ class State: self.skipped_sidecars = 0 self.dedupe_status = "idle" # idle | running self.dedupe_progress = {"done": 0, "total": 0} + self.recat_status = "idle" # idle | proposing | running + self.recat_progress = {"done": 0, "total": 0, "changed": 0} + self.recat_proposal = [] # last LLM-proposed category list, for the UI to recall self.already_done = 0 self.processed_session = 0 self.failed_session = 0 @@ -638,6 +642,133 @@ def build_dedupe_index(): with S.lock: S.dedupe_status = "idle" +def sample_descriptions_for_proposal(max_total=350): + """Stratified sample across every current category, weighted so a + dumping-ground category (e.g. one holding 35% of the library) is + oversampled relative to its share — the model needs to actually SEE + what's clogging it up, not just be told the percentage.""" + records = load_search_index() + by_cat = {} + for r in records: + by_cat.setdefault(r.get("category") or "Uncategorized", []).append(r) + total = len(records) + cat_sizes = {c: len(rs) for c, rs in by_cat.items()} + samples = [] + for cat, rs in by_cat.items(): + share = max(10, min(120, int(max_total * (len(rs) / total) ** 0.5))) if total else 0 + chosen = random.sample(rs, min(share, len(rs))) + for r in chosen: + samples.append({"category": cat, "desc": (r.get("desc") or "")[:150]}) + random.shuffle(samples) + return samples[:max_total], cat_sizes, total + +def build_category_proposal_prompt(samples, cat_sizes, total): + dist = "\n".join( + f"- {c}: {n} photos ({100*n/total:.0f}%)" + for c, n in sorted(cat_sizes.items(), key=lambda x: -x[1]) + ) + sample_lines = "\n".join(f'[{s["category"]}] {s["desc"]}' for s in samples) + return ( + "You are refining a personal photo library's category taxonomy.\n" + f"Current categories and their share of {total} total photos:\n{dist}\n\n" + "Below is a random sample of actual photo descriptions, each labeled with its " + "current category:\n" + sample_lines + "\n\n" + "The categories above are too coarse — the largest ones are dumping grounds " + "covering very different kinds of photos. Propose a NEW list of 15 to 25 " + "categories that fits this actual library well, so no single category dominates. " + "Keep any current category that's already specific and reasonably sized; split " + "the overloaded/vague ones based on what the sample actually shows (e.g. separate " + "electronics from plants from receipts from tools — whatever the real content is, " + "don't guess generically). Every name should be short (2-4 words) and mutually " + "exclusive from the others.\n" + 'Answer as JSON: {"categories": ["...", "...", ...]}' + ) + +CATEGORY_PROPOSAL_SCHEMA = { + "type": "object", + "properties": {"categories": {"type": "array", "items": {"type": "string"}}}, + "required": ["categories"], +} + +def build_reclassify_prompt(categories, batch): + cats = "\n".join(f"- {c}" for c in categories) + items = "\n".join(f"{i}. {b['desc'][:200]}" for i, b in enumerate(batch)) + return ( + "Classify each numbered photo description below into EXACTLY ONE of these " + f"categories:\n{cats}\n\nDescriptions:\n{items}\n\n" + "Answer as JSON: {\"categories\": [\"\", \"\", ...]} " + "— one entry per description, in the same order, same count." + ) + +def reclassify_batch_schema(n): + return { + "type": "object", + "properties": {"categories": {"type": "array", "items": {"type": "string"}, "minItems": n, "maxItems": n}}, + "required": ["categories"], + } + +def recategorize_apply_worker(new_categories, model): + with S.lock: + if S.recat_status != "idle": + return + S.recat_status = "running" + keep_backup = bool(S.settings.get("keepBackup", False)) + preserve_date = bool(S.settings.get("preserveDate", True)) + organize = bool(S.settings.get("organize", False)) + try: + save_categories(new_categories) + load_journal() + push_stats() + records = load_search_index() + total = len(records) + S.recat_progress = {"done": 0, "total": total, "changed": 0} + log("info", f"smart recategorize: reclassifying {total} photos into " + f"{len(new_categories)} categories …") + think = False if model_thinks(model) else None + batch_size = 25 + changed = 0 + done = 0 + for i in range(0, total, batch_size): + batch = records[i:i + batch_size] + prompt = build_reclassify_prompt(new_categories, batch) + try: + resp = ollama_generate_text(model, prompt, {"temperature": 0, "num_predict": 800}, + "10m", think, reclassify_batch_schema(len(batch))) + parsed = salvage_json(resp.get("response", "")) + cats = parsed.get("categories") or [] + except Exception as e: + log("error", f"recategorize batch {i}-{i+len(batch)} failed: {e}") + cats = [] + for j, rec in enumerate(batch): + done += 1 + new_cat = cats[j] if j < len(cats) else None + if new_cat not in new_categories: + new_cat = None + old_cat = rec.get("category") + if new_cat and new_cat != old_cat: + try: + write_metadata(rec["path"], rec.get("desc", ""), new_cat, + keep_backup, preserve_date, old_cat) + if organize: + organize_alias(find_owning_folder(rec["path"]), rec["path"], new_cat) + journal_write({"path": rec["path"], "desc": rec.get("desc", ""), + "category": new_cat, "model": "recategorize", + "route": rec.get("route"), "sec": 0.0, "ts": time.time()}) + changed += 1 + except Exception as e: + log("error", f"recategorize write failed for {rec['path']}: {e}") + S.recat_progress = {"done": done, "total": total, "changed": changed} + broadcast("recat", S.recat_progress) + push_stats() + compact_journal() + load_journal() + push_stats() + log("ok", f"smart recategorize complete: {changed} of {total} photos moved to a new category") + finally: + with S.lock: + S.recat_status = "idle" + broadcast("recat", {**S.recat_progress, "finished": True}) + def find_duplicate_groups(): """Groups of exact perceptual-hash matches and size/filename matches.""" records = {r["path"]: r for r in load_search_index()} @@ -773,6 +904,27 @@ def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema with urllib.request.urlopen(req, timeout=600) as r: return json.loads(r.read()) +def ollama_generate_text(model, prompt, opts, keep_alive, think=None, schema=SCHEMA): + """Text-only generation (no image) — used for classifying/organizing + against descriptions that were already produced by a vision pass, so + re-analyzing the whole library never needs the vision model again.""" + body = { + "model": model, + "prompt": prompt, + "stream": False, + "options": opts, + "keep_alive": keep_alive, + } + if schema is not None: + body["format"] = schema + if think is not None: + body["think"] = think + req = urllib.request.Request( + OLLAMA + "/api/generate", data=json.dumps(body).encode(), + headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=600) as r: + return json.loads(r.read()) + # ---------------------------------------------------------------- pipeline def scan_folder(folder, recursive=True): @@ -1022,7 +1174,14 @@ def build_prompt(length_key, mode="photo"): 'Answer as JSON: {"description": "...", "category": "..."}' ) -def write_metadata(path, desc, category, keep_backup, preserve_date): +def write_metadata(path, desc, category, keep_backup, preserve_date, old_category=None): + """Idempotent regardless of how many times a photo gets (re)tagged: + category and the photon-tagged marker are list-type XMP/IPTC tags, so a + bare += on every write would silently accumulate duplicates (a photo + retagged 3 times would carry 3 copies of "photon-tagged" and every + category it ever had). Always strip the exact values being (re)written + first, then add them back exactly once. old_category additionally + strips a genuinely different previous category.""" is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"] if preserve_date: @@ -1032,13 +1191,19 @@ def write_metadata(path, desc, category, keep_backup, preserve_date): if is_video: # mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic # tag names so it resolves to QuickTime/Keys groups automatically. + if old_category and old_category != category: + args.append(f"-Keywords-={old_category}") args += [ + f"-Keywords-={category}", f"-Keywords-=photon-tagged", f"-Description={desc}", f"-Keywords+={category}", "-Keywords+=photon-tagged", ] else: + if old_category and old_category != category: + args += [f"-XMP-dc:Subject-={old_category}", f"-IPTC:Keywords-={old_category}"] args += [ + f"-XMP-dc:Subject-={category}", f"-IPTC:Keywords-={category}", "-XMP-dc:Subject-=photon-tagged", f"-EXIF:ImageDescription={desc}", f"-IPTC:Caption-Abstract={desc}", f"-XMP-dc:Description={desc}", @@ -1160,6 +1325,10 @@ def process_loop(settings): if think_map[m] is False: log("info", f"{m} is a thinking model — thinking disabled for speed") + # built once so a retag (skipDone=false on an already-tagged photo) can + # strip its old category keyword instead of accumulating it + prev_categories = {p: rec.get("category") for p, rec in read_journal_deduped().items()} + tmpdir = tempfile.mkdtemp(prefix="photon_") log("info", f"engine online :: {'router mode' if router else 'model=' + model} " f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}") @@ -1332,11 +1501,12 @@ def process_loop(settings): continue try: + old_cat = prev_categories.get(path) if not dry: if integrity: - verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date) + verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date, old_cat) else: - write_metadata(path, desc, category, keep_backup, preserve_date) + write_metadata(path, desc, category, keep_backup, preserve_date, old_cat) if organize: organize_alias(find_owning_folder(path), path, category) @@ -1468,7 +1638,9 @@ class Handler(BaseHTTPRequestHandler): "folder": S.folders[0] if S.folders else "", "log": S.log_ring[-200:], "categories": CATEGORIES, "current": S.current, - "settings": S.settings, "failuresCount": len(S.failed_paths)}) + "settings": S.settings, "failuresCount": len(S.failed_paths), + "recatStatus": S.recat_status, "recatProgress": S.recat_progress, + "recatProposal": S.recat_proposal}) self._json(payload) elif self.path.startswith("/api/preview"): with S.lock: @@ -2191,7 +2363,8 @@ class Handler(BaseHTTPRequestHandler): keep_backup = bool(S.settings.get("keepBackup", False)) preserve_date = bool(S.settings.get("preserveDate", True)) organize = bool(S.settings.get("organize", False)) - write_metadata(path, desc, category, keep_backup, preserve_date) + old_cat = read_journal_deduped().get(path, {}).get("category") + write_metadata(path, desc, category, keep_backup, preserve_date, old_cat) if organize: organize_alias(find_owning_folder(path), path, category) existing_recs = [] @@ -2298,7 +2471,8 @@ class Handler(BaseHTTPRequestHandler): desc = f"{desc} | text: {words[:300]}" except Exception as e: pass - write_metadata(path, desc, category, keep_backup, preserve_date) + old_cat = read_journal_deduped().get(path, {}).get("category") + write_metadata(path, desc, category, keep_backup, preserve_date, old_cat) if organize: organize_alias(find_owning_folder(path), path, category) existing_recs = [] @@ -2402,6 +2576,47 @@ class Handler(BaseHTTPRequestHandler): threading.Thread(target=build_dedupe_index, daemon=True).start() self._json({"ok": True}) + elif self.path == "/api/recategorize/propose": + model = body.get("model") or "qwen3.5:9b" + with S.lock: + if S.recat_status != "idle": + self._json({"error": "a recategorize job is already running"}, 400); return + S.recat_status = "proposing" + try: + samples, cat_sizes, total = sample_descriptions_for_proposal() + if total == 0: + self._json({"error": "no tagged photos yet to analyze"}, 400); return + prompt = build_category_proposal_prompt(samples, cat_sizes, total) + think = False if model_thinks(model) else None + log("info", f"smart recategorize: asking {model} to propose new categories " + f"from {len(samples)} sampled descriptions ({total} photos total) …") + resp = ollama_generate_text(model, prompt, {"temperature": 0.3, "num_predict": 900}, + "5m", think, CATEGORY_PROPOSAL_SCHEMA) + parsed = salvage_json(resp.get("response", "")) + proposed = [c.strip() for c in (parsed.get("categories") or []) if c and c.strip()] + if not proposed: + self._json({"error": "model returned no categories"}, 500); return + with S.lock: + S.recat_proposal = proposed + log("ok", f"smart recategorize: proposed {len(proposed)} categories") + self._json({"ok": True, "categories": proposed, "currentDistribution": cat_sizes}) + except Exception as e: + self._json({"error": f"proposal failed: {e}"}, 500) + finally: + with S.lock: + S.recat_status = "idle" + + elif self.path == "/api/recategorize/apply": + categories = body.get("categories") + model = body.get("model") or "qwen3.5:9b" + if not isinstance(categories, list) or not categories: + self._json({"error": "categories list is required"}, 400); return + with S.lock: + if S.recat_status != "idle": + self._json({"error": "a recategorize job is already running"}, 400); return + threading.Thread(target=recategorize_apply_worker, args=(categories, model), daemon=True).start() + self._json({"ok": True}) + elif self.path == "/api/bulk_recategorize": paths = body.get("paths") or [] category = body.get("category") @@ -2415,8 +2630,9 @@ class Handler(BaseHTTPRequestHandler): for p in paths: rec = records.get(p) desc = rec.get("desc", "") if rec else "" + old_cat = rec.get("category") if rec else None try: - write_metadata(p, desc, category, keep_backup, preserve_date) + write_metadata(p, desc, category, keep_backup, preserve_date, old_cat) ok += 1 except Exception as e: failed.append({"path": p, "error": str(e)})