🧠smart recategorize — auto-discover categories from your library
+
+
Analyzes descriptions already sitting in your journal (no images re-scanned, no vision model) to propose a richer category list when one bucket is swallowing the library, then reclassifies every tagged photo into it using a fast local text model.
+
+
+
+
+
+
This rewrites the category keyword in every tagged photo's metadata (descriptions are untouched). Reversible per-photo the same way any recategorize is — nothing about the image itself changes.
+
+
+
+
starting…
+
+
+
+
🛡 trust & safety console
@@ -1985,6 +2004,51 @@ $("btnScanDupes").onclick = async ()=>{
}
};
+/* ---- smart recategorize ---- */
+function applyRecatProgress(p){
+ $("recatProgressBox").style.display = "block";
+ const pct = p.total ? Math.round(100 * p.done / p.total) : 0;
+ $("recatFill").style.width = pct + "%";
+ $("recatStatusText").textContent = p.finished
+ ? `Done — ${p.changed} of ${p.total} photos moved to a new category.`
+ : `Reclassifying… ${p.done}/${p.total} (${p.changed} changed so far)`;
+ if(p.finished){
+ $("btnProposeCats").disabled = false;
+ $("btnApplyRecat").disabled = false;
+ renderCatChips();
+ doSearch(true);
+ }
+}
+
+$("btnProposeCats").onclick = async ()=>{
+ $("btnProposeCats").disabled = true;
+ $("btnProposeCats").textContent = "Analyzing your library…";
+ const j = await api("/api/recategorize/propose", {model: "qwen3.5:9b"});
+ $("btnProposeCats").disabled = false;
+ $("btnProposeCats").textContent = "✨ Propose New Categories";
+ if(j.ok && j.categories){
+ $("recatProposalList").value = j.categories.join("\n");
+ $("recatProposalBox").style.display = "block";
+ showToast("Proposed " + j.categories.length + " categories — review before applying", "ok");
+ }
+};
+
+$("btnApplyRecat").onclick = async ()=>{
+ const cats = $("recatProposalList").value.split("\n").map(s=>s.trim()).filter(Boolean);
+ if(cats.length < 2){ showToast("Need at least a couple of categories", "error"); return; }
+ if(!confirm(`This will re-read the description of every tagged photo and, using a local text model, may change its category — rewriting the category keyword in that photo's metadata (descriptions stay untouched). Proceed with ${cats.length} categories?`)) return;
+ $("btnApplyRecat").disabled = true;
+ $("btnProposeCats").disabled = true;
+ const j = await api("/api/recategorize/apply", {categories: cats, model: "qwen3.5:9b"});
+ if(j.ok){
+ showToast("Recategorize started — watch progress below", "ok");
+ applyRecatProgress({done:0,total:0,changed:0});
+ } else {
+ $("btnApplyRecat").disabled = false;
+ $("btnProposeCats").disabled = false;
+ }
+};
+
/* ---- audit history dashboard ---- */
async function loadAuditHistory(){
const j = await api("/api/audit/history");
@@ -2116,6 +2180,9 @@ function connect(){
if(m.data.finished) loadDupeGroups();
}
}
+ else if(m.type==="recat"){
+ applyRecatProgress(m.data);
+ }
};
es.onerror = ()=>{ es.close(); setTimeout(connect, 2000); };
}
@@ -2132,6 +2199,16 @@ function connect(){
setBadge(st.status||"idle");
if(st.total>0) scanned=true, setBadge(st.status||"idle");
+ if(st.recatProposal && st.recatProposal.length){
+ $("recatProposalList").value = st.recatProposal.join("\n");
+ $("recatProposalBox").style.display = "block";
+ }
+ if(st.recatStatus === "running"){
+ $("btnProposeCats").disabled = true;
+ $("btnApplyRecat").disabled = true;
+ applyRecatProgress(st.recatProgress || {done:0,total:0,changed:0});
+ }
+
const mj = await api("/api/models");
const sel=$("model");
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
diff --git a/server.py b/server.py
index d19758a..8103fcd 100644
--- a/server.py
+++ b/server.py
@@ -13,6 +13,7 @@ import hashlib
import json
import os
import queue
+import random
import re
import shutil
import subprocess
@@ -195,6 +196,9 @@ class State:
self.skipped_sidecars = 0
self.dedupe_status = "idle" # idle | running
self.dedupe_progress = {"done": 0, "total": 0}
+ self.recat_status = "idle" # idle | proposing | running
+ self.recat_progress = {"done": 0, "total": 0, "changed": 0}
+ self.recat_proposal = [] # last LLM-proposed category list, for the UI to recall
self.already_done = 0
self.processed_session = 0
self.failed_session = 0
@@ -638,6 +642,133 @@ def build_dedupe_index():
with S.lock:
S.dedupe_status = "idle"
+def sample_descriptions_for_proposal(max_total=350):
+ """Stratified sample across every current category, weighted so a
+ dumping-ground category (e.g. one holding 35% of the library) is
+ oversampled relative to its share — the model needs to actually SEE
+ what's clogging it up, not just be told the percentage."""
+ records = load_search_index()
+ by_cat = {}
+ for r in records:
+ by_cat.setdefault(r.get("category") or "Uncategorized", []).append(r)
+ total = len(records)
+ cat_sizes = {c: len(rs) for c, rs in by_cat.items()}
+ samples = []
+ for cat, rs in by_cat.items():
+ share = max(10, min(120, int(max_total * (len(rs) / total) ** 0.5))) if total else 0
+ chosen = random.sample(rs, min(share, len(rs)))
+ for r in chosen:
+ samples.append({"category": cat, "desc": (r.get("desc") or "")[:150]})
+ random.shuffle(samples)
+ return samples[:max_total], cat_sizes, total
+
+def build_category_proposal_prompt(samples, cat_sizes, total):
+ dist = "\n".join(
+ f"- {c}: {n} photos ({100*n/total:.0f}%)"
+ for c, n in sorted(cat_sizes.items(), key=lambda x: -x[1])
+ )
+ sample_lines = "\n".join(f'[{s["category"]}] {s["desc"]}' for s in samples)
+ return (
+ "You are refining a personal photo library's category taxonomy.\n"
+ f"Current categories and their share of {total} total photos:\n{dist}\n\n"
+ "Below is a random sample of actual photo descriptions, each labeled with its "
+ "current category:\n" + sample_lines + "\n\n"
+ "The categories above are too coarse — the largest ones are dumping grounds "
+ "covering very different kinds of photos. Propose a NEW list of 15 to 25 "
+ "categories that fits this actual library well, so no single category dominates. "
+ "Keep any current category that's already specific and reasonably sized; split "
+ "the overloaded/vague ones based on what the sample actually shows (e.g. separate "
+ "electronics from plants from receipts from tools — whatever the real content is, "
+ "don't guess generically). Every name should be short (2-4 words) and mutually "
+ "exclusive from the others.\n"
+ 'Answer as JSON: {"categories": ["...", "...", ...]}'
+ )
+
+CATEGORY_PROPOSAL_SCHEMA = {
+ "type": "object",
+ "properties": {"categories": {"type": "array", "items": {"type": "string"}}},
+ "required": ["categories"],
+}
+
+def build_reclassify_prompt(categories, batch):
+ cats = "\n".join(f"- {c}" for c in categories)
+ items = "\n".join(f"{i}. {b['desc'][:200]}" for i, b in enumerate(batch))
+ return (
+ "Classify each numbered photo description below into EXACTLY ONE of these "
+ f"categories:\n{cats}\n\nDescriptions:\n{items}\n\n"
+ "Answer as JSON: {\"categories\": [\"\", \"\", ...]} "
+ "— one entry per description, in the same order, same count."
+ )
+
+def reclassify_batch_schema(n):
+ return {
+ "type": "object",
+ "properties": {"categories": {"type": "array", "items": {"type": "string"}, "minItems": n, "maxItems": n}},
+ "required": ["categories"],
+ }
+
+def recategorize_apply_worker(new_categories, model):
+ with S.lock:
+ if S.recat_status != "idle":
+ return
+ S.recat_status = "running"
+ keep_backup = bool(S.settings.get("keepBackup", False))
+ preserve_date = bool(S.settings.get("preserveDate", True))
+ organize = bool(S.settings.get("organize", False))
+ try:
+ save_categories(new_categories)
+ load_journal()
+ push_stats()
+ records = load_search_index()
+ total = len(records)
+ S.recat_progress = {"done": 0, "total": total, "changed": 0}
+ log("info", f"smart recategorize: reclassifying {total} photos into "
+ f"{len(new_categories)} categories …")
+ think = False if model_thinks(model) else None
+ batch_size = 25
+ changed = 0
+ done = 0
+ for i in range(0, total, batch_size):
+ batch = records[i:i + batch_size]
+ prompt = build_reclassify_prompt(new_categories, batch)
+ try:
+ resp = ollama_generate_text(model, prompt, {"temperature": 0, "num_predict": 800},
+ "10m", think, reclassify_batch_schema(len(batch)))
+ parsed = salvage_json(resp.get("response", ""))
+ cats = parsed.get("categories") or []
+ except Exception as e:
+ log("error", f"recategorize batch {i}-{i+len(batch)} failed: {e}")
+ cats = []
+ for j, rec in enumerate(batch):
+ done += 1
+ new_cat = cats[j] if j < len(cats) else None
+ if new_cat not in new_categories:
+ new_cat = None
+ old_cat = rec.get("category")
+ if new_cat and new_cat != old_cat:
+ try:
+ write_metadata(rec["path"], rec.get("desc", ""), new_cat,
+ keep_backup, preserve_date, old_cat)
+ if organize:
+ organize_alias(find_owning_folder(rec["path"]), rec["path"], new_cat)
+ journal_write({"path": rec["path"], "desc": rec.get("desc", ""),
+ "category": new_cat, "model": "recategorize",
+ "route": rec.get("route"), "sec": 0.0, "ts": time.time()})
+ changed += 1
+ except Exception as e:
+ log("error", f"recategorize write failed for {rec['path']}: {e}")
+ S.recat_progress = {"done": done, "total": total, "changed": changed}
+ broadcast("recat", S.recat_progress)
+ push_stats()
+ compact_journal()
+ load_journal()
+ push_stats()
+ log("ok", f"smart recategorize complete: {changed} of {total} photos moved to a new category")
+ finally:
+ with S.lock:
+ S.recat_status = "idle"
+ broadcast("recat", {**S.recat_progress, "finished": True})
+
def find_duplicate_groups():
"""Groups of exact perceptual-hash matches and size/filename matches."""
records = {r["path"]: r for r in load_search_index()}
@@ -773,6 +904,27 @@ def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read())
+def ollama_generate_text(model, prompt, opts, keep_alive, think=None, schema=SCHEMA):
+ """Text-only generation (no image) — used for classifying/organizing
+ against descriptions that were already produced by a vision pass, so
+ re-analyzing the whole library never needs the vision model again."""
+ body = {
+ "model": model,
+ "prompt": prompt,
+ "stream": False,
+ "options": opts,
+ "keep_alive": keep_alive,
+ }
+ if schema is not None:
+ body["format"] = schema
+ if think is not None:
+ body["think"] = think
+ req = urllib.request.Request(
+ OLLAMA + "/api/generate", data=json.dumps(body).encode(),
+ headers={"Content-Type": "application/json"})
+ with urllib.request.urlopen(req, timeout=600) as r:
+ return json.loads(r.read())
+
# ---------------------------------------------------------------- pipeline
def scan_folder(folder, recursive=True):
@@ -1022,7 +1174,14 @@ def build_prompt(length_key, mode="photo"):
'Answer as JSON: {"description": "...", "category": "..."}'
)
-def write_metadata(path, desc, category, keep_backup, preserve_date):
+def write_metadata(path, desc, category, keep_backup, preserve_date, old_category=None):
+ """Idempotent regardless of how many times a photo gets (re)tagged:
+ category and the photon-tagged marker are list-type XMP/IPTC tags, so a
+ bare += on every write would silently accumulate duplicates (a photo
+ retagged 3 times would carry 3 copies of "photon-tagged" and every
+ category it ever had). Always strip the exact values being (re)written
+ first, then add them back exactly once. old_category additionally
+ strips a genuinely different previous category."""
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"]
if preserve_date:
@@ -1032,13 +1191,19 @@ def write_metadata(path, desc, category, keep_backup, preserve_date):
if is_video:
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
# tag names so it resolves to QuickTime/Keys groups automatically.
+ if old_category and old_category != category:
+ args.append(f"-Keywords-={old_category}")
args += [
+ f"-Keywords-={category}", f"-Keywords-=photon-tagged",
f"-Description={desc}",
f"-Keywords+={category}",
"-Keywords+=photon-tagged",
]
else:
+ if old_category and old_category != category:
+ args += [f"-XMP-dc:Subject-={old_category}", f"-IPTC:Keywords-={old_category}"]
args += [
+ f"-XMP-dc:Subject-={category}", f"-IPTC:Keywords-={category}", "-XMP-dc:Subject-=photon-tagged",
f"-EXIF:ImageDescription={desc}",
f"-IPTC:Caption-Abstract={desc}",
f"-XMP-dc:Description={desc}",
@@ -1160,6 +1325,10 @@ def process_loop(settings):
if think_map[m] is False:
log("info", f"{m} is a thinking model — thinking disabled for speed")
+ # built once so a retag (skipDone=false on an already-tagged photo) can
+ # strip its old category keyword instead of accumulating it
+ prev_categories = {p: rec.get("category") for p, rec in read_journal_deduped().items()}
+
tmpdir = tempfile.mkdtemp(prefix="photon_")
log("info", f"engine online :: {'router mode' if router else 'model=' + model} "
f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}")
@@ -1332,11 +1501,12 @@ def process_loop(settings):
continue
try:
+ old_cat = prev_categories.get(path)
if not dry:
if integrity:
- verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date)
+ verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date, old_cat)
else:
- write_metadata(path, desc, category, keep_backup, preserve_date)
+ write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
@@ -1468,7 +1638,9 @@ class Handler(BaseHTTPRequestHandler):
"folder": S.folders[0] if S.folders else "",
"log": S.log_ring[-200:],
"categories": CATEGORIES, "current": S.current,
- "settings": S.settings, "failuresCount": len(S.failed_paths)})
+ "settings": S.settings, "failuresCount": len(S.failed_paths),
+ "recatStatus": S.recat_status, "recatProgress": S.recat_progress,
+ "recatProposal": S.recat_proposal})
self._json(payload)
elif self.path.startswith("/api/preview"):
with S.lock:
@@ -2191,7 +2363,8 @@ class Handler(BaseHTTPRequestHandler):
keep_backup = bool(S.settings.get("keepBackup", False))
preserve_date = bool(S.settings.get("preserveDate", True))
organize = bool(S.settings.get("organize", False))
- write_metadata(path, desc, category, keep_backup, preserve_date)
+ old_cat = read_journal_deduped().get(path, {}).get("category")
+ write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
existing_recs = []
@@ -2298,7 +2471,8 @@ class Handler(BaseHTTPRequestHandler):
desc = f"{desc} | text: {words[:300]}"
except Exception as e:
pass
- write_metadata(path, desc, category, keep_backup, preserve_date)
+ old_cat = read_journal_deduped().get(path, {}).get("category")
+ write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
existing_recs = []
@@ -2402,6 +2576,47 @@ class Handler(BaseHTTPRequestHandler):
threading.Thread(target=build_dedupe_index, daemon=True).start()
self._json({"ok": True})
+ elif self.path == "/api/recategorize/propose":
+ model = body.get("model") or "qwen3.5:9b"
+ with S.lock:
+ if S.recat_status != "idle":
+ self._json({"error": "a recategorize job is already running"}, 400); return
+ S.recat_status = "proposing"
+ try:
+ samples, cat_sizes, total = sample_descriptions_for_proposal()
+ if total == 0:
+ self._json({"error": "no tagged photos yet to analyze"}, 400); return
+ prompt = build_category_proposal_prompt(samples, cat_sizes, total)
+ think = False if model_thinks(model) else None
+ log("info", f"smart recategorize: asking {model} to propose new categories "
+ f"from {len(samples)} sampled descriptions ({total} photos total) …")
+ resp = ollama_generate_text(model, prompt, {"temperature": 0.3, "num_predict": 900},
+ "5m", think, CATEGORY_PROPOSAL_SCHEMA)
+ parsed = salvage_json(resp.get("response", ""))
+ proposed = [c.strip() for c in (parsed.get("categories") or []) if c and c.strip()]
+ if not proposed:
+ self._json({"error": "model returned no categories"}, 500); return
+ with S.lock:
+ S.recat_proposal = proposed
+ log("ok", f"smart recategorize: proposed {len(proposed)} categories")
+ self._json({"ok": True, "categories": proposed, "currentDistribution": cat_sizes})
+ except Exception as e:
+ self._json({"error": f"proposal failed: {e}"}, 500)
+ finally:
+ with S.lock:
+ S.recat_status = "idle"
+
+ elif self.path == "/api/recategorize/apply":
+ categories = body.get("categories")
+ model = body.get("model") or "qwen3.5:9b"
+ if not isinstance(categories, list) or not categories:
+ self._json({"error": "categories list is required"}, 400); return
+ with S.lock:
+ if S.recat_status != "idle":
+ self._json({"error": "a recategorize job is already running"}, 400); return
+ threading.Thread(target=recategorize_apply_worker, args=(categories, model), daemon=True).start()
+ self._json({"ok": True})
+
elif self.path == "/api/bulk_recategorize":
paths = body.get("paths") or []
category = body.get("category")
@@ -2415,8 +2630,9 @@ class Handler(BaseHTTPRequestHandler):
for p in paths:
rec = records.get(p)
desc = rec.get("desc", "") if rec else ""
+ old_cat = rec.get("category") if rec else None
try:
- write_metadata(p, desc, category, keep_backup, preserve_date)
+ write_metadata(p, desc, category, keep_backup, preserve_date, old_cat)
ok += 1
except Exception as e:
failed.append({"path": p, "error": str(e)})