Add smart recategorize (auto-discovered categories) and fix stale category-keyword accumulation

- New "Smart Recategorize" panel: analyzes existing journal descriptions
  (no vision re-run, no images touched) to propose a richer category list
  when one bucket dominates the library, then reclassifies every tagged
  photo into it using a fast local text-only model pass in batches of 25.
  Verified end-to-end on the real library: correctly split "Objects &
  Stuff" (35% of 15k photos) into specific categories like Electronics &
  Hardware, Cannabis Cultivation, Minerals & Crystals, Tools & Workshop
  based on what's actually in the descriptions.
- New ollama_generate_text() for text-only local model calls (the
  existing ollama_generate always sent an "images" key, unsuitable for
  pure text classification).
- FIX: write_metadata only ever appended category/photon-tagged keywords
  (+=) to the list-type XMP/IPTC tags, so retagging a photo left every
  category it ever had stacked in its metadata, and even rewriting with
  the same category duplicated "photon-tagged" each time. Found via a
  live test while building this feature (which recategorizes at library
  scale). Now strips old values before adding, making every write
  idempotent regardless of how many times a photo is retagged. Wired
  through all four write sites (pipeline, write_single, redo_single,
  bulk_recategorize).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-07-29 17:16:00 -07:00
parent 501bfe2916
commit 3f67efb379
2 changed files with 300 additions and 7 deletions

View File

@@ -720,6 +720,25 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
</div> </div>
</div> </div>
<!-- Smart Recategorize -->
<div class="panel">
<h2>🧠 <b>smart recategorize</b> — auto-discover categories from your library</h2>
<div class="set">
<div class="hint">Analyzes descriptions already sitting in your journal (no images re-scanned, no vision model) to propose a richer category list when one bucket is swallowing the library, then reclassifies every tagged photo into it using a fast local text model.</div>
<button id="btnProposeCats" class="btn-primary" style="width:100%; margin-top:10px">✨ Propose New Categories</button>
<div id="recatProposalBox" style="display:none; margin-top:12px">
<label>proposed categories — edit before applying (one per line)</label>
<textarea id="recatProposalList" rows="10" style="width:100%; font-family:var(--font-mono); font-size:11px; background:var(--panel2); border:1px solid var(--line); color:var(--txt); border-radius:4px; padding:8px" spellcheck="false"></textarea>
<button id="btnApplyRecat" class="btn-danger" style="width:100%; margin-top:10px">⚠ Apply — Reclassify Every Photo</button>
<div class="hint">This rewrites the category keyword in every tagged photo's metadata (descriptions are untouched). Reversible per-photo the same way any recategorize is — nothing about the image itself changes.</div>
</div>
<div id="recatProgressBox" style="display:none; margin-top:12px">
<div class="pbar"><div class="fill" id="recatFill"></div><div class="glint"></div></div>
<div class="hint" id="recatStatusText">starting…</div>
</div>
</div>
</div>
<!-- Safety Panel --> <!-- Safety Panel -->
<div class="panel"> <div class="panel">
<h2>🛡 <b>trust & safety console</b></h2> <h2>🛡 <b>trust & safety console</b></h2>
@@ -1985,6 +2004,51 @@ $("btnScanDupes").onclick = async ()=>{
} }
}; };
/* ---- smart recategorize ---- */
function applyRecatProgress(p){
$("recatProgressBox").style.display = "block";
const pct = p.total ? Math.round(100 * p.done / p.total) : 0;
$("recatFill").style.width = pct + "%";
$("recatStatusText").textContent = p.finished
? `Done — ${p.changed} of ${p.total} photos moved to a new category.`
: `Reclassifying… ${p.done}/${p.total} (${p.changed} changed so far)`;
if(p.finished){
$("btnProposeCats").disabled = false;
$("btnApplyRecat").disabled = false;
renderCatChips();
doSearch(true);
}
}
$("btnProposeCats").onclick = async ()=>{
$("btnProposeCats").disabled = true;
$("btnProposeCats").textContent = "Analyzing your library…";
const j = await api("/api/recategorize/propose", {model: "qwen3.5:9b"});
$("btnProposeCats").disabled = false;
$("btnProposeCats").textContent = "✨ Propose New Categories";
if(j.ok && j.categories){
$("recatProposalList").value = j.categories.join("\n");
$("recatProposalBox").style.display = "block";
showToast("Proposed " + j.categories.length + " categories — review before applying", "ok");
}
};
$("btnApplyRecat").onclick = async ()=>{
const cats = $("recatProposalList").value.split("\n").map(s=>s.trim()).filter(Boolean);
if(cats.length < 2){ showToast("Need at least a couple of categories", "error"); return; }
if(!confirm(`This will re-read the description of every tagged photo and, using a local text model, may change its category — rewriting the category keyword in that photo's metadata (descriptions stay untouched). Proceed with ${cats.length} categories?`)) return;
$("btnApplyRecat").disabled = true;
$("btnProposeCats").disabled = true;
const j = await api("/api/recategorize/apply", {categories: cats, model: "qwen3.5:9b"});
if(j.ok){
showToast("Recategorize started — watch progress below", "ok");
applyRecatProgress({done:0,total:0,changed:0});
} else {
$("btnApplyRecat").disabled = false;
$("btnProposeCats").disabled = false;
}
};
/* ---- audit history dashboard ---- */ /* ---- audit history dashboard ---- */
async function loadAuditHistory(){ async function loadAuditHistory(){
const j = await api("/api/audit/history"); const j = await api("/api/audit/history");
@@ -2116,6 +2180,9 @@ function connect(){
if(m.data.finished) loadDupeGroups(); if(m.data.finished) loadDupeGroups();
} }
} }
else if(m.type==="recat"){
applyRecatProgress(m.data);
}
}; };
es.onerror = ()=>{ es.close(); setTimeout(connect, 2000); }; es.onerror = ()=>{ es.close(); setTimeout(connect, 2000); };
} }
@@ -2132,6 +2199,16 @@ function connect(){
setBadge(st.status||"idle"); setBadge(st.status||"idle");
if(st.total>0) scanned=true, setBadge(st.status||"idle"); if(st.total>0) scanned=true, setBadge(st.status||"idle");
if(st.recatProposal && st.recatProposal.length){
$("recatProposalList").value = st.recatProposal.join("\n");
$("recatProposalBox").style.display = "block";
}
if(st.recatStatus === "running"){
$("btnProposeCats").disabled = true;
$("btnApplyRecat").disabled = true;
applyRecatProgress(st.recatProgress || {done:0,total:0,changed:0});
}
const mj = await api("/api/models"); const mj = await api("/api/models");
const sel=$("model"); const sel=$("model");
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken

230
server.py
View File

@@ -13,6 +13,7 @@ import hashlib
import json import json
import os import os
import queue import queue
import random
import re import re
import shutil import shutil
import subprocess import subprocess
@@ -195,6 +196,9 @@ class State:
self.skipped_sidecars = 0 self.skipped_sidecars = 0
self.dedupe_status = "idle" # idle | running self.dedupe_status = "idle" # idle | running
self.dedupe_progress = {"done": 0, "total": 0} self.dedupe_progress = {"done": 0, "total": 0}
self.recat_status = "idle" # idle | proposing | running
self.recat_progress = {"done": 0, "total": 0, "changed": 0}
self.recat_proposal = [] # last LLM-proposed category list, for the UI to recall
self.already_done = 0 self.already_done = 0
self.processed_session = 0 self.processed_session = 0
self.failed_session = 0 self.failed_session = 0
@@ -638,6 +642,133 @@ def build_dedupe_index():
with S.lock: with S.lock:
S.dedupe_status = "idle" S.dedupe_status = "idle"
def sample_descriptions_for_proposal(max_total=350):
"""Stratified sample across every current category, weighted so a
dumping-ground category (e.g. one holding 35% of the library) is
oversampled relative to its share — the model needs to actually SEE
what's clogging it up, not just be told the percentage."""
records = load_search_index()
by_cat = {}
for r in records:
by_cat.setdefault(r.get("category") or "Uncategorized", []).append(r)
total = len(records)
cat_sizes = {c: len(rs) for c, rs in by_cat.items()}
samples = []
for cat, rs in by_cat.items():
share = max(10, min(120, int(max_total * (len(rs) / total) ** 0.5))) if total else 0
chosen = random.sample(rs, min(share, len(rs)))
for r in chosen:
samples.append({"category": cat, "desc": (r.get("desc") or "")[:150]})
random.shuffle(samples)
return samples[:max_total], cat_sizes, total
def build_category_proposal_prompt(samples, cat_sizes, total):
dist = "\n".join(
f"- {c}: {n} photos ({100*n/total:.0f}%)"
for c, n in sorted(cat_sizes.items(), key=lambda x: -x[1])
)
sample_lines = "\n".join(f'[{s["category"]}] {s["desc"]}' for s in samples)
return (
"You are refining a personal photo library's category taxonomy.\n"
f"Current categories and their share of {total} total photos:\n{dist}\n\n"
"Below is a random sample of actual photo descriptions, each labeled with its "
"current category:\n" + sample_lines + "\n\n"
"The categories above are too coarse — the largest ones are dumping grounds "
"covering very different kinds of photos. Propose a NEW list of 15 to 25 "
"categories that fits this actual library well, so no single category dominates. "
"Keep any current category that's already specific and reasonably sized; split "
"the overloaded/vague ones based on what the sample actually shows (e.g. separate "
"electronics from plants from receipts from tools — whatever the real content is, "
"don't guess generically). Every name should be short (2-4 words) and mutually "
"exclusive from the others.\n"
'Answer as JSON: {"categories": ["...", "...", ...]}'
)
CATEGORY_PROPOSAL_SCHEMA = {
"type": "object",
"properties": {"categories": {"type": "array", "items": {"type": "string"}}},
"required": ["categories"],
}
def build_reclassify_prompt(categories, batch):
cats = "\n".join(f"- {c}" for c in categories)
items = "\n".join(f"{i}. {b['desc'][:200]}" for i, b in enumerate(batch))
return (
"Classify each numbered photo description below into EXACTLY ONE of these "
f"categories:\n{cats}\n\nDescriptions:\n{items}\n\n"
"Answer as JSON: {\"categories\": [\"<category for item 0>\", \"<category for item 1>\", ...]} "
"— one entry per description, in the same order, same count."
)
def reclassify_batch_schema(n):
return {
"type": "object",
"properties": {"categories": {"type": "array", "items": {"type": "string"}, "minItems": n, "maxItems": n}},
"required": ["categories"],
}
def recategorize_apply_worker(new_categories, model):
with S.lock:
if S.recat_status != "idle":
return
S.recat_status = "running"
keep_backup = bool(S.settings.get("keepBackup", False))
preserve_date = bool(S.settings.get("preserveDate", True))
organize = bool(S.settings.get("organize", False))
try:
save_categories(new_categories)
load_journal()
push_stats()
records = load_search_index()
total = len(records)
S.recat_progress = {"done": 0, "total": total, "changed": 0}
log("info", f"smart recategorize: reclassifying {total} photos into "
f"{len(new_categories)} categories …")
think = False if model_thinks(model) else None
batch_size = 25
changed = 0
done = 0
for i in range(0, total, batch_size):
batch = records[i:i + batch_size]
prompt = build_reclassify_prompt(new_categories, batch)
try:
resp = ollama_generate_text(model, prompt, {"temperature": 0, "num_predict": 800},
"10m", think, reclassify_batch_schema(len(batch)))
parsed = salvage_json(resp.get("response", ""))
cats = parsed.get("categories") or []
except Exception as e:
log("error", f"recategorize batch {i}-{i+len(batch)} failed: {e}")
cats = []
for j, rec in enumerate(batch):
done += 1
new_cat = cats[j] if j < len(cats) else None
if new_cat not in new_categories:
new_cat = None
old_cat = rec.get("category")
if new_cat and new_cat != old_cat:
try:
write_metadata(rec["path"], rec.get("desc", ""), new_cat,
keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(rec["path"]), rec["path"], new_cat)
journal_write({"path": rec["path"], "desc": rec.get("desc", ""),
"category": new_cat, "model": "recategorize",
"route": rec.get("route"), "sec": 0.0, "ts": time.time()})
changed += 1
except Exception as e:
log("error", f"recategorize write failed for {rec['path']}: {e}")
S.recat_progress = {"done": done, "total": total, "changed": changed}
broadcast("recat", S.recat_progress)
push_stats()
compact_journal()
load_journal()
push_stats()
log("ok", f"smart recategorize complete: {changed} of {total} photos moved to a new category")
finally:
with S.lock:
S.recat_status = "idle"
broadcast("recat", {**S.recat_progress, "finished": True})
def find_duplicate_groups(): def find_duplicate_groups():
"""Groups of exact perceptual-hash matches and size/filename matches.""" """Groups of exact perceptual-hash matches and size/filename matches."""
records = {r["path"]: r for r in load_search_index()} records = {r["path"]: r for r in load_search_index()}
@@ -773,6 +904,27 @@ def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema
with urllib.request.urlopen(req, timeout=600) as r: with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read()) return json.loads(r.read())
def ollama_generate_text(model, prompt, opts, keep_alive, think=None, schema=SCHEMA):
"""Text-only generation (no image) — used for classifying/organizing
against descriptions that were already produced by a vision pass, so
re-analyzing the whole library never needs the vision model again."""
body = {
"model": model,
"prompt": prompt,
"stream": False,
"options": opts,
"keep_alive": keep_alive,
}
if schema is not None:
body["format"] = schema
if think is not None:
body["think"] = think
req = urllib.request.Request(
OLLAMA + "/api/generate", data=json.dumps(body).encode(),
headers={"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read())
# ---------------------------------------------------------------- pipeline # ---------------------------------------------------------------- pipeline
def scan_folder(folder, recursive=True): def scan_folder(folder, recursive=True):
@@ -1022,7 +1174,14 @@ def build_prompt(length_key, mode="photo"):
'Answer as JSON: {"description": "...", "category": "..."}' 'Answer as JSON: {"description": "...", "category": "..."}'
) )
def write_metadata(path, desc, category, keep_backup, preserve_date): def write_metadata(path, desc, category, keep_backup, preserve_date, old_category=None):
"""Idempotent regardless of how many times a photo gets (re)tagged:
category and the photon-tagged marker are list-type XMP/IPTC tags, so a
bare += on every write would silently accumulate duplicates (a photo
retagged 3 times would carry 3 copies of "photon-tagged" and every
category it ever had). Always strip the exact values being (re)written
first, then add them back exactly once. old_category additionally
strips a genuinely different previous category."""
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"] args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"]
if preserve_date: if preserve_date:
@@ -1032,13 +1191,19 @@ def write_metadata(path, desc, category, keep_backup, preserve_date):
if is_video: if is_video:
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic # mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
# tag names so it resolves to QuickTime/Keys groups automatically. # tag names so it resolves to QuickTime/Keys groups automatically.
if old_category and old_category != category:
args.append(f"-Keywords-={old_category}")
args += [ args += [
f"-Keywords-={category}", f"-Keywords-=photon-tagged",
f"-Description={desc}", f"-Description={desc}",
f"-Keywords+={category}", f"-Keywords+={category}",
"-Keywords+=photon-tagged", "-Keywords+=photon-tagged",
] ]
else: else:
if old_category and old_category != category:
args += [f"-XMP-dc:Subject-={old_category}", f"-IPTC:Keywords-={old_category}"]
args += [ args += [
f"-XMP-dc:Subject-={category}", f"-IPTC:Keywords-={category}", "-XMP-dc:Subject-=photon-tagged",
f"-EXIF:ImageDescription={desc}", f"-EXIF:ImageDescription={desc}",
f"-IPTC:Caption-Abstract={desc}", f"-IPTC:Caption-Abstract={desc}",
f"-XMP-dc:Description={desc}", f"-XMP-dc:Description={desc}",
@@ -1160,6 +1325,10 @@ def process_loop(settings):
if think_map[m] is False: if think_map[m] is False:
log("info", f"{m} is a thinking model — thinking disabled for speed") log("info", f"{m} is a thinking model — thinking disabled for speed")
# built once so a retag (skipDone=false on an already-tagged photo) can
# strip its old category keyword instead of accumulating it
prev_categories = {p: rec.get("category") for p, rec in read_journal_deduped().items()}
tmpdir = tempfile.mkdtemp(prefix="photon_") tmpdir = tempfile.mkdtemp(prefix="photon_")
log("info", f"engine online :: {'router mode' if router else 'model=' + model} " log("info", f"engine online :: {'router mode' if router else 'model=' + model} "
f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}") f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}")
@@ -1332,11 +1501,12 @@ def process_loop(settings):
continue continue
try: try:
old_cat = prev_categories.get(path)
if not dry: if not dry:
if integrity: if integrity:
verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date) verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date, old_cat)
else: else:
write_metadata(path, desc, category, keep_backup, preserve_date) write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize: if organize:
organize_alias(find_owning_folder(path), path, category) organize_alias(find_owning_folder(path), path, category)
@@ -1468,7 +1638,9 @@ class Handler(BaseHTTPRequestHandler):
"folder": S.folders[0] if S.folders else "", "folder": S.folders[0] if S.folders else "",
"log": S.log_ring[-200:], "log": S.log_ring[-200:],
"categories": CATEGORIES, "current": S.current, "categories": CATEGORIES, "current": S.current,
"settings": S.settings, "failuresCount": len(S.failed_paths)}) "settings": S.settings, "failuresCount": len(S.failed_paths),
"recatStatus": S.recat_status, "recatProgress": S.recat_progress,
"recatProposal": S.recat_proposal})
self._json(payload) self._json(payload)
elif self.path.startswith("/api/preview"): elif self.path.startswith("/api/preview"):
with S.lock: with S.lock:
@@ -2191,7 +2363,8 @@ class Handler(BaseHTTPRequestHandler):
keep_backup = bool(S.settings.get("keepBackup", False)) keep_backup = bool(S.settings.get("keepBackup", False))
preserve_date = bool(S.settings.get("preserveDate", True)) preserve_date = bool(S.settings.get("preserveDate", True))
organize = bool(S.settings.get("organize", False)) organize = bool(S.settings.get("organize", False))
write_metadata(path, desc, category, keep_backup, preserve_date) old_cat = read_journal_deduped().get(path, {}).get("category")
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize: if organize:
organize_alias(find_owning_folder(path), path, category) organize_alias(find_owning_folder(path), path, category)
existing_recs = [] existing_recs = []
@@ -2298,7 +2471,8 @@ class Handler(BaseHTTPRequestHandler):
desc = f"{desc} | text: {words[:300]}" desc = f"{desc} | text: {words[:300]}"
except Exception as e: except Exception as e:
pass pass
write_metadata(path, desc, category, keep_backup, preserve_date) old_cat = read_journal_deduped().get(path, {}).get("category")
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize: if organize:
organize_alias(find_owning_folder(path), path, category) organize_alias(find_owning_folder(path), path, category)
existing_recs = [] existing_recs = []
@@ -2402,6 +2576,47 @@ class Handler(BaseHTTPRequestHandler):
threading.Thread(target=build_dedupe_index, daemon=True).start() threading.Thread(target=build_dedupe_index, daemon=True).start()
self._json({"ok": True}) self._json({"ok": True})
elif self.path == "/api/recategorize/propose":
model = body.get("model") or "qwen3.5:9b"
with S.lock:
if S.recat_status != "idle":
self._json({"error": "a recategorize job is already running"}, 400); return
S.recat_status = "proposing"
try:
samples, cat_sizes, total = sample_descriptions_for_proposal()
if total == 0:
self._json({"error": "no tagged photos yet to analyze"}, 400); return
prompt = build_category_proposal_prompt(samples, cat_sizes, total)
think = False if model_thinks(model) else None
log("info", f"smart recategorize: asking {model} to propose new categories "
f"from {len(samples)} sampled descriptions ({total} photos total) …")
resp = ollama_generate_text(model, prompt, {"temperature": 0.3, "num_predict": 900},
"5m", think, CATEGORY_PROPOSAL_SCHEMA)
parsed = salvage_json(resp.get("response", ""))
proposed = [c.strip() for c in (parsed.get("categories") or []) if c and c.strip()]
if not proposed:
self._json({"error": "model returned no categories"}, 500); return
with S.lock:
S.recat_proposal = proposed
log("ok", f"smart recategorize: proposed {len(proposed)} categories")
self._json({"ok": True, "categories": proposed, "currentDistribution": cat_sizes})
except Exception as e:
self._json({"error": f"proposal failed: {e}"}, 500)
finally:
with S.lock:
S.recat_status = "idle"
elif self.path == "/api/recategorize/apply":
categories = body.get("categories")
model = body.get("model") or "qwen3.5:9b"
if not isinstance(categories, list) or not categories:
self._json({"error": "categories list is required"}, 400); return
with S.lock:
if S.recat_status != "idle":
self._json({"error": "a recategorize job is already running"}, 400); return
threading.Thread(target=recategorize_apply_worker, args=(categories, model), daemon=True).start()
self._json({"ok": True})
elif self.path == "/api/bulk_recategorize": elif self.path == "/api/bulk_recategorize":
paths = body.get("paths") or [] paths = body.get("paths") or []
category = body.get("category") category = body.get("category")
@@ -2415,8 +2630,9 @@ class Handler(BaseHTTPRequestHandler):
for p in paths: for p in paths:
rec = records.get(p) rec = records.get(p)
desc = rec.get("desc", "") if rec else "" desc = rec.get("desc", "") if rec else ""
old_cat = rec.get("category") if rec else None
try: try:
write_metadata(p, desc, category, keep_backup, preserve_date) write_metadata(p, desc, category, keep_backup, preserve_date, old_cat)
ok += 1 ok += 1
except Exception as e: except Exception as e:
failed.append({"path": p, "error": str(e)}) failed.append({"path": p, "error": str(e)})