Add smart recategorize (auto-discovered categories) and fix stale category-keyword accumulation

- New "Smart Recategorize" panel: analyzes existing journal descriptions
  (no vision re-run, no images touched) to propose a richer category list
  when one bucket dominates the library, then reclassifies every tagged
  photo into it using a fast local text-only model pass in batches of 25.
  Verified end-to-end on the real library: correctly split "Objects &
  Stuff" (35% of 15k photos) into specific categories like Electronics &
  Hardware, Cannabis Cultivation, Minerals & Crystals, Tools & Workshop
  based on what's actually in the descriptions.
- New ollama_generate_text() for text-only local model calls (the
  existing ollama_generate always sent an "images" key, unsuitable for
  pure text classification).
- FIX: write_metadata only ever appended category/photon-tagged keywords
  (+=) to the list-type XMP/IPTC tags, so retagging a photo left every
  category it ever had stacked in its metadata, and even rewriting with
  the same category duplicated "photon-tagged" each time. Found via a
  live test while building this feature (which recategorizes at library
  scale). Now strips old values before adding, making every write
  idempotent regardless of how many times a photo is retagged. Wired
  through all four write sites (pipeline, write_single, redo_single,
  bulk_recategorize).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-07-29 17:16:00 -07:00
parent 501bfe2916
commit 3f67efb379
2 changed files with 300 additions and 7 deletions

230
server.py
View File

@@ -13,6 +13,7 @@ import hashlib
import json
import os
import queue
import random
import re
import shutil
import subprocess
@@ -195,6 +196,9 @@ class State:
self.skipped_sidecars = 0
self.dedupe_status = "idle" # idle | running
self.dedupe_progress = {"done": 0, "total": 0}
self.recat_status = "idle" # idle | proposing | running
self.recat_progress = {"done": 0, "total": 0, "changed": 0}
self.recat_proposal = [] # last LLM-proposed category list, for the UI to recall
self.already_done = 0
self.processed_session = 0
self.failed_session = 0
@@ -638,6 +642,133 @@ def build_dedupe_index():
with S.lock:
S.dedupe_status = "idle"
def sample_descriptions_for_proposal(max_total=350):
"""Stratified sample across every current category, weighted so a
dumping-ground category (e.g. one holding 35% of the library) is
oversampled relative to its share — the model needs to actually SEE
what's clogging it up, not just be told the percentage."""
records = load_search_index()
by_cat = {}
for r in records:
by_cat.setdefault(r.get("category") or "Uncategorized", []).append(r)
total = len(records)
cat_sizes = {c: len(rs) for c, rs in by_cat.items()}
samples = []
for cat, rs in by_cat.items():
share = max(10, min(120, int(max_total * (len(rs) / total) ** 0.5))) if total else 0
chosen = random.sample(rs, min(share, len(rs)))
for r in chosen:
samples.append({"category": cat, "desc": (r.get("desc") or "")[:150]})
random.shuffle(samples)
return samples[:max_total], cat_sizes, total
def build_category_proposal_prompt(samples, cat_sizes, total):
dist = "\n".join(
f"- {c}: {n} photos ({100*n/total:.0f}%)"
for c, n in sorted(cat_sizes.items(), key=lambda x: -x[1])
)
sample_lines = "\n".join(f'[{s["category"]}] {s["desc"]}' for s in samples)
return (
"You are refining a personal photo library's category taxonomy.\n"
f"Current categories and their share of {total} total photos:\n{dist}\n\n"
"Below is a random sample of actual photo descriptions, each labeled with its "
"current category:\n" + sample_lines + "\n\n"
"The categories above are too coarse — the largest ones are dumping grounds "
"covering very different kinds of photos. Propose a NEW list of 15 to 25 "
"categories that fits this actual library well, so no single category dominates. "
"Keep any current category that's already specific and reasonably sized; split "
"the overloaded/vague ones based on what the sample actually shows (e.g. separate "
"electronics from plants from receipts from tools — whatever the real content is, "
"don't guess generically). Every name should be short (2-4 words) and mutually "
"exclusive from the others.\n"
'Answer as JSON: {"categories": ["...", "...", ...]}'
)
CATEGORY_PROPOSAL_SCHEMA = {
"type": "object",
"properties": {"categories": {"type": "array", "items": {"type": "string"}}},
"required": ["categories"],
}
def build_reclassify_prompt(categories, batch):
cats = "\n".join(f"- {c}" for c in categories)
items = "\n".join(f"{i}. {b['desc'][:200]}" for i, b in enumerate(batch))
return (
"Classify each numbered photo description below into EXACTLY ONE of these "
f"categories:\n{cats}\n\nDescriptions:\n{items}\n\n"
"Answer as JSON: {\"categories\": [\"<category for item 0>\", \"<category for item 1>\", ...]} "
"— one entry per description, in the same order, same count."
)
def reclassify_batch_schema(n):
return {
"type": "object",
"properties": {"categories": {"type": "array", "items": {"type": "string"}, "minItems": n, "maxItems": n}},
"required": ["categories"],
}
def recategorize_apply_worker(new_categories, model):
with S.lock:
if S.recat_status != "idle":
return
S.recat_status = "running"
keep_backup = bool(S.settings.get("keepBackup", False))
preserve_date = bool(S.settings.get("preserveDate", True))
organize = bool(S.settings.get("organize", False))
try:
save_categories(new_categories)
load_journal()
push_stats()
records = load_search_index()
total = len(records)
S.recat_progress = {"done": 0, "total": total, "changed": 0}
log("info", f"smart recategorize: reclassifying {total} photos into "
f"{len(new_categories)} categories …")
think = False if model_thinks(model) else None
batch_size = 25
changed = 0
done = 0
for i in range(0, total, batch_size):
batch = records[i:i + batch_size]
prompt = build_reclassify_prompt(new_categories, batch)
try:
resp = ollama_generate_text(model, prompt, {"temperature": 0, "num_predict": 800},
"10m", think, reclassify_batch_schema(len(batch)))
parsed = salvage_json(resp.get("response", ""))
cats = parsed.get("categories") or []
except Exception as e:
log("error", f"recategorize batch {i}-{i+len(batch)} failed: {e}")
cats = []
for j, rec in enumerate(batch):
done += 1
new_cat = cats[j] if j < len(cats) else None
if new_cat not in new_categories:
new_cat = None
old_cat = rec.get("category")
if new_cat and new_cat != old_cat:
try:
write_metadata(rec["path"], rec.get("desc", ""), new_cat,
keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(rec["path"]), rec["path"], new_cat)
journal_write({"path": rec["path"], "desc": rec.get("desc", ""),
"category": new_cat, "model": "recategorize",
"route": rec.get("route"), "sec": 0.0, "ts": time.time()})
changed += 1
except Exception as e:
log("error", f"recategorize write failed for {rec['path']}: {e}")
S.recat_progress = {"done": done, "total": total, "changed": changed}
broadcast("recat", S.recat_progress)
push_stats()
compact_journal()
load_journal()
push_stats()
log("ok", f"smart recategorize complete: {changed} of {total} photos moved to a new category")
finally:
with S.lock:
S.recat_status = "idle"
broadcast("recat", {**S.recat_progress, "finished": True})
def find_duplicate_groups():
"""Groups of exact perceptual-hash matches and size/filename matches."""
records = {r["path"]: r for r in load_search_index()}
@@ -773,6 +904,27 @@ def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read())
def ollama_generate_text(model, prompt, opts, keep_alive, think=None, schema=SCHEMA):
"""Text-only generation (no image) — used for classifying/organizing
against descriptions that were already produced by a vision pass, so
re-analyzing the whole library never needs the vision model again."""
body = {
"model": model,
"prompt": prompt,
"stream": False,
"options": opts,
"keep_alive": keep_alive,
}
if schema is not None:
body["format"] = schema
if think is not None:
body["think"] = think
req = urllib.request.Request(
OLLAMA + "/api/generate", data=json.dumps(body).encode(),
headers={"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read())
# ---------------------------------------------------------------- pipeline
def scan_folder(folder, recursive=True):
@@ -1022,7 +1174,14 @@ def build_prompt(length_key, mode="photo"):
'Answer as JSON: {"description": "...", "category": "..."}'
)
def write_metadata(path, desc, category, keep_backup, preserve_date):
def write_metadata(path, desc, category, keep_backup, preserve_date, old_category=None):
"""Idempotent regardless of how many times a photo gets (re)tagged:
category and the photon-tagged marker are list-type XMP/IPTC tags, so a
bare += on every write would silently accumulate duplicates (a photo
retagged 3 times would carry 3 copies of "photon-tagged" and every
category it ever had). Always strip the exact values being (re)written
first, then add them back exactly once. old_category additionally
strips a genuinely different previous category."""
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"]
if preserve_date:
@@ -1032,13 +1191,19 @@ def write_metadata(path, desc, category, keep_backup, preserve_date):
if is_video:
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
# tag names so it resolves to QuickTime/Keys groups automatically.
if old_category and old_category != category:
args.append(f"-Keywords-={old_category}")
args += [
f"-Keywords-={category}", f"-Keywords-=photon-tagged",
f"-Description={desc}",
f"-Keywords+={category}",
"-Keywords+=photon-tagged",
]
else:
if old_category and old_category != category:
args += [f"-XMP-dc:Subject-={old_category}", f"-IPTC:Keywords-={old_category}"]
args += [
f"-XMP-dc:Subject-={category}", f"-IPTC:Keywords-={category}", "-XMP-dc:Subject-=photon-tagged",
f"-EXIF:ImageDescription={desc}",
f"-IPTC:Caption-Abstract={desc}",
f"-XMP-dc:Description={desc}",
@@ -1160,6 +1325,10 @@ def process_loop(settings):
if think_map[m] is False:
log("info", f"{m} is a thinking model — thinking disabled for speed")
# built once so a retag (skipDone=false on an already-tagged photo) can
# strip its old category keyword instead of accumulating it
prev_categories = {p: rec.get("category") for p, rec in read_journal_deduped().items()}
tmpdir = tempfile.mkdtemp(prefix="photon_")
log("info", f"engine online :: {'router mode' if router else 'model=' + model} "
f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}")
@@ -1332,11 +1501,12 @@ def process_loop(settings):
continue
try:
old_cat = prev_categories.get(path)
if not dry:
if integrity:
verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date)
verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date, old_cat)
else:
write_metadata(path, desc, category, keep_backup, preserve_date)
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
@@ -1468,7 +1638,9 @@ class Handler(BaseHTTPRequestHandler):
"folder": S.folders[0] if S.folders else "",
"log": S.log_ring[-200:],
"categories": CATEGORIES, "current": S.current,
"settings": S.settings, "failuresCount": len(S.failed_paths)})
"settings": S.settings, "failuresCount": len(S.failed_paths),
"recatStatus": S.recat_status, "recatProgress": S.recat_progress,
"recatProposal": S.recat_proposal})
self._json(payload)
elif self.path.startswith("/api/preview"):
with S.lock:
@@ -2191,7 +2363,8 @@ class Handler(BaseHTTPRequestHandler):
keep_backup = bool(S.settings.get("keepBackup", False))
preserve_date = bool(S.settings.get("preserveDate", True))
organize = bool(S.settings.get("organize", False))
write_metadata(path, desc, category, keep_backup, preserve_date)
old_cat = read_journal_deduped().get(path, {}).get("category")
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
existing_recs = []
@@ -2298,7 +2471,8 @@ class Handler(BaseHTTPRequestHandler):
desc = f"{desc} | text: {words[:300]}"
except Exception as e:
pass
write_metadata(path, desc, category, keep_backup, preserve_date)
old_cat = read_journal_deduped().get(path, {}).get("category")
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
if organize:
organize_alias(find_owning_folder(path), path, category)
existing_recs = []
@@ -2402,6 +2576,47 @@ class Handler(BaseHTTPRequestHandler):
threading.Thread(target=build_dedupe_index, daemon=True).start()
self._json({"ok": True})
elif self.path == "/api/recategorize/propose":
model = body.get("model") or "qwen3.5:9b"
with S.lock:
if S.recat_status != "idle":
self._json({"error": "a recategorize job is already running"}, 400); return
S.recat_status = "proposing"
try:
samples, cat_sizes, total = sample_descriptions_for_proposal()
if total == 0:
self._json({"error": "no tagged photos yet to analyze"}, 400); return
prompt = build_category_proposal_prompt(samples, cat_sizes, total)
think = False if model_thinks(model) else None
log("info", f"smart recategorize: asking {model} to propose new categories "
f"from {len(samples)} sampled descriptions ({total} photos total) …")
resp = ollama_generate_text(model, prompt, {"temperature": 0.3, "num_predict": 900},
"5m", think, CATEGORY_PROPOSAL_SCHEMA)
parsed = salvage_json(resp.get("response", ""))
proposed = [c.strip() for c in (parsed.get("categories") or []) if c and c.strip()]
if not proposed:
self._json({"error": "model returned no categories"}, 500); return
with S.lock:
S.recat_proposal = proposed
log("ok", f"smart recategorize: proposed {len(proposed)} categories")
self._json({"ok": True, "categories": proposed, "currentDistribution": cat_sizes})
except Exception as e:
self._json({"error": f"proposal failed: {e}"}, 500)
finally:
with S.lock:
S.recat_status = "idle"
elif self.path == "/api/recategorize/apply":
categories = body.get("categories")
model = body.get("model") or "qwen3.5:9b"
if not isinstance(categories, list) or not categories:
self._json({"error": "categories list is required"}, 400); return
with S.lock:
if S.recat_status != "idle":
self._json({"error": "a recategorize job is already running"}, 400); return
threading.Thread(target=recategorize_apply_worker, args=(categories, model), daemon=True).start()
self._json({"ok": True})
elif self.path == "/api/bulk_recategorize":
paths = body.get("paths") or []
category = body.get("category")
@@ -2415,8 +2630,9 @@ class Handler(BaseHTTPRequestHandler):
for p in paths:
rec = records.get(p)
desc = rec.get("desc", "") if rec else ""
old_cat = rec.get("category") if rec else None
try:
write_metadata(p, desc, category, keep_backup, preserve_date)
write_metadata(p, desc, category, keep_backup, preserve_date, old_cat)
ok += 1
except Exception as e:
failed.append({"path": p, "error": str(e)})