Add smart recategorize (auto-discovered categories) and fix stale category-keyword accumulation
- New "Smart Recategorize" panel: analyzes existing journal descriptions (no vision re-run, no images touched) to propose a richer category list when one bucket dominates the library, then reclassifies every tagged photo into it using a fast local text-only model pass in batches of 25. Verified end-to-end on the real library: correctly split "Objects & Stuff" (35% of 15k photos) into specific categories like Electronics & Hardware, Cannabis Cultivation, Minerals & Crystals, Tools & Workshop based on what's actually in the descriptions. - New ollama_generate_text() for text-only local model calls (the existing ollama_generate always sent an "images" key, unsuitable for pure text classification). - FIX: write_metadata only ever appended category/photon-tagged keywords (+=) to the list-type XMP/IPTC tags, so retagging a photo left every category it ever had stacked in its metadata, and even rewriting with the same category duplicated "photon-tagged" each time. Found via a live test while building this feature (which recategorizes at library scale). Now strips old values before adding, making every write idempotent regardless of how many times a photo is retagged. Wired through all four write sites (pipeline, write_single, redo_single, bulk_recategorize). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
230
server.py
230
server.py
@@ -13,6 +13,7 @@ import hashlib
|
||||
import json
|
||||
import os
|
||||
import queue
|
||||
import random
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
@@ -195,6 +196,9 @@ class State:
|
||||
self.skipped_sidecars = 0
|
||||
self.dedupe_status = "idle" # idle | running
|
||||
self.dedupe_progress = {"done": 0, "total": 0}
|
||||
self.recat_status = "idle" # idle | proposing | running
|
||||
self.recat_progress = {"done": 0, "total": 0, "changed": 0}
|
||||
self.recat_proposal = [] # last LLM-proposed category list, for the UI to recall
|
||||
self.already_done = 0
|
||||
self.processed_session = 0
|
||||
self.failed_session = 0
|
||||
@@ -638,6 +642,133 @@ def build_dedupe_index():
|
||||
with S.lock:
|
||||
S.dedupe_status = "idle"
|
||||
|
||||
def sample_descriptions_for_proposal(max_total=350):
|
||||
"""Stratified sample across every current category, weighted so a
|
||||
dumping-ground category (e.g. one holding 35% of the library) is
|
||||
oversampled relative to its share — the model needs to actually SEE
|
||||
what's clogging it up, not just be told the percentage."""
|
||||
records = load_search_index()
|
||||
by_cat = {}
|
||||
for r in records:
|
||||
by_cat.setdefault(r.get("category") or "Uncategorized", []).append(r)
|
||||
total = len(records)
|
||||
cat_sizes = {c: len(rs) for c, rs in by_cat.items()}
|
||||
samples = []
|
||||
for cat, rs in by_cat.items():
|
||||
share = max(10, min(120, int(max_total * (len(rs) / total) ** 0.5))) if total else 0
|
||||
chosen = random.sample(rs, min(share, len(rs)))
|
||||
for r in chosen:
|
||||
samples.append({"category": cat, "desc": (r.get("desc") or "")[:150]})
|
||||
random.shuffle(samples)
|
||||
return samples[:max_total], cat_sizes, total
|
||||
|
||||
def build_category_proposal_prompt(samples, cat_sizes, total):
|
||||
dist = "\n".join(
|
||||
f"- {c}: {n} photos ({100*n/total:.0f}%)"
|
||||
for c, n in sorted(cat_sizes.items(), key=lambda x: -x[1])
|
||||
)
|
||||
sample_lines = "\n".join(f'[{s["category"]}] {s["desc"]}' for s in samples)
|
||||
return (
|
||||
"You are refining a personal photo library's category taxonomy.\n"
|
||||
f"Current categories and their share of {total} total photos:\n{dist}\n\n"
|
||||
"Below is a random sample of actual photo descriptions, each labeled with its "
|
||||
"current category:\n" + sample_lines + "\n\n"
|
||||
"The categories above are too coarse — the largest ones are dumping grounds "
|
||||
"covering very different kinds of photos. Propose a NEW list of 15 to 25 "
|
||||
"categories that fits this actual library well, so no single category dominates. "
|
||||
"Keep any current category that's already specific and reasonably sized; split "
|
||||
"the overloaded/vague ones based on what the sample actually shows (e.g. separate "
|
||||
"electronics from plants from receipts from tools — whatever the real content is, "
|
||||
"don't guess generically). Every name should be short (2-4 words) and mutually "
|
||||
"exclusive from the others.\n"
|
||||
'Answer as JSON: {"categories": ["...", "...", ...]}'
|
||||
)
|
||||
|
||||
CATEGORY_PROPOSAL_SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"categories": {"type": "array", "items": {"type": "string"}}},
|
||||
"required": ["categories"],
|
||||
}
|
||||
|
||||
def build_reclassify_prompt(categories, batch):
|
||||
cats = "\n".join(f"- {c}" for c in categories)
|
||||
items = "\n".join(f"{i}. {b['desc'][:200]}" for i, b in enumerate(batch))
|
||||
return (
|
||||
"Classify each numbered photo description below into EXACTLY ONE of these "
|
||||
f"categories:\n{cats}\n\nDescriptions:\n{items}\n\n"
|
||||
"Answer as JSON: {\"categories\": [\"<category for item 0>\", \"<category for item 1>\", ...]} "
|
||||
"— one entry per description, in the same order, same count."
|
||||
)
|
||||
|
||||
def reclassify_batch_schema(n):
|
||||
return {
|
||||
"type": "object",
|
||||
"properties": {"categories": {"type": "array", "items": {"type": "string"}, "minItems": n, "maxItems": n}},
|
||||
"required": ["categories"],
|
||||
}
|
||||
|
||||
def recategorize_apply_worker(new_categories, model):
|
||||
with S.lock:
|
||||
if S.recat_status != "idle":
|
||||
return
|
||||
S.recat_status = "running"
|
||||
keep_backup = bool(S.settings.get("keepBackup", False))
|
||||
preserve_date = bool(S.settings.get("preserveDate", True))
|
||||
organize = bool(S.settings.get("organize", False))
|
||||
try:
|
||||
save_categories(new_categories)
|
||||
load_journal()
|
||||
push_stats()
|
||||
records = load_search_index()
|
||||
total = len(records)
|
||||
S.recat_progress = {"done": 0, "total": total, "changed": 0}
|
||||
log("info", f"smart recategorize: reclassifying {total} photos into "
|
||||
f"{len(new_categories)} categories …")
|
||||
think = False if model_thinks(model) else None
|
||||
batch_size = 25
|
||||
changed = 0
|
||||
done = 0
|
||||
for i in range(0, total, batch_size):
|
||||
batch = records[i:i + batch_size]
|
||||
prompt = build_reclassify_prompt(new_categories, batch)
|
||||
try:
|
||||
resp = ollama_generate_text(model, prompt, {"temperature": 0, "num_predict": 800},
|
||||
"10m", think, reclassify_batch_schema(len(batch)))
|
||||
parsed = salvage_json(resp.get("response", ""))
|
||||
cats = parsed.get("categories") or []
|
||||
except Exception as e:
|
||||
log("error", f"recategorize batch {i}-{i+len(batch)} failed: {e}")
|
||||
cats = []
|
||||
for j, rec in enumerate(batch):
|
||||
done += 1
|
||||
new_cat = cats[j] if j < len(cats) else None
|
||||
if new_cat not in new_categories:
|
||||
new_cat = None
|
||||
old_cat = rec.get("category")
|
||||
if new_cat and new_cat != old_cat:
|
||||
try:
|
||||
write_metadata(rec["path"], rec.get("desc", ""), new_cat,
|
||||
keep_backup, preserve_date, old_cat)
|
||||
if organize:
|
||||
organize_alias(find_owning_folder(rec["path"]), rec["path"], new_cat)
|
||||
journal_write({"path": rec["path"], "desc": rec.get("desc", ""),
|
||||
"category": new_cat, "model": "recategorize",
|
||||
"route": rec.get("route"), "sec": 0.0, "ts": time.time()})
|
||||
changed += 1
|
||||
except Exception as e:
|
||||
log("error", f"recategorize write failed for {rec['path']}: {e}")
|
||||
S.recat_progress = {"done": done, "total": total, "changed": changed}
|
||||
broadcast("recat", S.recat_progress)
|
||||
push_stats()
|
||||
compact_journal()
|
||||
load_journal()
|
||||
push_stats()
|
||||
log("ok", f"smart recategorize complete: {changed} of {total} photos moved to a new category")
|
||||
finally:
|
||||
with S.lock:
|
||||
S.recat_status = "idle"
|
||||
broadcast("recat", {**S.recat_progress, "finished": True})
|
||||
|
||||
def find_duplicate_groups():
|
||||
"""Groups of exact perceptual-hash matches and size/filename matches."""
|
||||
records = {r["path"]: r for r in load_search_index()}
|
||||
@@ -773,6 +904,27 @@ def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
return json.loads(r.read())
|
||||
|
||||
def ollama_generate_text(model, prompt, opts, keep_alive, think=None, schema=SCHEMA):
|
||||
"""Text-only generation (no image) — used for classifying/organizing
|
||||
against descriptions that were already produced by a vision pass, so
|
||||
re-analyzing the whole library never needs the vision model again."""
|
||||
body = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"stream": False,
|
||||
"options": opts,
|
||||
"keep_alive": keep_alive,
|
||||
}
|
||||
if schema is not None:
|
||||
body["format"] = schema
|
||||
if think is not None:
|
||||
body["think"] = think
|
||||
req = urllib.request.Request(
|
||||
OLLAMA + "/api/generate", data=json.dumps(body).encode(),
|
||||
headers={"Content-Type": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
return json.loads(r.read())
|
||||
|
||||
# ---------------------------------------------------------------- pipeline
|
||||
|
||||
def scan_folder(folder, recursive=True):
|
||||
@@ -1022,7 +1174,14 @@ def build_prompt(length_key, mode="photo"):
|
||||
'Answer as JSON: {"description": "...", "category": "..."}'
|
||||
)
|
||||
|
||||
def write_metadata(path, desc, category, keep_backup, preserve_date):
|
||||
def write_metadata(path, desc, category, keep_backup, preserve_date, old_category=None):
|
||||
"""Idempotent regardless of how many times a photo gets (re)tagged:
|
||||
category and the photon-tagged marker are list-type XMP/IPTC tags, so a
|
||||
bare += on every write would silently accumulate duplicates (a photo
|
||||
retagged 3 times would carry 3 copies of "photon-tagged" and every
|
||||
category it ever had). Always strip the exact values being (re)written
|
||||
first, then add them back exactly once. old_category additionally
|
||||
strips a genuinely different previous category."""
|
||||
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
||||
args = ["exiftool", "-m", "-q", "-codedcharacterset=utf8"]
|
||||
if preserve_date:
|
||||
@@ -1032,13 +1191,19 @@ def write_metadata(path, desc, category, keep_backup, preserve_date):
|
||||
if is_video:
|
||||
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
|
||||
# tag names so it resolves to QuickTime/Keys groups automatically.
|
||||
if old_category and old_category != category:
|
||||
args.append(f"-Keywords-={old_category}")
|
||||
args += [
|
||||
f"-Keywords-={category}", f"-Keywords-=photon-tagged",
|
||||
f"-Description={desc}",
|
||||
f"-Keywords+={category}",
|
||||
"-Keywords+=photon-tagged",
|
||||
]
|
||||
else:
|
||||
if old_category and old_category != category:
|
||||
args += [f"-XMP-dc:Subject-={old_category}", f"-IPTC:Keywords-={old_category}"]
|
||||
args += [
|
||||
f"-XMP-dc:Subject-={category}", f"-IPTC:Keywords-={category}", "-XMP-dc:Subject-=photon-tagged",
|
||||
f"-EXIF:ImageDescription={desc}",
|
||||
f"-IPTC:Caption-Abstract={desc}",
|
||||
f"-XMP-dc:Description={desc}",
|
||||
@@ -1160,6 +1325,10 @@ def process_loop(settings):
|
||||
if think_map[m] is False:
|
||||
log("info", f"{m} is a thinking model — thinking disabled for speed")
|
||||
|
||||
# built once so a retag (skipDone=false on an already-tagged photo) can
|
||||
# strip its old category keyword instead of accumulating it
|
||||
prev_categories = {p: rec.get("category") for p, rec in read_journal_deduped().items()}
|
||||
|
||||
tmpdir = tempfile.mkdtemp(prefix="photon_")
|
||||
log("info", f"engine online :: {'router mode' if router else 'model=' + model} "
|
||||
f"resize={max_px or 'off'}px len={length_key} temp={temp} dry_run={dry} integrity_check={integrity} organize_aliases={organize}")
|
||||
@@ -1332,11 +1501,12 @@ def process_loop(settings):
|
||||
continue
|
||||
|
||||
try:
|
||||
old_cat = prev_categories.get(path)
|
||||
if not dry:
|
||||
if integrity:
|
||||
verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date)
|
||||
verify_pixel_integrity(path, write_metadata, path, desc, category, keep_backup, preserve_date, old_cat)
|
||||
else:
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date)
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
|
||||
if organize:
|
||||
organize_alias(find_owning_folder(path), path, category)
|
||||
|
||||
@@ -1468,7 +1638,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"folder": S.folders[0] if S.folders else "",
|
||||
"log": S.log_ring[-200:],
|
||||
"categories": CATEGORIES, "current": S.current,
|
||||
"settings": S.settings, "failuresCount": len(S.failed_paths)})
|
||||
"settings": S.settings, "failuresCount": len(S.failed_paths),
|
||||
"recatStatus": S.recat_status, "recatProgress": S.recat_progress,
|
||||
"recatProposal": S.recat_proposal})
|
||||
self._json(payload)
|
||||
elif self.path.startswith("/api/preview"):
|
||||
with S.lock:
|
||||
@@ -2191,7 +2363,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
keep_backup = bool(S.settings.get("keepBackup", False))
|
||||
preserve_date = bool(S.settings.get("preserveDate", True))
|
||||
organize = bool(S.settings.get("organize", False))
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date)
|
||||
old_cat = read_journal_deduped().get(path, {}).get("category")
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
|
||||
if organize:
|
||||
organize_alias(find_owning_folder(path), path, category)
|
||||
existing_recs = []
|
||||
@@ -2298,7 +2471,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
desc = f"{desc} | text: {words[:300]}"
|
||||
except Exception as e:
|
||||
pass
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date)
|
||||
old_cat = read_journal_deduped().get(path, {}).get("category")
|
||||
write_metadata(path, desc, category, keep_backup, preserve_date, old_cat)
|
||||
if organize:
|
||||
organize_alias(find_owning_folder(path), path, category)
|
||||
existing_recs = []
|
||||
@@ -2402,6 +2576,47 @@ class Handler(BaseHTTPRequestHandler):
|
||||
threading.Thread(target=build_dedupe_index, daemon=True).start()
|
||||
self._json({"ok": True})
|
||||
|
||||
elif self.path == "/api/recategorize/propose":
|
||||
model = body.get("model") or "qwen3.5:9b"
|
||||
with S.lock:
|
||||
if S.recat_status != "idle":
|
||||
self._json({"error": "a recategorize job is already running"}, 400); return
|
||||
S.recat_status = "proposing"
|
||||
try:
|
||||
samples, cat_sizes, total = sample_descriptions_for_proposal()
|
||||
if total == 0:
|
||||
self._json({"error": "no tagged photos yet to analyze"}, 400); return
|
||||
prompt = build_category_proposal_prompt(samples, cat_sizes, total)
|
||||
think = False if model_thinks(model) else None
|
||||
log("info", f"smart recategorize: asking {model} to propose new categories "
|
||||
f"from {len(samples)} sampled descriptions ({total} photos total) …")
|
||||
resp = ollama_generate_text(model, prompt, {"temperature": 0.3, "num_predict": 900},
|
||||
"5m", think, CATEGORY_PROPOSAL_SCHEMA)
|
||||
parsed = salvage_json(resp.get("response", ""))
|
||||
proposed = [c.strip() for c in (parsed.get("categories") or []) if c and c.strip()]
|
||||
if not proposed:
|
||||
self._json({"error": "model returned no categories"}, 500); return
|
||||
with S.lock:
|
||||
S.recat_proposal = proposed
|
||||
log("ok", f"smart recategorize: proposed {len(proposed)} categories")
|
||||
self._json({"ok": True, "categories": proposed, "currentDistribution": cat_sizes})
|
||||
except Exception as e:
|
||||
self._json({"error": f"proposal failed: {e}"}, 500)
|
||||
finally:
|
||||
with S.lock:
|
||||
S.recat_status = "idle"
|
||||
|
||||
elif self.path == "/api/recategorize/apply":
|
||||
categories = body.get("categories")
|
||||
model = body.get("model") or "qwen3.5:9b"
|
||||
if not isinstance(categories, list) or not categories:
|
||||
self._json({"error": "categories list is required"}, 400); return
|
||||
with S.lock:
|
||||
if S.recat_status != "idle":
|
||||
self._json({"error": "a recategorize job is already running"}, 400); return
|
||||
threading.Thread(target=recategorize_apply_worker, args=(categories, model), daemon=True).start()
|
||||
self._json({"ok": True})
|
||||
|
||||
elif self.path == "/api/bulk_recategorize":
|
||||
paths = body.get("paths") or []
|
||||
category = body.get("category")
|
||||
@@ -2415,8 +2630,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
for p in paths:
|
||||
rec = records.get(p)
|
||||
desc = rec.get("desc", "") if rec else ""
|
||||
old_cat = rec.get("category") if rec else None
|
||||
try:
|
||||
write_metadata(p, desc, category, keep_backup, preserve_date)
|
||||
write_metadata(p, desc, category, keep_backup, preserve_date, old_cat)
|
||||
ok += 1
|
||||
except Exception as e:
|
||||
failed.append({"path": p, "error": str(e)})
|
||||
|
||||
Reference in New Issue
Block a user