diff --git a/.gitignore b/.gitignore
index b55708f..cacf79a 100644
--- a/.gitignore
+++ b/.gitignore
@@ -6,4 +6,6 @@ photon_audits.jsonl
photon_phash.json
photon_folder.json
photon_categories.json
+.photon.pid
+.photon_server.log
.claude/
diff --git a/PHOTON.app/Contents/Info.plist b/PHOTON.app/Contents/Info.plist
new file mode 100644
index 0000000..b577361
--- /dev/null
+++ b/PHOTON.app/Contents/Info.plist
@@ -0,0 +1,28 @@
+
+
+
+
+ CFBundleExecutable
+ PHOTON
+ CFBundleIdentifier
+ local.photon.console
+ CFBundleName
+ PHOTON
+ CFBundleDisplayName
+ PHOTON
+ CFBundlePackageType
+ APPL
+ CFBundleShortVersionString
+ 1.0
+ CFBundleVersion
+ 1
+ LSUIElement
+
+ LSMinimumSystemVersion
+ 11.0
+ NSHighResolutionCapable
+
+ LSApplicationCategoryType
+ public.app-category.utilities
+
+
diff --git a/PHOTON.app/Contents/MacOS/PHOTON b/PHOTON.app/Contents/MacOS/PHOTON
new file mode 100755
index 0000000..ae3f594
Binary files /dev/null and b/PHOTON.app/Contents/MacOS/PHOTON differ
diff --git a/PHOTON.app/Contents/_CodeSignature/CodeResources b/PHOTON.app/Contents/_CodeSignature/CodeResources
new file mode 100644
index 0000000..d5d0fd7
--- /dev/null
+++ b/PHOTON.app/Contents/_CodeSignature/CodeResources
@@ -0,0 +1,115 @@
+
+
+
+
+ files
+
+ files2
+
+ rules
+
+ ^Resources/
+
+ ^Resources/.*\.lproj/
+
+ optional
+
+ weight
+ 1000
+
+ ^Resources/.*\.lproj/locversion.plist$
+
+ omit
+
+ weight
+ 1100
+
+ ^Resources/Base\.lproj/
+
+ weight
+ 1010
+
+ ^version.plist$
+
+
+ rules2
+
+ .*\.dSYM($|/)
+
+ weight
+ 11
+
+ ^(.*/)?\.DS_Store$
+
+ omit
+
+ weight
+ 2000
+
+ ^(Frameworks|SharedFrameworks|PlugIns|Plug-ins|XPCServices|Helpers|MacOS|Library/(Automator|Spotlight|LoginItems))/
+
+ nested
+
+ weight
+ 10
+
+ ^.*
+
+ ^Info\.plist$
+
+ omit
+
+ weight
+ 20
+
+ ^PkgInfo$
+
+ omit
+
+ weight
+ 20
+
+ ^Resources/
+
+ weight
+ 20
+
+ ^Resources/.*\.lproj/
+
+ optional
+
+ weight
+ 1000
+
+ ^Resources/.*\.lproj/locversion.plist$
+
+ omit
+
+ weight
+ 1100
+
+ ^Resources/Base\.lproj/
+
+ weight
+ 1010
+
+ ^[^/]+$
+
+ nested
+
+ weight
+ 10
+
+ ^embedded\.provisionprofile$
+
+ weight
+ 20
+
+ ^version\.plist$
+
+ weight
+ 20
+
+
+
+
diff --git a/server.py b/server.py
index 0579043..d7368f9 100644
--- a/server.py
+++ b/server.py
@@ -227,30 +227,55 @@ def resolve_portable_path(path):
return cand
return path
+def read_journal_deduped():
+ """Read the journal into {abs_path: latest_record}. journal_write only
+ ever appends (fast during a tagging run), so a photo retagged more than
+ once leaves stale earlier lines behind — this keeps just the newest
+ record per path so counts/search never double up a single photo."""
+ latest = {}
+ if not os.path.exists(JOURNAL):
+ return latest
+ with open(JOURNAL, "r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ try:
+ rec = json.loads(line)
+ path = resolve_portable_path(rec.get("path", ""))
+ latest[path] = rec
+ except Exception:
+ pass
+ return latest
+
+def compact_journal():
+ """Rewrite the journal keeping only the latest record per path, so
+ on-disk duplicate lines from past retags don't linger forever."""
+ latest = read_journal_deduped()
+ if not latest:
+ return 0
+ with open(JOURNAL, "r", encoding="utf-8") as f:
+ line_count = sum(1 for l in f if l.strip())
+ removed = line_count - len(latest)
+ if removed > 0:
+ with open(JOURNAL, "w", encoding="utf-8") as f:
+ for rec in latest.values():
+ f.write(json.dumps(rec, ensure_ascii=False) + "\n")
+ return removed
+
def load_journal():
- n = 0
with S.lock:
S.cat_counts = {c: 0 for c in CATEGORIES}
S.done_paths = set()
- if os.path.exists(JOURNAL):
- with open(JOURNAL, "r", encoding="utf-8") as f:
- for line in f:
- line = line.strip()
- if not line:
- continue
- try:
- rec = json.loads(line)
- path = resolve_portable_path(rec.get("path", ""))
- with S.lock:
- S.done_paths.add(path)
- cat = rec.get("category")
- if cat not in S.cat_counts:
- S.cat_counts[cat] = 0
- S.cat_counts[cat] += 1
- n += 1
- except Exception:
- pass
- return n
+ latest = read_journal_deduped()
+ with S.lock:
+ for path, rec in latest.items():
+ S.done_paths.add(path)
+ cat = rec.get("category")
+ if cat not in S.cat_counts:
+ S.cat_counts[cat] = 0
+ S.cat_counts[cat] += 1
+ return len(latest)
def journal_write(rec):
with open(JOURNAL, "a", encoding="utf-8") as f:
@@ -273,77 +298,117 @@ def trash_or_remove_file(path):
log("error", f"Failed to delete file {path}: {e}")
return False
-def remove_photo_records(path):
- abs_path = os.path.abspath(path)
+def cleanup_organized_symlinks(abs_paths):
+ """Remove _organized alias symlinks pointing at any of the given files.
+ One folder walk covers the whole batch instead of one walk per file."""
+ if not abs_paths:
+ return
with S.lock:
folder = S.folder
org_root = os.path.join(folder, "_organized")
- if os.path.exists(org_root):
- for root, dirs, files in os.walk(org_root):
- dirs[:] = [d for d in dirs if not d.startswith(".")]
- for name in files:
- p = os.path.join(root, name)
- if os.path.islink(p):
+ if not os.path.exists(org_root):
+ return
+ targets = set(abs_paths)
+ for root, dirs, files in os.walk(org_root):
+ dirs[:] = [d for d in dirs if not d.startswith(".")]
+ for name in files:
+ p = os.path.join(root, name)
+ if os.path.islink(p):
+ try:
+ if os.path.realpath(p) in targets:
+ os.remove(p)
+ except Exception:
+ pass
+
+def cleanup_thumb_cache(abs_paths):
+ for abs_path in abs_paths:
+ try:
+ h = hashlib.md5(abs_path.encode()).hexdigest()
+ for sub in ("photon_thumbs", "photon_view"):
+ t = os.path.join(tempfile.gettempdir(), sub, f"{h}.jpg")
+ if os.path.exists(t):
try:
- if os.path.realpath(p) == abs_path:
- os.remove(p)
+ os.remove(t)
except Exception:
pass
+ except Exception:
+ pass
- try:
- h = hashlib.md5(abs_path.encode()).hexdigest()
- t1 = os.path.join(tempfile.gettempdir(), "photon_thumbs", f"{h}.jpg")
- t2 = os.path.join(tempfile.gettempdir(), "photon_view", f"{h}.jpg")
- for t in (t1, t2):
- if os.path.exists(t):
- try: os.remove(t)
- except Exception: pass
- except Exception:
- pass
-
- if os.path.exists(JOURNAL):
- lines_to_keep = []
- with open(JOURNAL, "r", encoding="utf-8") as f:
- for line in f:
- line_str = line.strip()
- if not line_str:
- continue
- try:
- rec = json.loads(line_str)
- if os.path.abspath(rec.get("path", "")) != abs_path:
- lines_to_keep.append(line_str)
- except Exception:
+def remove_journal_entries(abs_paths):
+ """Strip every journal line matching any path in abs_paths in a single
+ read+write pass, instead of rewriting the whole journal once per file."""
+ if not os.path.exists(JOURNAL) or not abs_paths:
+ return
+ targets = set(abs_paths)
+ lines_to_keep = []
+ with open(JOURNAL, "r", encoding="utf-8") as f:
+ for line in f:
+ line_str = line.strip()
+ if not line_str:
+ continue
+ try:
+ rec = json.loads(line_str)
+ if os.path.abspath(rec.get("path", "")) not in targets:
lines_to_keep.append(line_str)
- with open(JOURNAL, "w", encoding="utf-8") as f:
- for l in lines_to_keep:
- f.write(l + "\n")
+ except Exception:
+ lines_to_keep.append(line_str)
+ with open(JOURNAL, "w", encoding="utf-8") as f:
+ for l in lines_to_keep:
+ f.write(l + "\n")
+def remove_photo_records_batch(paths):
+ """Purge journal/cache/in-memory records for a whole batch of files at
+ once. Used by both single and bulk delete so a duplicate-cleanup of
+ hundreds of files does one journal pass instead of hundreds."""
+ abs_paths = [os.path.abspath(p) for p in paths]
+ if not abs_paths:
+ return
+ cleanup_organized_symlinks(abs_paths)
+ cleanup_thumb_cache(abs_paths)
+ remove_journal_entries(abs_paths)
+
+ abs_set = set(abs_paths)
+ removed_failed = None
with S.lock:
- if abs_path in S.done_paths:
- S.done_paths.remove(abs_path)
- if abs_path in S.files:
- S.files.remove(abs_path)
- if abs_path in S.video_files:
- S.video_files.remove(abs_path)
- if abs_path in S.failed_paths:
- S.failed_paths.remove(abs_path)
- write_failures()
-
+ S.done_paths -= abs_set
+ S.files = [p for p in S.files if p not in abs_set]
+ S.video_files = [p for p in S.video_files if p not in abs_set]
+ removed_failed = S.failed_paths & abs_set
+ if removed_failed:
+ S.failed_paths -= removed_failed
+ if removed_failed:
+ write_failures()
+
global SEARCH_CACHE
SEARCH_CACHE = {"mtime": None, "records": []}
load_journal()
push_stats()
-def delete_photo_file_and_record(path):
+def remove_photo_records(path):
+ remove_photo_records_batch([path])
+
+def delete_photo_file_and_record(path, skip_records=False):
+ """skip_records=True defers journal/cache cleanup to a single batched
+ call afterwards — used by bulk delete so N files cost one journal pass,
+ not N. The single-delete endpoint uses the default (clean up right away)."""
if not path:
return False, "path is required"
abs_path = os.path.abspath(path)
+ with S.lock:
+ folder = os.path.abspath(S.folder) if S.folder else None
+ try:
+ inside_folder = bool(folder) and os.path.commonpath([folder, abs_path]) == folder
+ except ValueError:
+ inside_folder = False
+ if not inside_folder:
+ return False, f"refusing to delete outside the configured photo folder: {abs_path}"
deleted_disk = False
if os.path.exists(abs_path):
deleted_disk = trash_or_remove_file(abs_path)
else:
deleted_disk = True
- remove_photo_records(abs_path)
+ if not skip_records:
+ remove_photo_records(abs_path)
return deleted_disk, None
SEARCH_CACHE = {"mtime": None, "records": []}
@@ -358,32 +423,26 @@ def load_search_index():
if SEARCH_CACHE["mtime"] == mtime:
return SEARCH_CACHE["records"]
records = []
- with open(JOURNAL, "r", encoding="utf-8") as f:
- for line in f:
- line = line.strip()
- if not line:
- continue
+ for path, rec in read_journal_deduped().items():
+ try:
+ rec["path"] = path
+ name = os.path.basename(path)
+ rec["name"] = name
+ rec["_ext"] = os.path.splitext(name)[1].lower()
+ rec["_hay"] = " ".join([
+ path.lower(), (rec.get("desc") or "").lower(),
+ (rec.get("category") or "").lower(), name.lower()
+ ])
try:
- rec = json.loads(line)
- path = resolve_portable_path(rec.get("path", ""))
- rec["path"] = path
- name = os.path.basename(path)
- rec["name"] = name
- rec["_ext"] = os.path.splitext(name)[1].lower()
- rec["_hay"] = " ".join([
- path.lower(), (rec.get("desc") or "").lower(),
- (rec.get("category") or "").lower(), name.lower()
- ])
- try:
- st = os.stat(path)
- rec["_mtime"] = st.st_mtime
- rec["sizeBytes"] = st.st_size
- except OSError:
- rec["_mtime"] = rec.get("ts", 0)
- rec["sizeBytes"] = None
- records.append(rec)
- except Exception:
- pass
+ st = os.stat(path)
+ rec["_mtime"] = st.st_mtime
+ rec["sizeBytes"] = st.st_size
+ except OSError:
+ rec["_mtime"] = rec.get("ts", 0)
+ rec["sizeBytes"] = None
+ records.append(rec)
+ except Exception:
+ pass
SEARCH_CACHE = {"mtime": mtime, "records": records}
return records
@@ -987,6 +1046,8 @@ def process_loop(settings):
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
organize = bool(settings.get("organize", False))
integrity = bool(settings.get("integrity", False))
+ with S.lock:
+ organize_folder = S.folder or DEFAULT_FOLDER
photo_prompt = build_prompt(length_key, "photo")
shot_prompt = build_prompt(length_key, "screenshot")
@@ -1034,49 +1095,66 @@ def process_loop(settings):
stop_pipeline = threading.Event()
consec_fail = 0
+ def should_halt():
+ return S.stop_evt.is_set() or stop_pipeline.is_set()
+
+ def put_until_halt(q, item):
+ """Blocking put that keeps re-checking stop flags instead of hanging on
+ a full queue for up to 60s when the consumer has already exited."""
+ while not should_halt():
+ try:
+ q.put(item, timeout=0.5)
+ return True
+ except queue.Full:
+ continue
+ return False
+
# Stage 1: Downscaler Thread
def downscale_worker():
for i, path in enumerate(pending):
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
while not S.pause_evt.is_set():
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
time.sleep(0.2)
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
try:
t0 = time.time()
img_bytes = downscale(path, max_px, tmpdir)
b64 = base64.b64encode(img_bytes).decode()
- downscale_queue.put((i, path, img_bytes, b64, t0), timeout=60)
+ if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)):
+ break
except Exception as e:
log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}")
- downscale_queue.put((i, path, None, str(e), time.time()), timeout=60)
- downscale_queue.put(None)
+ if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())):
+ break
+ put_until_halt(downscale_queue, None)
# Stage 2: Inference Thread
def inference_worker():
while True:
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
while not S.pause_evt.is_set():
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
time.sleep(0.2)
- if S.stop_evt.is_set() or stop_pipeline.is_set():
+ if should_halt():
break
try:
item = downscale_queue.get(timeout=1.0)
except queue.Empty:
continue
if item is None:
- write_queue.put(None)
+ put_until_halt(write_queue, None)
break
i, path, img_bytes, b64_or_err, t0 = item
name = os.path.basename(path)
if img_bytes is None:
- write_queue.put((i, path, None, None, None, f"Downscale error: {b64_or_err}", t0))
+ if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)):
+ break
continue
with S.lock:
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total}
@@ -1116,9 +1194,9 @@ def process_loop(settings):
desc = f"{desc} | text: {words[:300]}"
except Exception as e:
pass
- write_queue.put((i, path, desc, category, route, None, t0))
+ put_until_halt(write_queue, (i, path, desc, category, route, None, t0))
except Exception as e:
- write_queue.put((i, path, None, None, None, f"Inference error: {e}", t0))
+ put_until_halt(write_queue, (i, path, None, None, None, f"Inference error: {e}", t0))
t_downscale = threading.Thread(target=downscale_worker, daemon=True)
t_inference = threading.Thread(target=inference_worker, daemon=True)
@@ -1162,7 +1240,7 @@ def process_loop(settings):
else:
write_metadata(path, desc, category, keep_backup, preserve_date)
if organize:
- organize_alias(settings.get("folder") or DEFAULT_FOLDER, path, category)
+ organize_alias(organize_folder, path, category)
dt = time.time() - t0
with S.lock:
@@ -2120,16 +2198,19 @@ class Handler(BaseHTTPRequestHandler):
paths = body.get("paths") or []
if not paths:
self._json({"error": "paths array is required"}, 400); return
- deleted_count = 0
+ deleted_paths = []
failed = []
for p in paths:
- success, err = delete_photo_file_and_record(p)
+ success, err = delete_photo_file_and_record(p, skip_records=True)
if success:
- deleted_count += 1
+ deleted_paths.append(p)
else:
failed.append({"path": p, "error": err})
- log("ok", f"Batch deleted {deleted_count} photos from disk" + (f" ({len(failed)} failed)" if failed else ""))
- self._json({"ok": True, "deleted": deleted_count, "failed": failed})
+ # One batched journal/cache pass for the whole set, instead of
+ # one full journal rewrite per file.
+ remove_photo_records_batch(deleted_paths)
+ log("ok", f"Batch deleted {len(deleted_paths)} photos from disk" + (f" ({len(failed)} failed)" if failed else ""))
+ self._json({"ok": True, "deleted": len(deleted_paths), "failed": failed})
elif self.path == "/api/dedupe/build":
with S.lock:
@@ -2182,6 +2263,9 @@ class Handler(BaseHTTPRequestHandler):
self.send_response(404); self.end_headers()
def main():
+ removed = compact_journal()
+ if removed:
+ print(f"journal compacted: removed {removed} duplicate/stale entries from past retags")
n = load_journal()
load_failures()
load_phash_cache()