diff --git a/.gitignore b/.gitignore index b55708f..cacf79a 100644 --- a/.gitignore +++ b/.gitignore @@ -6,4 +6,6 @@ photon_audits.jsonl photon_phash.json photon_folder.json photon_categories.json +.photon.pid +.photon_server.log .claude/ diff --git a/PHOTON.app/Contents/Info.plist b/PHOTON.app/Contents/Info.plist new file mode 100644 index 0000000..b577361 --- /dev/null +++ b/PHOTON.app/Contents/Info.plist @@ -0,0 +1,28 @@ + + + + + CFBundleExecutable + PHOTON + CFBundleIdentifier + local.photon.console + CFBundleName + PHOTON + CFBundleDisplayName + PHOTON + CFBundlePackageType + APPL + CFBundleShortVersionString + 1.0 + CFBundleVersion + 1 + LSUIElement + + LSMinimumSystemVersion + 11.0 + NSHighResolutionCapable + + LSApplicationCategoryType + public.app-category.utilities + + diff --git a/PHOTON.app/Contents/MacOS/PHOTON b/PHOTON.app/Contents/MacOS/PHOTON new file mode 100755 index 0000000..ae3f594 Binary files /dev/null and b/PHOTON.app/Contents/MacOS/PHOTON differ diff --git a/PHOTON.app/Contents/_CodeSignature/CodeResources b/PHOTON.app/Contents/_CodeSignature/CodeResources new file mode 100644 index 0000000..d5d0fd7 --- /dev/null +++ b/PHOTON.app/Contents/_CodeSignature/CodeResources @@ -0,0 +1,115 @@ + + + + + files + + files2 + + rules + + ^Resources/ + + ^Resources/.*\.lproj/ + + optional + + weight + 1000 + + ^Resources/.*\.lproj/locversion.plist$ + + omit + + weight + 1100 + + ^Resources/Base\.lproj/ + + weight + 1010 + + ^version.plist$ + + + rules2 + + .*\.dSYM($|/) + + weight + 11 + + ^(.*/)?\.DS_Store$ + + omit + + weight + 2000 + + ^(Frameworks|SharedFrameworks|PlugIns|Plug-ins|XPCServices|Helpers|MacOS|Library/(Automator|Spotlight|LoginItems))/ + + nested + + weight + 10 + + ^.* + + ^Info\.plist$ + + omit + + weight + 20 + + ^PkgInfo$ + + omit + + weight + 20 + + ^Resources/ + + weight + 20 + + ^Resources/.*\.lproj/ + + optional + + weight + 1000 + + ^Resources/.*\.lproj/locversion.plist$ + + omit + + weight + 1100 + + ^Resources/Base\.lproj/ + + weight + 1010 + + ^[^/]+$ + + nested + + weight + 10 + + ^embedded\.provisionprofile$ + + weight + 20 + + ^version\.plist$ + + weight + 20 + + + + diff --git a/server.py b/server.py index 0579043..d7368f9 100644 --- a/server.py +++ b/server.py @@ -227,30 +227,55 @@ def resolve_portable_path(path): return cand return path +def read_journal_deduped(): + """Read the journal into {abs_path: latest_record}. journal_write only + ever appends (fast during a tagging run), so a photo retagged more than + once leaves stale earlier lines behind — this keeps just the newest + record per path so counts/search never double up a single photo.""" + latest = {} + if not os.path.exists(JOURNAL): + return latest + with open(JOURNAL, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + path = resolve_portable_path(rec.get("path", "")) + latest[path] = rec + except Exception: + pass + return latest + +def compact_journal(): + """Rewrite the journal keeping only the latest record per path, so + on-disk duplicate lines from past retags don't linger forever.""" + latest = read_journal_deduped() + if not latest: + return 0 + with open(JOURNAL, "r", encoding="utf-8") as f: + line_count = sum(1 for l in f if l.strip()) + removed = line_count - len(latest) + if removed > 0: + with open(JOURNAL, "w", encoding="utf-8") as f: + for rec in latest.values(): + f.write(json.dumps(rec, ensure_ascii=False) + "\n") + return removed + def load_journal(): - n = 0 with S.lock: S.cat_counts = {c: 0 for c in CATEGORIES} S.done_paths = set() - if os.path.exists(JOURNAL): - with open(JOURNAL, "r", encoding="utf-8") as f: - for line in f: - line = line.strip() - if not line: - continue - try: - rec = json.loads(line) - path = resolve_portable_path(rec.get("path", "")) - with S.lock: - S.done_paths.add(path) - cat = rec.get("category") - if cat not in S.cat_counts: - S.cat_counts[cat] = 0 - S.cat_counts[cat] += 1 - n += 1 - except Exception: - pass - return n + latest = read_journal_deduped() + with S.lock: + for path, rec in latest.items(): + S.done_paths.add(path) + cat = rec.get("category") + if cat not in S.cat_counts: + S.cat_counts[cat] = 0 + S.cat_counts[cat] += 1 + return len(latest) def journal_write(rec): with open(JOURNAL, "a", encoding="utf-8") as f: @@ -273,77 +298,117 @@ def trash_or_remove_file(path): log("error", f"Failed to delete file {path}: {e}") return False -def remove_photo_records(path): - abs_path = os.path.abspath(path) +def cleanup_organized_symlinks(abs_paths): + """Remove _organized alias symlinks pointing at any of the given files. + One folder walk covers the whole batch instead of one walk per file.""" + if not abs_paths: + return with S.lock: folder = S.folder org_root = os.path.join(folder, "_organized") - if os.path.exists(org_root): - for root, dirs, files in os.walk(org_root): - dirs[:] = [d for d in dirs if not d.startswith(".")] - for name in files: - p = os.path.join(root, name) - if os.path.islink(p): + if not os.path.exists(org_root): + return + targets = set(abs_paths) + for root, dirs, files in os.walk(org_root): + dirs[:] = [d for d in dirs if not d.startswith(".")] + for name in files: + p = os.path.join(root, name) + if os.path.islink(p): + try: + if os.path.realpath(p) in targets: + os.remove(p) + except Exception: + pass + +def cleanup_thumb_cache(abs_paths): + for abs_path in abs_paths: + try: + h = hashlib.md5(abs_path.encode()).hexdigest() + for sub in ("photon_thumbs", "photon_view"): + t = os.path.join(tempfile.gettempdir(), sub, f"{h}.jpg") + if os.path.exists(t): try: - if os.path.realpath(p) == abs_path: - os.remove(p) + os.remove(t) except Exception: pass + except Exception: + pass - try: - h = hashlib.md5(abs_path.encode()).hexdigest() - t1 = os.path.join(tempfile.gettempdir(), "photon_thumbs", f"{h}.jpg") - t2 = os.path.join(tempfile.gettempdir(), "photon_view", f"{h}.jpg") - for t in (t1, t2): - if os.path.exists(t): - try: os.remove(t) - except Exception: pass - except Exception: - pass - - if os.path.exists(JOURNAL): - lines_to_keep = [] - with open(JOURNAL, "r", encoding="utf-8") as f: - for line in f: - line_str = line.strip() - if not line_str: - continue - try: - rec = json.loads(line_str) - if os.path.abspath(rec.get("path", "")) != abs_path: - lines_to_keep.append(line_str) - except Exception: +def remove_journal_entries(abs_paths): + """Strip every journal line matching any path in abs_paths in a single + read+write pass, instead of rewriting the whole journal once per file.""" + if not os.path.exists(JOURNAL) or not abs_paths: + return + targets = set(abs_paths) + lines_to_keep = [] + with open(JOURNAL, "r", encoding="utf-8") as f: + for line in f: + line_str = line.strip() + if not line_str: + continue + try: + rec = json.loads(line_str) + if os.path.abspath(rec.get("path", "")) not in targets: lines_to_keep.append(line_str) - with open(JOURNAL, "w", encoding="utf-8") as f: - for l in lines_to_keep: - f.write(l + "\n") + except Exception: + lines_to_keep.append(line_str) + with open(JOURNAL, "w", encoding="utf-8") as f: + for l in lines_to_keep: + f.write(l + "\n") +def remove_photo_records_batch(paths): + """Purge journal/cache/in-memory records for a whole batch of files at + once. Used by both single and bulk delete so a duplicate-cleanup of + hundreds of files does one journal pass instead of hundreds.""" + abs_paths = [os.path.abspath(p) for p in paths] + if not abs_paths: + return + cleanup_organized_symlinks(abs_paths) + cleanup_thumb_cache(abs_paths) + remove_journal_entries(abs_paths) + + abs_set = set(abs_paths) + removed_failed = None with S.lock: - if abs_path in S.done_paths: - S.done_paths.remove(abs_path) - if abs_path in S.files: - S.files.remove(abs_path) - if abs_path in S.video_files: - S.video_files.remove(abs_path) - if abs_path in S.failed_paths: - S.failed_paths.remove(abs_path) - write_failures() - + S.done_paths -= abs_set + S.files = [p for p in S.files if p not in abs_set] + S.video_files = [p for p in S.video_files if p not in abs_set] + removed_failed = S.failed_paths & abs_set + if removed_failed: + S.failed_paths -= removed_failed + if removed_failed: + write_failures() + global SEARCH_CACHE SEARCH_CACHE = {"mtime": None, "records": []} load_journal() push_stats() -def delete_photo_file_and_record(path): +def remove_photo_records(path): + remove_photo_records_batch([path]) + +def delete_photo_file_and_record(path, skip_records=False): + """skip_records=True defers journal/cache cleanup to a single batched + call afterwards — used by bulk delete so N files cost one journal pass, + not N. The single-delete endpoint uses the default (clean up right away).""" if not path: return False, "path is required" abs_path = os.path.abspath(path) + with S.lock: + folder = os.path.abspath(S.folder) if S.folder else None + try: + inside_folder = bool(folder) and os.path.commonpath([folder, abs_path]) == folder + except ValueError: + inside_folder = False + if not inside_folder: + return False, f"refusing to delete outside the configured photo folder: {abs_path}" deleted_disk = False if os.path.exists(abs_path): deleted_disk = trash_or_remove_file(abs_path) else: deleted_disk = True - remove_photo_records(abs_path) + if not skip_records: + remove_photo_records(abs_path) return deleted_disk, None SEARCH_CACHE = {"mtime": None, "records": []} @@ -358,32 +423,26 @@ def load_search_index(): if SEARCH_CACHE["mtime"] == mtime: return SEARCH_CACHE["records"] records = [] - with open(JOURNAL, "r", encoding="utf-8") as f: - for line in f: - line = line.strip() - if not line: - continue + for path, rec in read_journal_deduped().items(): + try: + rec["path"] = path + name = os.path.basename(path) + rec["name"] = name + rec["_ext"] = os.path.splitext(name)[1].lower() + rec["_hay"] = " ".join([ + path.lower(), (rec.get("desc") or "").lower(), + (rec.get("category") or "").lower(), name.lower() + ]) try: - rec = json.loads(line) - path = resolve_portable_path(rec.get("path", "")) - rec["path"] = path - name = os.path.basename(path) - rec["name"] = name - rec["_ext"] = os.path.splitext(name)[1].lower() - rec["_hay"] = " ".join([ - path.lower(), (rec.get("desc") or "").lower(), - (rec.get("category") or "").lower(), name.lower() - ]) - try: - st = os.stat(path) - rec["_mtime"] = st.st_mtime - rec["sizeBytes"] = st.st_size - except OSError: - rec["_mtime"] = rec.get("ts", 0) - rec["sizeBytes"] = None - records.append(rec) - except Exception: - pass + st = os.stat(path) + rec["_mtime"] = st.st_mtime + rec["sizeBytes"] = st.st_size + except OSError: + rec["_mtime"] = rec.get("ts", 0) + rec["sizeBytes"] = None + records.append(rec) + except Exception: + pass SEARCH_CACHE = {"mtime": mtime, "records": records} return records @@ -987,6 +1046,8 @@ def process_loop(settings): ocr_model = settings.get("ocrModel") or "glm-ocr:latest" organize = bool(settings.get("organize", False)) integrity = bool(settings.get("integrity", False)) + with S.lock: + organize_folder = S.folder or DEFAULT_FOLDER photo_prompt = build_prompt(length_key, "photo") shot_prompt = build_prompt(length_key, "screenshot") @@ -1034,49 +1095,66 @@ def process_loop(settings): stop_pipeline = threading.Event() consec_fail = 0 + def should_halt(): + return S.stop_evt.is_set() or stop_pipeline.is_set() + + def put_until_halt(q, item): + """Blocking put that keeps re-checking stop flags instead of hanging on + a full queue for up to 60s when the consumer has already exited.""" + while not should_halt(): + try: + q.put(item, timeout=0.5) + return True + except queue.Full: + continue + return False + # Stage 1: Downscaler Thread def downscale_worker(): for i, path in enumerate(pending): - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break while not S.pause_evt.is_set(): - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break time.sleep(0.2) - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break try: t0 = time.time() img_bytes = downscale(path, max_px, tmpdir) b64 = base64.b64encode(img_bytes).decode() - downscale_queue.put((i, path, img_bytes, b64, t0), timeout=60) + if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)): + break except Exception as e: log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}") - downscale_queue.put((i, path, None, str(e), time.time()), timeout=60) - downscale_queue.put(None) + if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())): + break + put_until_halt(downscale_queue, None) # Stage 2: Inference Thread def inference_worker(): while True: - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break while not S.pause_evt.is_set(): - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break time.sleep(0.2) - if S.stop_evt.is_set() or stop_pipeline.is_set(): + if should_halt(): break try: item = downscale_queue.get(timeout=1.0) except queue.Empty: continue if item is None: - write_queue.put(None) + put_until_halt(write_queue, None) break i, path, img_bytes, b64_or_err, t0 = item name = os.path.basename(path) if img_bytes is None: - write_queue.put((i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)) + if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)): + break continue with S.lock: S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total} @@ -1116,9 +1194,9 @@ def process_loop(settings): desc = f"{desc} | text: {words[:300]}" except Exception as e: pass - write_queue.put((i, path, desc, category, route, None, t0)) + put_until_halt(write_queue, (i, path, desc, category, route, None, t0)) except Exception as e: - write_queue.put((i, path, None, None, None, f"Inference error: {e}", t0)) + put_until_halt(write_queue, (i, path, None, None, None, f"Inference error: {e}", t0)) t_downscale = threading.Thread(target=downscale_worker, daemon=True) t_inference = threading.Thread(target=inference_worker, daemon=True) @@ -1162,7 +1240,7 @@ def process_loop(settings): else: write_metadata(path, desc, category, keep_backup, preserve_date) if organize: - organize_alias(settings.get("folder") or DEFAULT_FOLDER, path, category) + organize_alias(organize_folder, path, category) dt = time.time() - t0 with S.lock: @@ -2120,16 +2198,19 @@ class Handler(BaseHTTPRequestHandler): paths = body.get("paths") or [] if not paths: self._json({"error": "paths array is required"}, 400); return - deleted_count = 0 + deleted_paths = [] failed = [] for p in paths: - success, err = delete_photo_file_and_record(p) + success, err = delete_photo_file_and_record(p, skip_records=True) if success: - deleted_count += 1 + deleted_paths.append(p) else: failed.append({"path": p, "error": err}) - log("ok", f"Batch deleted {deleted_count} photos from disk" + (f" ({len(failed)} failed)" if failed else "")) - self._json({"ok": True, "deleted": deleted_count, "failed": failed}) + # One batched journal/cache pass for the whole set, instead of + # one full journal rewrite per file. + remove_photo_records_batch(deleted_paths) + log("ok", f"Batch deleted {len(deleted_paths)} photos from disk" + (f" ({len(failed)} failed)" if failed else "")) + self._json({"ok": True, "deleted": len(deleted_paths), "failed": failed}) elif self.path == "/api/dedupe/build": with S.lock: @@ -2182,6 +2263,9 @@ class Handler(BaseHTTPRequestHandler): self.send_response(404); self.end_headers() def main(): + removed = compact_journal() + if removed: + print(f"journal compacted: removed {removed} duplicate/stale entries from past retags") n = load_journal() load_failures() load_phash_cache()