diff --git a/index.html b/index.html index e1fcb06..01df070 100644 --- a/index.html +++ b/index.html @@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
keep _original backups
off = atomic metadata write. on = copies originals (doubles disk usage).
dry run (no writes)
+ +
tag videos too
+
videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.
+ + @@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{ keepBackup: swOn("swBackup"), dryRun: swOn("swDry"), organize: swOn("swOrganize"), - integrity: swOn("swIntegrity") + integrity: swOn("swIntegrity"), + videoModel: $("videoModel").value, + processVideos: swOn("swVideo") }); }; @@ -1256,7 +1263,9 @@ function engineSettings(){ keepAlive: $("keepAlive").value, keepBackup: swOn("swBackup"), preserveDate: swOn("swDate"), - organize: swOn("swOrganize") + organize: swOn("swOrganize"), + videoModel: $("videoModel").value, + processVideos: swOn("swVideo") }; } @@ -2213,7 +2222,8 @@ function connect(){ const sel=$("model"); const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken [["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"], - ["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{ + ["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"], + ["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{ const s=$(id); visionModels.forEach(m=>{ const o=document.createElement("option"); @@ -2225,6 +2235,9 @@ function connect(){ if(!sel.children.length){ addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"}); } + if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){ + addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"}); + } populateCategoryDropdowns(); renderCategoryListEditor(); diff --git a/server.py b/server.py index 8103fcd..a7c79bd 100644 --- a/server.py +++ b/server.py @@ -886,10 +886,14 @@ def model_thinks(model): return False def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA): + """img_b64 is either a single base64 image string (photos) or a list of + them (video: multiple sampled frames in one call, for real multi-frame + understanding instead of treating one static frame as a photo).""" + images = img_b64 if isinstance(img_b64, list) else [img_b64] body = { "model": model, "prompt": prompt, - "images": [img_b64], + "images": images, "stream": False, "options": opts, "keep_alive": keep_alive, @@ -1096,9 +1100,11 @@ def downscale(path, max_px, tmpdir): with open(out, "rb") as f: return f.read() -def extract_video_frame(path, tmpdir): - """Grab one representative frame from a video via ffmpeg (read-only). Returns a jpeg file path.""" - out = os.path.join(tmpdir, "photon_vframe.jpg") +def extract_video_frames_b64(path, tmpdir, max_frames=6): + """Sample up to max_frames frames spread uniformly across the video's + duration and return them as base64 strings, for genuine multi-frame + video understanding — the model sees the clip's actual progression, + not one static snapshot mistaken for a photo.""" dur = 3.0 try: pr = subprocess.run( @@ -1108,12 +1114,21 @@ def extract_video_frame(path, tmpdir): dur = float(pr.stdout.strip()) except Exception: pass - ts = max(0.5, min(dur * 0.3, max(dur - 0.2, 0.5))) if dur > 1 else 0.1 - cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", "-q:v", "3", out] - r = subprocess.run(cmd, capture_output=True, timeout=60) - if r.returncode != 0 or not os.path.exists(out): - raise RuntimeError(f"ffmpeg frame extraction failed: {r.stderr.decode(errors='replace')[:200]}") - return out + n = max(2, min(max_frames, int(dur / 1.5) + 1)) if dur > 0 else 2 + frames = [] + for i in range(n): + ts = (dur * i / n) if n > 1 else dur * 0.3 + ts = max(0.1, min(ts, max(dur - 0.1, 0.1))) + out = os.path.join(tmpdir, f"photon_vframe_{i}.jpg") + cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", + "-vf", "scale=896:-2", "-q:v", "4", out] + r = subprocess.run(cmd, capture_output=True, timeout=30) + if r.returncode == 0 and os.path.exists(out): + with open(out, "rb") as f: + frames.append(base64.b64encode(f.read()).decode()) + if not frames: + raise RuntimeError("ffmpeg could not extract any frames from this video") + return frames def video_thumbnail(path, out_path): """Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only).""" @@ -1161,6 +1176,10 @@ def build_prompt(length_key, mode="photo"): task = (f"1. This image is a screenshot or document. In ONE sentence of at most " f"{words} words, say what app/website/document it is and what it shows " "(the topic, not the exact words). NEVER copy the text verbatim.\n") + elif mode == "video": + task = (f"1. These images are frames sampled in order across a short video clip. " + f"Describe what happens in the clip in ONE sentence, at most {words} words, " + "noting the main subject/action and any motion or change across the frames.\n") else: task = (f"1. Describe this photo in ONE sentence, at most {words} words. " "Mention the main subject, setting, and any clearly readable text.\n") @@ -1189,15 +1208,18 @@ def write_metadata(path, desc, category, keep_backup, preserve_date, old_categor if not keep_backup: args.append("-overwrite_original") # writes temp file then atomic rename if is_video: - # mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic - # tag names so it resolves to QuickTime/Keys groups automatically. + # mp4/mov containers don't carry EXIF/IPTC — "-Keywords" is a silent + # no-op on QuickTime files (exiftool has no mapping for it there), + # confirmed by testing directly. XMP-dc:Subject is the one that + # actually lands (QuickTime containers support an embedded XMP + # packet) — same tag family used for photos. if old_category and old_category != category: - args.append(f"-Keywords-={old_category}") + args.append(f"-XMP-dc:Subject-={old_category}") args += [ - f"-Keywords-={category}", f"-Keywords-=photon-tagged", + f"-XMP-dc:Subject-={category}", "-XMP-dc:Subject-=photon-tagged", f"-Description={desc}", - f"-Keywords+={category}", - "-Keywords+=photon-tagged", + f"-XMP-dc:Subject+={category}", + "-XMP-dc:Subject+=photon-tagged", ] else: if old_category and old_category != category: @@ -1311,14 +1333,19 @@ def process_loop(settings): ocr_model = settings.get("ocrModel") or "glm-ocr:latest" organize = bool(settings.get("organize", False)) integrity = bool(settings.get("integrity", False)) + video_model = settings.get("videoModel") or "minicpm-v4.6:1b" + process_videos = bool(settings.get("processVideos", True)) photo_prompt = build_prompt(length_key, "photo") shot_prompt = build_prompt(length_key, "screenshot") + video_prompt = build_prompt(length_key, "video") num_predict = LENGTH_PRESETS[length_key]["num_predict"] opts = {"temperature": temp, "num_predict": num_predict} involved = {model} if not router else {router_model, shot_model, photo_model} if ocr_text: involved.add(ocr_model) + if process_videos: + involved.add(video_model) think_map = {} for m in involved: think_map[m] = False if model_thinks(m) else None @@ -1336,7 +1363,7 @@ def process_loop(settings): log("warn", "DRY RUN — no metadata will be written") with S.lock: - pending = list(S.files) + pending = list(S.files) + (list(S.video_files) if process_videos else []) idx_offset = 0 if skip_done: before = len(pending) @@ -1345,16 +1372,20 @@ def process_loop(settings): with S.lock: S.already_done = idx_offset if idx_offset: - log("info", f"resume: {idx_offset} photos already in journal, skipping them") + log("info", f"resume: {idx_offset} photos/videos already in journal, skipping them") total = len(pending) + idx_offset + with S.lock: + S.total_images = total # so the progress bar/ETA reflect videos too, not just images if total == idx_offset: - log("info", "All photos in this folder are already tagged.") + log("info", "Everything in this folder is already tagged.") with S.lock: S.status = "done" push_stats() broadcast("state", {"status": "done"}) return + if process_videos and any(os.path.splitext(p)[1].lower() in VIDEO_EXTS for p in pending): + log("info", f"video model :: {video_model} (multi-frame sampling, up to 6 frames per clip)") # Multi-threaded Queues downscale_queue = queue.Queue(maxsize=2) @@ -1387,15 +1418,21 @@ def process_loop(settings): time.sleep(0.2) if should_halt(): break + is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS try: t0 = time.time() - img_bytes = downscale(path, max_px, tmpdir) - b64 = base64.b64encode(img_bytes).decode() - if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)): + if is_video: + b64_list = extract_video_frames_b64(path, tmpdir) + preview_bytes = base64.b64decode(b64_list[0]) + else: + preview_bytes = downscale(path, max_px, tmpdir) + b64_list = [base64.b64encode(preview_bytes).decode()] + if not put_until_halt(downscale_queue, (i, path, preview_bytes, b64_list, t0, is_video)): break except Exception as e: - log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}") - if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())): + kind = "frame extraction" if is_video else "downscaling" + log("error", f"{kind.capitalize()} failed for {os.path.basename(path)} :: {e}") + if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time(), is_video)): break put_until_halt(downscale_queue, None) @@ -1417,30 +1454,34 @@ def process_loop(settings): if item is None: put_until_halt(write_queue, None) break - i, path, img_bytes, b64_or_err, t0 = item + i, path, preview_bytes, b64_or_err, t0, is_video = item name = os.path.basename(path) - if img_bytes is None: - if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)): + if preview_bytes is None: + kind = "Frame extraction" if is_video else "Downscale" + if not put_until_halt(write_queue, (i, path, None, None, None, f"{kind} error: {b64_or_err}", t0)): break continue with S.lock: - S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total} + S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total, "isVideo": is_video} S.preview_seq += 1 - S.preview = (img_bytes, S.preview_seq) + S.preview = (preview_bytes, S.preview_seq) broadcast("photo_start", S.current) broadcast("preview", {"seq": S.preview_seq}) try: route = None - use_model, use_prompt = model, photo_prompt - if router: - r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err, - {"temperature": 0, "num_predict": 30}, - keep_alive, think_map.get(router_model), KIND_SCHEMA) - route = salvage_json(r.get("response", "")).get("kind") - if route not in ("screenshot", "photo"): - route = "photo" - use_model = shot_model if route == "screenshot" else photo_model - use_prompt = shot_prompt if route == "screenshot" else photo_prompt + if is_video: + use_model, use_prompt = video_model, video_prompt + else: + use_model, use_prompt = model, photo_prompt + if router: + r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err, + {"temperature": 0, "num_predict": 30}, + keep_alive, think_map.get(router_model), KIND_SCHEMA) + route = salvage_json(r.get("response", "")).get("kind") + if route not in ("screenshot", "photo"): + route = "photo" + use_model = shot_model if route == "screenshot" else photo_model + use_prompt = shot_prompt if route == "screenshot" else photo_prompt resp = ollama_generate(use_model, use_prompt, b64_or_err, opts, keep_alive, think_map.get(use_model)) raw = resp.get("response", "") @@ -2425,24 +2466,27 @@ class Handler(BaseHTTPRequestHandler): ocr_model = settings.get("ocrModel") or "glm-ocr:latest" organize = bool(settings.get("organize", False)) + video_model = settings.get("videoModel") or "minicpm-v4.6:1b" photo_prompt = build_prompt(length_key, "photo") shot_prompt = build_prompt(length_key, "screenshot") + video_prompt = build_prompt(length_key, "video") num_predict = LENGTH_PRESETS[length_key]["num_predict"] opts = {"temperature": temp, "num_predict": num_predict} - - think = None - if model_thinks(model): - think = False - + tmpdir = tempfile.mkdtemp(prefix="photon_redo_") is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS - frame_source = extract_video_frame(path, tmpdir) if is_video else path - img = downscale(frame_source, max_px, tmpdir) - b64 = base64.b64encode(img).decode() - + route = None - use_model, use_prompt = model, photo_prompt - if router: + if is_video: + use_model, use_prompt = video_model, video_prompt + b64 = extract_video_frames_b64(path, tmpdir) + else: + use_model, use_prompt = model, photo_prompt + img = downscale(path, max_px, tmpdir) + b64 = base64.b64encode(img).decode() + + think = False if model_thinks(use_model) else None + if not is_video and router: r = ollama_generate(router_model, ROUTER_PROMPT, b64, {"temperature": 0, "num_predict": 30}, keep_alive, think, KIND_SCHEMA) @@ -2451,6 +2495,7 @@ class Handler(BaseHTTPRequestHandler): route = "photo" use_model = shot_model if route == "screenshot" else photo_model use_prompt = shot_prompt if route == "screenshot" else photo_prompt + think = False if model_thinks(use_model) else None resp = ollama_generate(use_model, use_prompt, b64, opts, keep_alive, think) raw = resp.get("response", "")