off = atomic metadata write. on = copies originals (doubles disk usage).
dry run (no writes)
+
+
tag videos too
+
videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.
+
+
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
keepBackup: swOn("swBackup"),
dryRun: swOn("swDry"),
organize: swOn("swOrganize"),
- integrity: swOn("swIntegrity")
+ integrity: swOn("swIntegrity"),
+ videoModel: $("videoModel").value,
+ processVideos: swOn("swVideo")
});
};
@@ -1256,7 +1263,9 @@ function engineSettings(){
keepAlive: $("keepAlive").value,
keepBackup: swOn("swBackup"),
preserveDate: swOn("swDate"),
- organize: swOn("swOrganize")
+ organize: swOn("swOrganize"),
+ videoModel: $("videoModel").value,
+ processVideos: swOn("swVideo")
};
}
@@ -2213,7 +2222,8 @@ function connect(){
const sel=$("model");
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
- ["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{
+ ["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
+ ["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
const s=$(id);
visionModels.forEach(m=>{
const o=document.createElement("option");
@@ -2225,6 +2235,9 @@ function connect(){
if(!sel.children.length){
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
}
+ if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
+ addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
+ }
populateCategoryDropdowns();
renderCategoryListEditor();
diff --git a/server.py b/server.py
index 8103fcd..a7c79bd 100644
--- a/server.py
+++ b/server.py
@@ -886,10 +886,14 @@ def model_thinks(model):
return False
def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA):
+ """img_b64 is either a single base64 image string (photos) or a list of
+ them (video: multiple sampled frames in one call, for real multi-frame
+ understanding instead of treating one static frame as a photo)."""
+ images = img_b64 if isinstance(img_b64, list) else [img_b64]
body = {
"model": model,
"prompt": prompt,
- "images": [img_b64],
+ "images": images,
"stream": False,
"options": opts,
"keep_alive": keep_alive,
@@ -1096,9 +1100,11 @@ def downscale(path, max_px, tmpdir):
with open(out, "rb") as f:
return f.read()
-def extract_video_frame(path, tmpdir):
- """Grab one representative frame from a video via ffmpeg (read-only). Returns a jpeg file path."""
- out = os.path.join(tmpdir, "photon_vframe.jpg")
+def extract_video_frames_b64(path, tmpdir, max_frames=6):
+ """Sample up to max_frames frames spread uniformly across the video's
+ duration and return them as base64 strings, for genuine multi-frame
+ video understanding — the model sees the clip's actual progression,
+ not one static snapshot mistaken for a photo."""
dur = 3.0
try:
pr = subprocess.run(
@@ -1108,12 +1114,21 @@ def extract_video_frame(path, tmpdir):
dur = float(pr.stdout.strip())
except Exception:
pass
- ts = max(0.5, min(dur * 0.3, max(dur - 0.2, 0.5))) if dur > 1 else 0.1
- cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", "-q:v", "3", out]
- r = subprocess.run(cmd, capture_output=True, timeout=60)
- if r.returncode != 0 or not os.path.exists(out):
- raise RuntimeError(f"ffmpeg frame extraction failed: {r.stderr.decode(errors='replace')[:200]}")
- return out
+ n = max(2, min(max_frames, int(dur / 1.5) + 1)) if dur > 0 else 2
+ frames = []
+ for i in range(n):
+ ts = (dur * i / n) if n > 1 else dur * 0.3
+ ts = max(0.1, min(ts, max(dur - 0.1, 0.1)))
+ out = os.path.join(tmpdir, f"photon_vframe_{i}.jpg")
+ cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1",
+ "-vf", "scale=896:-2", "-q:v", "4", out]
+ r = subprocess.run(cmd, capture_output=True, timeout=30)
+ if r.returncode == 0 and os.path.exists(out):
+ with open(out, "rb") as f:
+ frames.append(base64.b64encode(f.read()).decode())
+ if not frames:
+ raise RuntimeError("ffmpeg could not extract any frames from this video")
+ return frames
def video_thumbnail(path, out_path):
"""Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only)."""
@@ -1161,6 +1176,10 @@ def build_prompt(length_key, mode="photo"):
task = (f"1. This image is a screenshot or document. In ONE sentence of at most "
f"{words} words, say what app/website/document it is and what it shows "
"(the topic, not the exact words). NEVER copy the text verbatim.\n")
+ elif mode == "video":
+ task = (f"1. These images are frames sampled in order across a short video clip. "
+ f"Describe what happens in the clip in ONE sentence, at most {words} words, "
+ "noting the main subject/action and any motion or change across the frames.\n")
else:
task = (f"1. Describe this photo in ONE sentence, at most {words} words. "
"Mention the main subject, setting, and any clearly readable text.\n")
@@ -1189,15 +1208,18 @@ def write_metadata(path, desc, category, keep_backup, preserve_date, old_categor
if not keep_backup:
args.append("-overwrite_original") # writes temp file then atomic rename
if is_video:
- # mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
- # tag names so it resolves to QuickTime/Keys groups automatically.
+ # mp4/mov containers don't carry EXIF/IPTC — "-Keywords" is a silent
+ # no-op on QuickTime files (exiftool has no mapping for it there),
+ # confirmed by testing directly. XMP-dc:Subject is the one that
+ # actually lands (QuickTime containers support an embedded XMP
+ # packet) — same tag family used for photos.
if old_category and old_category != category:
- args.append(f"-Keywords-={old_category}")
+ args.append(f"-XMP-dc:Subject-={old_category}")
args += [
- f"-Keywords-={category}", f"-Keywords-=photon-tagged",
+ f"-XMP-dc:Subject-={category}", "-XMP-dc:Subject-=photon-tagged",
f"-Description={desc}",
- f"-Keywords+={category}",
- "-Keywords+=photon-tagged",
+ f"-XMP-dc:Subject+={category}",
+ "-XMP-dc:Subject+=photon-tagged",
]
else:
if old_category and old_category != category:
@@ -1311,14 +1333,19 @@ def process_loop(settings):
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
organize = bool(settings.get("organize", False))
integrity = bool(settings.get("integrity", False))
+ video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
+ process_videos = bool(settings.get("processVideos", True))
photo_prompt = build_prompt(length_key, "photo")
shot_prompt = build_prompt(length_key, "screenshot")
+ video_prompt = build_prompt(length_key, "video")
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
opts = {"temperature": temp, "num_predict": num_predict}
involved = {model} if not router else {router_model, shot_model, photo_model}
if ocr_text:
involved.add(ocr_model)
+ if process_videos:
+ involved.add(video_model)
think_map = {}
for m in involved:
think_map[m] = False if model_thinks(m) else None
@@ -1336,7 +1363,7 @@ def process_loop(settings):
log("warn", "DRY RUN — no metadata will be written")
with S.lock:
- pending = list(S.files)
+ pending = list(S.files) + (list(S.video_files) if process_videos else [])
idx_offset = 0
if skip_done:
before = len(pending)
@@ -1345,16 +1372,20 @@ def process_loop(settings):
with S.lock:
S.already_done = idx_offset
if idx_offset:
- log("info", f"resume: {idx_offset} photos already in journal, skipping them")
+ log("info", f"resume: {idx_offset} photos/videos already in journal, skipping them")
total = len(pending) + idx_offset
+ with S.lock:
+ S.total_images = total # so the progress bar/ETA reflect videos too, not just images
if total == idx_offset:
- log("info", "All photos in this folder are already tagged.")
+ log("info", "Everything in this folder is already tagged.")
with S.lock:
S.status = "done"
push_stats()
broadcast("state", {"status": "done"})
return
+ if process_videos and any(os.path.splitext(p)[1].lower() in VIDEO_EXTS for p in pending):
+ log("info", f"video model :: {video_model} (multi-frame sampling, up to 6 frames per clip)")
# Multi-threaded Queues
downscale_queue = queue.Queue(maxsize=2)
@@ -1387,15 +1418,21 @@ def process_loop(settings):
time.sleep(0.2)
if should_halt():
break
+ is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
try:
t0 = time.time()
- img_bytes = downscale(path, max_px, tmpdir)
- b64 = base64.b64encode(img_bytes).decode()
- if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)):
+ if is_video:
+ b64_list = extract_video_frames_b64(path, tmpdir)
+ preview_bytes = base64.b64decode(b64_list[0])
+ else:
+ preview_bytes = downscale(path, max_px, tmpdir)
+ b64_list = [base64.b64encode(preview_bytes).decode()]
+ if not put_until_halt(downscale_queue, (i, path, preview_bytes, b64_list, t0, is_video)):
break
except Exception as e:
- log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}")
- if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())):
+ kind = "frame extraction" if is_video else "downscaling"
+ log("error", f"{kind.capitalize()} failed for {os.path.basename(path)} :: {e}")
+ if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time(), is_video)):
break
put_until_halt(downscale_queue, None)
@@ -1417,30 +1454,34 @@ def process_loop(settings):
if item is None:
put_until_halt(write_queue, None)
break
- i, path, img_bytes, b64_or_err, t0 = item
+ i, path, preview_bytes, b64_or_err, t0, is_video = item
name = os.path.basename(path)
- if img_bytes is None:
- if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)):
+ if preview_bytes is None:
+ kind = "Frame extraction" if is_video else "Downscale"
+ if not put_until_halt(write_queue, (i, path, None, None, None, f"{kind} error: {b64_or_err}", t0)):
break
continue
with S.lock:
- S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total}
+ S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total, "isVideo": is_video}
S.preview_seq += 1
- S.preview = (img_bytes, S.preview_seq)
+ S.preview = (preview_bytes, S.preview_seq)
broadcast("photo_start", S.current)
broadcast("preview", {"seq": S.preview_seq})
try:
route = None
- use_model, use_prompt = model, photo_prompt
- if router:
- r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
- {"temperature": 0, "num_predict": 30},
- keep_alive, think_map.get(router_model), KIND_SCHEMA)
- route = salvage_json(r.get("response", "")).get("kind")
- if route not in ("screenshot", "photo"):
- route = "photo"
- use_model = shot_model if route == "screenshot" else photo_model
- use_prompt = shot_prompt if route == "screenshot" else photo_prompt
+ if is_video:
+ use_model, use_prompt = video_model, video_prompt
+ else:
+ use_model, use_prompt = model, photo_prompt
+ if router:
+ r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
+ {"temperature": 0, "num_predict": 30},
+ keep_alive, think_map.get(router_model), KIND_SCHEMA)
+ route = salvage_json(r.get("response", "")).get("kind")
+ if route not in ("screenshot", "photo"):
+ route = "photo"
+ use_model = shot_model if route == "screenshot" else photo_model
+ use_prompt = shot_prompt if route == "screenshot" else photo_prompt
resp = ollama_generate(use_model, use_prompt, b64_or_err,
opts, keep_alive, think_map.get(use_model))
raw = resp.get("response", "")
@@ -2425,24 +2466,27 @@ class Handler(BaseHTTPRequestHandler):
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
organize = bool(settings.get("organize", False))
+ video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
photo_prompt = build_prompt(length_key, "photo")
shot_prompt = build_prompt(length_key, "screenshot")
+ video_prompt = build_prompt(length_key, "video")
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
opts = {"temperature": temp, "num_predict": num_predict}
-
- think = None
- if model_thinks(model):
- think = False
-
+
tmpdir = tempfile.mkdtemp(prefix="photon_redo_")
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
- frame_source = extract_video_frame(path, tmpdir) if is_video else path
- img = downscale(frame_source, max_px, tmpdir)
- b64 = base64.b64encode(img).decode()
-
+
route = None
- use_model, use_prompt = model, photo_prompt
- if router:
+ if is_video:
+ use_model, use_prompt = video_model, video_prompt
+ b64 = extract_video_frames_b64(path, tmpdir)
+ else:
+ use_model, use_prompt = model, photo_prompt
+ img = downscale(path, max_px, tmpdir)
+ b64 = base64.b64encode(img).decode()
+
+ think = False if model_thinks(use_model) else None
+ if not is_video and router:
r = ollama_generate(router_model, ROUTER_PROMPT, b64,
{"temperature": 0, "num_predict": 30},
keep_alive, think, KIND_SCHEMA)
@@ -2451,6 +2495,7 @@ class Handler(BaseHTTPRequestHandler):
route = "photo"
use_model = shot_model if route == "screenshot" else photo_model
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
+ think = False if model_thinks(use_model) else None
resp = ollama_generate(use_model, use_prompt, b64,
opts, keep_alive, think)
raw = resp.get("response", "")