Add real video tagging via minicpm-v4.6:1b, fix broken video keyword writes

Videos were never part of the automated pipeline before -- only reachable
one at a time via a manual "Tag Now" button, and even that just grabbed
one static frame and ran it through the photo model. Now videos are
first-class:

- New extract_video_frames_b64(): samples up to 6 frames spread across
  the clip's duration and passes them all to the video model in one
  call, so it sees actual motion/progression instead of one snapshot.
  Verified live: two different real test videos got distinct,
  content-aware descriptions that correctly named what was actually
  happening in each, not generic placeholders.
- ollama_generate() now accepts a list of images (photos still pass a
  single one, unchanged) so the same call path serves both.
- process_loop merges S.video_files into the same pending queue as
  photos, routes videos to a separate configurable video model
  (default minicpm-v4.6:1b, a small dedicated vision model) and prompt,
  skipping the photo/screenshot router entirely. redo_single (the
  lightbox "AI Redo" button) updated the same way for consistency.
- S.total_images is now set to the actual combined pending count for
  the run so the progress bar/ETA reflect videos too, not just photos.
- FOUND WHILE TESTING: write_metadata's video branch used "-Keywords"
  for the category/photon-tagged marker, which is a silent no-op on
  QuickTime/.mov files (exiftool has no mapping for it there) --
  confirmed by direct testing. Every video "tagged" before this would
  have gotten a description but never an actual category keyword
  embedded. Switched to XMP-dc:Subject (the same tag family used for
  photos, which QuickTime containers do support via an embedded XMP
  packet) -- verified the category now lands and stays idempotent
  across repeat writes.
- New frontend controls: video model dropdown and a "tag videos too"
  toggle (on by default) in Engine Configurations.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-07-29 18:28:04 -07:00
parent 3f67efb379
commit fc1e8b9c33
2 changed files with 111 additions and 53 deletions

View File

@@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
<div class="tog"><span>tag videos too</span><div class="sw on" id="swVideo"><i></i></div></div>
<div class="hint">videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.</div>
<label>video model</label>
<select id="videoModel"></select>
</div>
</div>
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
keepBackup: swOn("swBackup"),
dryRun: swOn("swDry"),
organize: swOn("swOrganize"),
integrity: swOn("swIntegrity")
integrity: swOn("swIntegrity"),
videoModel: $("videoModel").value,
processVideos: swOn("swVideo")
});
};
@@ -1256,7 +1263,9 @@ function engineSettings(){
keepAlive: $("keepAlive").value,
keepBackup: swOn("swBackup"),
preserveDate: swOn("swDate"),
organize: swOn("swOrganize")
organize: swOn("swOrganize"),
videoModel: $("videoModel").value,
processVideos: swOn("swVideo")
};
}
@@ -2213,7 +2222,8 @@ function connect(){
const sel=$("model");
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
const s=$(id);
visionModels.forEach(m=>{
const o=document.createElement("option");
@@ -2225,6 +2235,9 @@ function connect(){
if(!sel.children.length){
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
}
if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
}
populateCategoryDropdowns();
renderCategoryListEditor();