Add real video tagging via minicpm-v4.6:1b, fix broken video keyword writes
Videos were never part of the automated pipeline before -- only reachable one at a time via a manual "Tag Now" button, and even that just grabbed one static frame and ran it through the photo model. Now videos are first-class: - New extract_video_frames_b64(): samples up to 6 frames spread across the clip's duration and passes them all to the video model in one call, so it sees actual motion/progression instead of one snapshot. Verified live: two different real test videos got distinct, content-aware descriptions that correctly named what was actually happening in each, not generic placeholders. - ollama_generate() now accepts a list of images (photos still pass a single one, unchanged) so the same call path serves both. - process_loop merges S.video_files into the same pending queue as photos, routes videos to a separate configurable video model (default minicpm-v4.6:1b, a small dedicated vision model) and prompt, skipping the photo/screenshot router entirely. redo_single (the lightbox "AI Redo" button) updated the same way for consistency. - S.total_images is now set to the actual combined pending count for the run so the progress bar/ETA reflect videos too, not just photos. - FOUND WHILE TESTING: write_metadata's video branch used "-Keywords" for the category/photon-tagged marker, which is a silent no-op on QuickTime/.mov files (exiftool has no mapping for it there) -- confirmed by direct testing. Every video "tagged" before this would have gotten a description but never an actual category keyword embedded. Switched to XMP-dc:Subject (the same tag family used for photos, which QuickTime containers do support via an embedded XMP packet) -- verified the category now lands and stays idempotent across repeat writes. - New frontend controls: video model dropdown and a "tag videos too" toggle (on by default) in Engine Configurations. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
19
index.html
19
index.html
@@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
|
||||
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
|
||||
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
|
||||
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
|
||||
|
||||
<div class="tog"><span>tag videos too</span><div class="sw on" id="swVideo"><i></i></div></div>
|
||||
<div class="hint">videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.</div>
|
||||
<label>video model</label>
|
||||
<select id="videoModel"></select>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
|
||||
keepBackup: swOn("swBackup"),
|
||||
dryRun: swOn("swDry"),
|
||||
organize: swOn("swOrganize"),
|
||||
integrity: swOn("swIntegrity")
|
||||
integrity: swOn("swIntegrity"),
|
||||
videoModel: $("videoModel").value,
|
||||
processVideos: swOn("swVideo")
|
||||
});
|
||||
};
|
||||
|
||||
@@ -1256,7 +1263,9 @@ function engineSettings(){
|
||||
keepAlive: $("keepAlive").value,
|
||||
keepBackup: swOn("swBackup"),
|
||||
preserveDate: swOn("swDate"),
|
||||
organize: swOn("swOrganize")
|
||||
organize: swOn("swOrganize"),
|
||||
videoModel: $("videoModel").value,
|
||||
processVideos: swOn("swVideo")
|
||||
};
|
||||
}
|
||||
|
||||
@@ -2213,7 +2222,8 @@ function connect(){
|
||||
const sel=$("model");
|
||||
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
|
||||
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
|
||||
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{
|
||||
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
|
||||
["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
|
||||
const s=$(id);
|
||||
visionModels.forEach(m=>{
|
||||
const o=document.createElement("option");
|
||||
@@ -2225,6 +2235,9 @@ function connect(){
|
||||
if(!sel.children.length){
|
||||
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
|
||||
}
|
||||
if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
|
||||
addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
|
||||
}
|
||||
|
||||
populateCategoryDropdowns();
|
||||
renderCategoryListEditor();
|
||||
|
||||
Reference in New Issue
Block a user