Add real video tagging via minicpm-v4.6:1b, fix broken video keyword writes

Videos were never part of the automated pipeline before -- only reachable
one at a time via a manual "Tag Now" button, and even that just grabbed
one static frame and ran it through the photo model. Now videos are
first-class:

- New extract_video_frames_b64(): samples up to 6 frames spread across
  the clip's duration and passes them all to the video model in one
  call, so it sees actual motion/progression instead of one snapshot.
  Verified live: two different real test videos got distinct,
  content-aware descriptions that correctly named what was actually
  happening in each, not generic placeholders.
- ollama_generate() now accepts a list of images (photos still pass a
  single one, unchanged) so the same call path serves both.
- process_loop merges S.video_files into the same pending queue as
  photos, routes videos to a separate configurable video model
  (default minicpm-v4.6:1b, a small dedicated vision model) and prompt,
  skipping the photo/screenshot router entirely. redo_single (the
  lightbox "AI Redo" button) updated the same way for consistency.
- S.total_images is now set to the actual combined pending count for
  the run so the progress bar/ETA reflect videos too, not just photos.
- FOUND WHILE TESTING: write_metadata's video branch used "-Keywords"
  for the category/photon-tagged marker, which is a silent no-op on
  QuickTime/.mov files (exiftool has no mapping for it there) --
  confirmed by direct testing. Every video "tagged" before this would
  have gotten a description but never an actual category keyword
  embedded. Switched to XMP-dc:Subject (the same tag family used for
  photos, which QuickTime containers do support via an embedded XMP
  packet) -- verified the category now lands and stays idempotent
  across repeat writes.
- New frontend controls: video model dropdown and a "tag videos too"
  toggle (on by default) in Engine Configurations.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
drjones
2026-07-29 18:28:04 -07:00
parent 3f67efb379
commit fc1e8b9c33
2 changed files with 111 additions and 53 deletions

View File

@@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div> <div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div> <div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div> <div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
<div class="tog"><span>tag videos too</span><div class="sw on" id="swVideo"><i></i></div></div>
<div class="hint">videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.</div>
<label>video model</label>
<select id="videoModel"></select>
</div> </div>
</div> </div>
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
keepBackup: swOn("swBackup"), keepBackup: swOn("swBackup"),
dryRun: swOn("swDry"), dryRun: swOn("swDry"),
organize: swOn("swOrganize"), organize: swOn("swOrganize"),
integrity: swOn("swIntegrity") integrity: swOn("swIntegrity"),
videoModel: $("videoModel").value,
processVideos: swOn("swVideo")
}); });
}; };
@@ -1256,7 +1263,9 @@ function engineSettings(){
keepAlive: $("keepAlive").value, keepAlive: $("keepAlive").value,
keepBackup: swOn("swBackup"), keepBackup: swOn("swBackup"),
preserveDate: swOn("swDate"), preserveDate: swOn("swDate"),
organize: swOn("swOrganize") organize: swOn("swOrganize"),
videoModel: $("videoModel").value,
processVideos: swOn("swVideo")
}; };
} }
@@ -2213,7 +2222,8 @@ function connect(){
const sel=$("model"); const sel=$("model");
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"], [["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{ ["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
const s=$(id); const s=$(id);
visionModels.forEach(m=>{ visionModels.forEach(m=>{
const o=document.createElement("option"); const o=document.createElement("option");
@@ -2225,6 +2235,9 @@ function connect(){
if(!sel.children.length){ if(!sel.children.length){
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"}); addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
} }
if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
}
populateCategoryDropdowns(); populateCategoryDropdowns();
renderCategoryListEditor(); renderCategoryListEditor();

119
server.py
View File

@@ -886,10 +886,14 @@ def model_thinks(model):
return False return False
def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA): def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA):
"""img_b64 is either a single base64 image string (photos) or a list of
them (video: multiple sampled frames in one call, for real multi-frame
understanding instead of treating one static frame as a photo)."""
images = img_b64 if isinstance(img_b64, list) else [img_b64]
body = { body = {
"model": model, "model": model,
"prompt": prompt, "prompt": prompt,
"images": [img_b64], "images": images,
"stream": False, "stream": False,
"options": opts, "options": opts,
"keep_alive": keep_alive, "keep_alive": keep_alive,
@@ -1096,9 +1100,11 @@ def downscale(path, max_px, tmpdir):
with open(out, "rb") as f: with open(out, "rb") as f:
return f.read() return f.read()
def extract_video_frame(path, tmpdir): def extract_video_frames_b64(path, tmpdir, max_frames=6):
"""Grab one representative frame from a video via ffmpeg (read-only). Returns a jpeg file path.""" """Sample up to max_frames frames spread uniformly across the video's
out = os.path.join(tmpdir, "photon_vframe.jpg") duration and return them as base64 strings, for genuine multi-frame
video understanding — the model sees the clip's actual progression,
not one static snapshot mistaken for a photo."""
dur = 3.0 dur = 3.0
try: try:
pr = subprocess.run( pr = subprocess.run(
@@ -1108,12 +1114,21 @@ def extract_video_frame(path, tmpdir):
dur = float(pr.stdout.strip()) dur = float(pr.stdout.strip())
except Exception: except Exception:
pass pass
ts = max(0.5, min(dur * 0.3, max(dur - 0.2, 0.5))) if dur > 1 else 0.1 n = max(2, min(max_frames, int(dur / 1.5) + 1)) if dur > 0 else 2
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", "-q:v", "3", out] frames = []
r = subprocess.run(cmd, capture_output=True, timeout=60) for i in range(n):
if r.returncode != 0 or not os.path.exists(out): ts = (dur * i / n) if n > 1 else dur * 0.3
raise RuntimeError(f"ffmpeg frame extraction failed: {r.stderr.decode(errors='replace')[:200]}") ts = max(0.1, min(ts, max(dur - 0.1, 0.1)))
return out out = os.path.join(tmpdir, f"photon_vframe_{i}.jpg")
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1",
"-vf", "scale=896:-2", "-q:v", "4", out]
r = subprocess.run(cmd, capture_output=True, timeout=30)
if r.returncode == 0 and os.path.exists(out):
with open(out, "rb") as f:
frames.append(base64.b64encode(f.read()).decode())
if not frames:
raise RuntimeError("ffmpeg could not extract any frames from this video")
return frames
def video_thumbnail(path, out_path): def video_thumbnail(path, out_path):
"""Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only).""" """Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only)."""
@@ -1161,6 +1176,10 @@ def build_prompt(length_key, mode="photo"):
task = (f"1. This image is a screenshot or document. In ONE sentence of at most " task = (f"1. This image is a screenshot or document. In ONE sentence of at most "
f"{words} words, say what app/website/document it is and what it shows " f"{words} words, say what app/website/document it is and what it shows "
"(the topic, not the exact words). NEVER copy the text verbatim.\n") "(the topic, not the exact words). NEVER copy the text verbatim.\n")
elif mode == "video":
task = (f"1. These images are frames sampled in order across a short video clip. "
f"Describe what happens in the clip in ONE sentence, at most {words} words, "
"noting the main subject/action and any motion or change across the frames.\n")
else: else:
task = (f"1. Describe this photo in ONE sentence, at most {words} words. " task = (f"1. Describe this photo in ONE sentence, at most {words} words. "
"Mention the main subject, setting, and any clearly readable text.\n") "Mention the main subject, setting, and any clearly readable text.\n")
@@ -1189,15 +1208,18 @@ def write_metadata(path, desc, category, keep_backup, preserve_date, old_categor
if not keep_backup: if not keep_backup:
args.append("-overwrite_original") # writes temp file then atomic rename args.append("-overwrite_original") # writes temp file then atomic rename
if is_video: if is_video:
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic # mp4/mov containers don't carry EXIF/IPTC — "-Keywords" is a silent
# tag names so it resolves to QuickTime/Keys groups automatically. # no-op on QuickTime files (exiftool has no mapping for it there),
# confirmed by testing directly. XMP-dc:Subject is the one that
# actually lands (QuickTime containers support an embedded XMP
# packet) — same tag family used for photos.
if old_category and old_category != category: if old_category and old_category != category:
args.append(f"-Keywords-={old_category}") args.append(f"-XMP-dc:Subject-={old_category}")
args += [ args += [
f"-Keywords-={category}", f"-Keywords-=photon-tagged", f"-XMP-dc:Subject-={category}", "-XMP-dc:Subject-=photon-tagged",
f"-Description={desc}", f"-Description={desc}",
f"-Keywords+={category}", f"-XMP-dc:Subject+={category}",
"-Keywords+=photon-tagged", "-XMP-dc:Subject+=photon-tagged",
] ]
else: else:
if old_category and old_category != category: if old_category and old_category != category:
@@ -1311,14 +1333,19 @@ def process_loop(settings):
ocr_model = settings.get("ocrModel") or "glm-ocr:latest" ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
organize = bool(settings.get("organize", False)) organize = bool(settings.get("organize", False))
integrity = bool(settings.get("integrity", False)) integrity = bool(settings.get("integrity", False))
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
process_videos = bool(settings.get("processVideos", True))
photo_prompt = build_prompt(length_key, "photo") photo_prompt = build_prompt(length_key, "photo")
shot_prompt = build_prompt(length_key, "screenshot") shot_prompt = build_prompt(length_key, "screenshot")
video_prompt = build_prompt(length_key, "video")
num_predict = LENGTH_PRESETS[length_key]["num_predict"] num_predict = LENGTH_PRESETS[length_key]["num_predict"]
opts = {"temperature": temp, "num_predict": num_predict} opts = {"temperature": temp, "num_predict": num_predict}
involved = {model} if not router else {router_model, shot_model, photo_model} involved = {model} if not router else {router_model, shot_model, photo_model}
if ocr_text: if ocr_text:
involved.add(ocr_model) involved.add(ocr_model)
if process_videos:
involved.add(video_model)
think_map = {} think_map = {}
for m in involved: for m in involved:
think_map[m] = False if model_thinks(m) else None think_map[m] = False if model_thinks(m) else None
@@ -1336,7 +1363,7 @@ def process_loop(settings):
log("warn", "DRY RUN — no metadata will be written") log("warn", "DRY RUN — no metadata will be written")
with S.lock: with S.lock:
pending = list(S.files) pending = list(S.files) + (list(S.video_files) if process_videos else [])
idx_offset = 0 idx_offset = 0
if skip_done: if skip_done:
before = len(pending) before = len(pending)
@@ -1345,16 +1372,20 @@ def process_loop(settings):
with S.lock: with S.lock:
S.already_done = idx_offset S.already_done = idx_offset
if idx_offset: if idx_offset:
log("info", f"resume: {idx_offset} photos already in journal, skipping them") log("info", f"resume: {idx_offset} photos/videos already in journal, skipping them")
total = len(pending) + idx_offset total = len(pending) + idx_offset
with S.lock:
S.total_images = total # so the progress bar/ETA reflect videos too, not just images
if total == idx_offset: if total == idx_offset:
log("info", "All photos in this folder are already tagged.") log("info", "Everything in this folder is already tagged.")
with S.lock: with S.lock:
S.status = "done" S.status = "done"
push_stats() push_stats()
broadcast("state", {"status": "done"}) broadcast("state", {"status": "done"})
return return
if process_videos and any(os.path.splitext(p)[1].lower() in VIDEO_EXTS for p in pending):
log("info", f"video model :: {video_model} (multi-frame sampling, up to 6 frames per clip)")
# Multi-threaded Queues # Multi-threaded Queues
downscale_queue = queue.Queue(maxsize=2) downscale_queue = queue.Queue(maxsize=2)
@@ -1387,15 +1418,21 @@ def process_loop(settings):
time.sleep(0.2) time.sleep(0.2)
if should_halt(): if should_halt():
break break
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
try: try:
t0 = time.time() t0 = time.time()
img_bytes = downscale(path, max_px, tmpdir) if is_video:
b64 = base64.b64encode(img_bytes).decode() b64_list = extract_video_frames_b64(path, tmpdir)
if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)): preview_bytes = base64.b64decode(b64_list[0])
else:
preview_bytes = downscale(path, max_px, tmpdir)
b64_list = [base64.b64encode(preview_bytes).decode()]
if not put_until_halt(downscale_queue, (i, path, preview_bytes, b64_list, t0, is_video)):
break break
except Exception as e: except Exception as e:
log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}") kind = "frame extraction" if is_video else "downscaling"
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())): log("error", f"{kind.capitalize()} failed for {os.path.basename(path)} :: {e}")
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time(), is_video)):
break break
put_until_halt(downscale_queue, None) put_until_halt(downscale_queue, None)
@@ -1417,20 +1454,24 @@ def process_loop(settings):
if item is None: if item is None:
put_until_halt(write_queue, None) put_until_halt(write_queue, None)
break break
i, path, img_bytes, b64_or_err, t0 = item i, path, preview_bytes, b64_or_err, t0, is_video = item
name = os.path.basename(path) name = os.path.basename(path)
if img_bytes is None: if preview_bytes is None:
if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)): kind = "Frame extraction" if is_video else "Downscale"
if not put_until_halt(write_queue, (i, path, None, None, None, f"{kind} error: {b64_or_err}", t0)):
break break
continue continue
with S.lock: with S.lock:
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total} S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total, "isVideo": is_video}
S.preview_seq += 1 S.preview_seq += 1
S.preview = (img_bytes, S.preview_seq) S.preview = (preview_bytes, S.preview_seq)
broadcast("photo_start", S.current) broadcast("photo_start", S.current)
broadcast("preview", {"seq": S.preview_seq}) broadcast("preview", {"seq": S.preview_seq})
try: try:
route = None route = None
if is_video:
use_model, use_prompt = video_model, video_prompt
else:
use_model, use_prompt = model, photo_prompt use_model, use_prompt = model, photo_prompt
if router: if router:
r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err, r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
@@ -2425,24 +2466,27 @@ class Handler(BaseHTTPRequestHandler):
ocr_model = settings.get("ocrModel") or "glm-ocr:latest" ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
organize = bool(settings.get("organize", False)) organize = bool(settings.get("organize", False))
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
photo_prompt = build_prompt(length_key, "photo") photo_prompt = build_prompt(length_key, "photo")
shot_prompt = build_prompt(length_key, "screenshot") shot_prompt = build_prompt(length_key, "screenshot")
video_prompt = build_prompt(length_key, "video")
num_predict = LENGTH_PRESETS[length_key]["num_predict"] num_predict = LENGTH_PRESETS[length_key]["num_predict"]
opts = {"temperature": temp, "num_predict": num_predict} opts = {"temperature": temp, "num_predict": num_predict}
think = None
if model_thinks(model):
think = False
tmpdir = tempfile.mkdtemp(prefix="photon_redo_") tmpdir = tempfile.mkdtemp(prefix="photon_redo_")
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
frame_source = extract_video_frame(path, tmpdir) if is_video else path
img = downscale(frame_source, max_px, tmpdir)
b64 = base64.b64encode(img).decode()
route = None route = None
if is_video:
use_model, use_prompt = video_model, video_prompt
b64 = extract_video_frames_b64(path, tmpdir)
else:
use_model, use_prompt = model, photo_prompt use_model, use_prompt = model, photo_prompt
if router: img = downscale(path, max_px, tmpdir)
b64 = base64.b64encode(img).decode()
think = False if model_thinks(use_model) else None
if not is_video and router:
r = ollama_generate(router_model, ROUTER_PROMPT, b64, r = ollama_generate(router_model, ROUTER_PROMPT, b64,
{"temperature": 0, "num_predict": 30}, {"temperature": 0, "num_predict": 30},
keep_alive, think, KIND_SCHEMA) keep_alive, think, KIND_SCHEMA)
@@ -2451,6 +2495,7 @@ class Handler(BaseHTTPRequestHandler):
route = "photo" route = "photo"
use_model = shot_model if route == "screenshot" else photo_model use_model = shot_model if route == "screenshot" else photo_model
use_prompt = shot_prompt if route == "screenshot" else photo_prompt use_prompt = shot_prompt if route == "screenshot" else photo_prompt
think = False if model_thinks(use_model) else None
resp = ollama_generate(use_model, use_prompt, b64, resp = ollama_generate(use_model, use_prompt, b64,
opts, keep_alive, think) opts, keep_alive, think)
raw = resp.get("response", "") raw = resp.get("response", "")