Add real video tagging via minicpm-v4.6:1b, fix broken video keyword writes
Videos were never part of the automated pipeline before -- only reachable one at a time via a manual "Tag Now" button, and even that just grabbed one static frame and ran it through the photo model. Now videos are first-class: - New extract_video_frames_b64(): samples up to 6 frames spread across the clip's duration and passes them all to the video model in one call, so it sees actual motion/progression instead of one snapshot. Verified live: two different real test videos got distinct, content-aware descriptions that correctly named what was actually happening in each, not generic placeholders. - ollama_generate() now accepts a list of images (photos still pass a single one, unchanged) so the same call path serves both. - process_loop merges S.video_files into the same pending queue as photos, routes videos to a separate configurable video model (default minicpm-v4.6:1b, a small dedicated vision model) and prompt, skipping the photo/screenshot router entirely. redo_single (the lightbox "AI Redo" button) updated the same way for consistency. - S.total_images is now set to the actual combined pending count for the run so the progress bar/ETA reflect videos too, not just photos. - FOUND WHILE TESTING: write_metadata's video branch used "-Keywords" for the category/photon-tagged marker, which is a silent no-op on QuickTime/.mov files (exiftool has no mapping for it there) -- confirmed by direct testing. Every video "tagged" before this would have gotten a description but never an actual category keyword embedded. Switched to XMP-dc:Subject (the same tag family used for photos, which QuickTime containers do support via an embedded XMP packet) -- verified the category now lands and stays idempotent across repeat writes. - New frontend controls: video model dropdown and a "tag videos too" toggle (on by default) in Engine Configurations. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
19
index.html
19
index.html
@@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
|
||||
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
|
||||
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
|
||||
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
|
||||
|
||||
<div class="tog"><span>tag videos too</span><div class="sw on" id="swVideo"><i></i></div></div>
|
||||
<div class="hint">videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.</div>
|
||||
<label>video model</label>
|
||||
<select id="videoModel"></select>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
|
||||
keepBackup: swOn("swBackup"),
|
||||
dryRun: swOn("swDry"),
|
||||
organize: swOn("swOrganize"),
|
||||
integrity: swOn("swIntegrity")
|
||||
integrity: swOn("swIntegrity"),
|
||||
videoModel: $("videoModel").value,
|
||||
processVideos: swOn("swVideo")
|
||||
});
|
||||
};
|
||||
|
||||
@@ -1256,7 +1263,9 @@ function engineSettings(){
|
||||
keepAlive: $("keepAlive").value,
|
||||
keepBackup: swOn("swBackup"),
|
||||
preserveDate: swOn("swDate"),
|
||||
organize: swOn("swOrganize")
|
||||
organize: swOn("swOrganize"),
|
||||
videoModel: $("videoModel").value,
|
||||
processVideos: swOn("swVideo")
|
||||
};
|
||||
}
|
||||
|
||||
@@ -2213,7 +2222,8 @@ function connect(){
|
||||
const sel=$("model");
|
||||
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
|
||||
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
|
||||
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{
|
||||
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
|
||||
["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
|
||||
const s=$(id);
|
||||
visionModels.forEach(m=>{
|
||||
const o=document.createElement("option");
|
||||
@@ -2225,6 +2235,9 @@ function connect(){
|
||||
if(!sel.children.length){
|
||||
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
|
||||
}
|
||||
if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
|
||||
addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
|
||||
}
|
||||
|
||||
populateCategoryDropdowns();
|
||||
renderCategoryListEditor();
|
||||
|
||||
145
server.py
145
server.py
@@ -886,10 +886,14 @@ def model_thinks(model):
|
||||
return False
|
||||
|
||||
def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA):
|
||||
"""img_b64 is either a single base64 image string (photos) or a list of
|
||||
them (video: multiple sampled frames in one call, for real multi-frame
|
||||
understanding instead of treating one static frame as a photo)."""
|
||||
images = img_b64 if isinstance(img_b64, list) else [img_b64]
|
||||
body = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"images": [img_b64],
|
||||
"images": images,
|
||||
"stream": False,
|
||||
"options": opts,
|
||||
"keep_alive": keep_alive,
|
||||
@@ -1096,9 +1100,11 @@ def downscale(path, max_px, tmpdir):
|
||||
with open(out, "rb") as f:
|
||||
return f.read()
|
||||
|
||||
def extract_video_frame(path, tmpdir):
|
||||
"""Grab one representative frame from a video via ffmpeg (read-only). Returns a jpeg file path."""
|
||||
out = os.path.join(tmpdir, "photon_vframe.jpg")
|
||||
def extract_video_frames_b64(path, tmpdir, max_frames=6):
|
||||
"""Sample up to max_frames frames spread uniformly across the video's
|
||||
duration and return them as base64 strings, for genuine multi-frame
|
||||
video understanding — the model sees the clip's actual progression,
|
||||
not one static snapshot mistaken for a photo."""
|
||||
dur = 3.0
|
||||
try:
|
||||
pr = subprocess.run(
|
||||
@@ -1108,12 +1114,21 @@ def extract_video_frame(path, tmpdir):
|
||||
dur = float(pr.stdout.strip())
|
||||
except Exception:
|
||||
pass
|
||||
ts = max(0.5, min(dur * 0.3, max(dur - 0.2, 0.5))) if dur > 1 else 0.1
|
||||
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", "-q:v", "3", out]
|
||||
r = subprocess.run(cmd, capture_output=True, timeout=60)
|
||||
if r.returncode != 0 or not os.path.exists(out):
|
||||
raise RuntimeError(f"ffmpeg frame extraction failed: {r.stderr.decode(errors='replace')[:200]}")
|
||||
return out
|
||||
n = max(2, min(max_frames, int(dur / 1.5) + 1)) if dur > 0 else 2
|
||||
frames = []
|
||||
for i in range(n):
|
||||
ts = (dur * i / n) if n > 1 else dur * 0.3
|
||||
ts = max(0.1, min(ts, max(dur - 0.1, 0.1)))
|
||||
out = os.path.join(tmpdir, f"photon_vframe_{i}.jpg")
|
||||
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1",
|
||||
"-vf", "scale=896:-2", "-q:v", "4", out]
|
||||
r = subprocess.run(cmd, capture_output=True, timeout=30)
|
||||
if r.returncode == 0 and os.path.exists(out):
|
||||
with open(out, "rb") as f:
|
||||
frames.append(base64.b64encode(f.read()).decode())
|
||||
if not frames:
|
||||
raise RuntimeError("ffmpeg could not extract any frames from this video")
|
||||
return frames
|
||||
|
||||
def video_thumbnail(path, out_path):
|
||||
"""Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only)."""
|
||||
@@ -1161,6 +1176,10 @@ def build_prompt(length_key, mode="photo"):
|
||||
task = (f"1. This image is a screenshot or document. In ONE sentence of at most "
|
||||
f"{words} words, say what app/website/document it is and what it shows "
|
||||
"(the topic, not the exact words). NEVER copy the text verbatim.\n")
|
||||
elif mode == "video":
|
||||
task = (f"1. These images are frames sampled in order across a short video clip. "
|
||||
f"Describe what happens in the clip in ONE sentence, at most {words} words, "
|
||||
"noting the main subject/action and any motion or change across the frames.\n")
|
||||
else:
|
||||
task = (f"1. Describe this photo in ONE sentence, at most {words} words. "
|
||||
"Mention the main subject, setting, and any clearly readable text.\n")
|
||||
@@ -1189,15 +1208,18 @@ def write_metadata(path, desc, category, keep_backup, preserve_date, old_categor
|
||||
if not keep_backup:
|
||||
args.append("-overwrite_original") # writes temp file then atomic rename
|
||||
if is_video:
|
||||
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
|
||||
# tag names so it resolves to QuickTime/Keys groups automatically.
|
||||
# mp4/mov containers don't carry EXIF/IPTC — "-Keywords" is a silent
|
||||
# no-op on QuickTime files (exiftool has no mapping for it there),
|
||||
# confirmed by testing directly. XMP-dc:Subject is the one that
|
||||
# actually lands (QuickTime containers support an embedded XMP
|
||||
# packet) — same tag family used for photos.
|
||||
if old_category and old_category != category:
|
||||
args.append(f"-Keywords-={old_category}")
|
||||
args.append(f"-XMP-dc:Subject-={old_category}")
|
||||
args += [
|
||||
f"-Keywords-={category}", f"-Keywords-=photon-tagged",
|
||||
f"-XMP-dc:Subject-={category}", "-XMP-dc:Subject-=photon-tagged",
|
||||
f"-Description={desc}",
|
||||
f"-Keywords+={category}",
|
||||
"-Keywords+=photon-tagged",
|
||||
f"-XMP-dc:Subject+={category}",
|
||||
"-XMP-dc:Subject+=photon-tagged",
|
||||
]
|
||||
else:
|
||||
if old_category and old_category != category:
|
||||
@@ -1311,14 +1333,19 @@ def process_loop(settings):
|
||||
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
||||
organize = bool(settings.get("organize", False))
|
||||
integrity = bool(settings.get("integrity", False))
|
||||
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
|
||||
process_videos = bool(settings.get("processVideos", True))
|
||||
|
||||
photo_prompt = build_prompt(length_key, "photo")
|
||||
shot_prompt = build_prompt(length_key, "screenshot")
|
||||
video_prompt = build_prompt(length_key, "video")
|
||||
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
||||
opts = {"temperature": temp, "num_predict": num_predict}
|
||||
involved = {model} if not router else {router_model, shot_model, photo_model}
|
||||
if ocr_text:
|
||||
involved.add(ocr_model)
|
||||
if process_videos:
|
||||
involved.add(video_model)
|
||||
think_map = {}
|
||||
for m in involved:
|
||||
think_map[m] = False if model_thinks(m) else None
|
||||
@@ -1336,7 +1363,7 @@ def process_loop(settings):
|
||||
log("warn", "DRY RUN — no metadata will be written")
|
||||
|
||||
with S.lock:
|
||||
pending = list(S.files)
|
||||
pending = list(S.files) + (list(S.video_files) if process_videos else [])
|
||||
idx_offset = 0
|
||||
if skip_done:
|
||||
before = len(pending)
|
||||
@@ -1345,16 +1372,20 @@ def process_loop(settings):
|
||||
with S.lock:
|
||||
S.already_done = idx_offset
|
||||
if idx_offset:
|
||||
log("info", f"resume: {idx_offset} photos already in journal, skipping them")
|
||||
log("info", f"resume: {idx_offset} photos/videos already in journal, skipping them")
|
||||
|
||||
total = len(pending) + idx_offset
|
||||
with S.lock:
|
||||
S.total_images = total # so the progress bar/ETA reflect videos too, not just images
|
||||
if total == idx_offset:
|
||||
log("info", "All photos in this folder are already tagged.")
|
||||
log("info", "Everything in this folder is already tagged.")
|
||||
with S.lock:
|
||||
S.status = "done"
|
||||
push_stats()
|
||||
broadcast("state", {"status": "done"})
|
||||
return
|
||||
if process_videos and any(os.path.splitext(p)[1].lower() in VIDEO_EXTS for p in pending):
|
||||
log("info", f"video model :: {video_model} (multi-frame sampling, up to 6 frames per clip)")
|
||||
|
||||
# Multi-threaded Queues
|
||||
downscale_queue = queue.Queue(maxsize=2)
|
||||
@@ -1387,15 +1418,21 @@ def process_loop(settings):
|
||||
time.sleep(0.2)
|
||||
if should_halt():
|
||||
break
|
||||
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
||||
try:
|
||||
t0 = time.time()
|
||||
img_bytes = downscale(path, max_px, tmpdir)
|
||||
b64 = base64.b64encode(img_bytes).decode()
|
||||
if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)):
|
||||
if is_video:
|
||||
b64_list = extract_video_frames_b64(path, tmpdir)
|
||||
preview_bytes = base64.b64decode(b64_list[0])
|
||||
else:
|
||||
preview_bytes = downscale(path, max_px, tmpdir)
|
||||
b64_list = [base64.b64encode(preview_bytes).decode()]
|
||||
if not put_until_halt(downscale_queue, (i, path, preview_bytes, b64_list, t0, is_video)):
|
||||
break
|
||||
except Exception as e:
|
||||
log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}")
|
||||
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())):
|
||||
kind = "frame extraction" if is_video else "downscaling"
|
||||
log("error", f"{kind.capitalize()} failed for {os.path.basename(path)} :: {e}")
|
||||
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time(), is_video)):
|
||||
break
|
||||
put_until_halt(downscale_queue, None)
|
||||
|
||||
@@ -1417,30 +1454,34 @@ def process_loop(settings):
|
||||
if item is None:
|
||||
put_until_halt(write_queue, None)
|
||||
break
|
||||
i, path, img_bytes, b64_or_err, t0 = item
|
||||
i, path, preview_bytes, b64_or_err, t0, is_video = item
|
||||
name = os.path.basename(path)
|
||||
if img_bytes is None:
|
||||
if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)):
|
||||
if preview_bytes is None:
|
||||
kind = "Frame extraction" if is_video else "Downscale"
|
||||
if not put_until_halt(write_queue, (i, path, None, None, None, f"{kind} error: {b64_or_err}", t0)):
|
||||
break
|
||||
continue
|
||||
with S.lock:
|
||||
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total}
|
||||
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total, "isVideo": is_video}
|
||||
S.preview_seq += 1
|
||||
S.preview = (img_bytes, S.preview_seq)
|
||||
S.preview = (preview_bytes, S.preview_seq)
|
||||
broadcast("photo_start", S.current)
|
||||
broadcast("preview", {"seq": S.preview_seq})
|
||||
try:
|
||||
route = None
|
||||
use_model, use_prompt = model, photo_prompt
|
||||
if router:
|
||||
r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
|
||||
{"temperature": 0, "num_predict": 30},
|
||||
keep_alive, think_map.get(router_model), KIND_SCHEMA)
|
||||
route = salvage_json(r.get("response", "")).get("kind")
|
||||
if route not in ("screenshot", "photo"):
|
||||
route = "photo"
|
||||
use_model = shot_model if route == "screenshot" else photo_model
|
||||
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
||||
if is_video:
|
||||
use_model, use_prompt = video_model, video_prompt
|
||||
else:
|
||||
use_model, use_prompt = model, photo_prompt
|
||||
if router:
|
||||
r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
|
||||
{"temperature": 0, "num_predict": 30},
|
||||
keep_alive, think_map.get(router_model), KIND_SCHEMA)
|
||||
route = salvage_json(r.get("response", "")).get("kind")
|
||||
if route not in ("screenshot", "photo"):
|
||||
route = "photo"
|
||||
use_model = shot_model if route == "screenshot" else photo_model
|
||||
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
||||
resp = ollama_generate(use_model, use_prompt, b64_or_err,
|
||||
opts, keep_alive, think_map.get(use_model))
|
||||
raw = resp.get("response", "")
|
||||
@@ -2425,24 +2466,27 @@ class Handler(BaseHTTPRequestHandler):
|
||||
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
||||
organize = bool(settings.get("organize", False))
|
||||
|
||||
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
|
||||
photo_prompt = build_prompt(length_key, "photo")
|
||||
shot_prompt = build_prompt(length_key, "screenshot")
|
||||
video_prompt = build_prompt(length_key, "video")
|
||||
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
||||
opts = {"temperature": temp, "num_predict": num_predict}
|
||||
|
||||
think = None
|
||||
if model_thinks(model):
|
||||
think = False
|
||||
|
||||
|
||||
tmpdir = tempfile.mkdtemp(prefix="photon_redo_")
|
||||
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
||||
frame_source = extract_video_frame(path, tmpdir) if is_video else path
|
||||
img = downscale(frame_source, max_px, tmpdir)
|
||||
b64 = base64.b64encode(img).decode()
|
||||
|
||||
|
||||
route = None
|
||||
use_model, use_prompt = model, photo_prompt
|
||||
if router:
|
||||
if is_video:
|
||||
use_model, use_prompt = video_model, video_prompt
|
||||
b64 = extract_video_frames_b64(path, tmpdir)
|
||||
else:
|
||||
use_model, use_prompt = model, photo_prompt
|
||||
img = downscale(path, max_px, tmpdir)
|
||||
b64 = base64.b64encode(img).decode()
|
||||
|
||||
think = False if model_thinks(use_model) else None
|
||||
if not is_video and router:
|
||||
r = ollama_generate(router_model, ROUTER_PROMPT, b64,
|
||||
{"temperature": 0, "num_predict": 30},
|
||||
keep_alive, think, KIND_SCHEMA)
|
||||
@@ -2451,6 +2495,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
route = "photo"
|
||||
use_model = shot_model if route == "screenshot" else photo_model
|
||||
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
||||
think = False if model_thinks(use_model) else None
|
||||
resp = ollama_generate(use_model, use_prompt, b64,
|
||||
opts, keep_alive, think)
|
||||
raw = resp.get("response", "")
|
||||
|
||||
Reference in New Issue
Block a user