Add real video tagging via minicpm-v4.6:1b, fix broken video keyword writes
Videos were never part of the automated pipeline before -- only reachable one at a time via a manual "Tag Now" button, and even that just grabbed one static frame and ran it through the photo model. Now videos are first-class: - New extract_video_frames_b64(): samples up to 6 frames spread across the clip's duration and passes them all to the video model in one call, so it sees actual motion/progression instead of one snapshot. Verified live: two different real test videos got distinct, content-aware descriptions that correctly named what was actually happening in each, not generic placeholders. - ollama_generate() now accepts a list of images (photos still pass a single one, unchanged) so the same call path serves both. - process_loop merges S.video_files into the same pending queue as photos, routes videos to a separate configurable video model (default minicpm-v4.6:1b, a small dedicated vision model) and prompt, skipping the photo/screenshot router entirely. redo_single (the lightbox "AI Redo" button) updated the same way for consistency. - S.total_images is now set to the actual combined pending count for the run so the progress bar/ETA reflect videos too, not just photos. - FOUND WHILE TESTING: write_metadata's video branch used "-Keywords" for the category/photon-tagged marker, which is a silent no-op on QuickTime/.mov files (exiftool has no mapping for it there) -- confirmed by direct testing. Every video "tagged" before this would have gotten a description but never an actual category keyword embedded. Switched to XMP-dc:Subject (the same tag family used for photos, which QuickTime containers do support via an embedded XMP packet) -- verified the category now lands and stays idempotent across repeat writes. - New frontend controls: video model dropdown and a "tag videos too" toggle (on by default) in Engine Configurations. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
19
index.html
19
index.html
@@ -702,6 +702,11 @@ mark{background:rgba(34,211,238,.28);color:#fff;border-radius:2px;padding:0 1px}
|
|||||||
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
|
<div class="tog"><span>keep _original backups</span><div class="sw" id="swBackup"><i></i></div></div>
|
||||||
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
|
<div class="hint">off = atomic metadata write. on = copies originals (doubles disk usage).</div>
|
||||||
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
|
<div class="tog"><span>dry run (no writes)</span><div class="sw" id="swDry"><i></i></div></div>
|
||||||
|
|
||||||
|
<div class="tog"><span>tag videos too</span><div class="sw on" id="swVideo"><i></i></div></div>
|
||||||
|
<div class="hint">videos are included in the same pipeline run — several frames sampled across each clip go to the video model in one call for real motion-aware description.</div>
|
||||||
|
<label>video model</label>
|
||||||
|
<select id="videoModel"></select>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -1141,7 +1146,9 @@ $("btnStart").onclick = ()=>{
|
|||||||
keepBackup: swOn("swBackup"),
|
keepBackup: swOn("swBackup"),
|
||||||
dryRun: swOn("swDry"),
|
dryRun: swOn("swDry"),
|
||||||
organize: swOn("swOrganize"),
|
organize: swOn("swOrganize"),
|
||||||
integrity: swOn("swIntegrity")
|
integrity: swOn("swIntegrity"),
|
||||||
|
videoModel: $("videoModel").value,
|
||||||
|
processVideos: swOn("swVideo")
|
||||||
});
|
});
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -1256,7 +1263,9 @@ function engineSettings(){
|
|||||||
keepAlive: $("keepAlive").value,
|
keepAlive: $("keepAlive").value,
|
||||||
keepBackup: swOn("swBackup"),
|
keepBackup: swOn("swBackup"),
|
||||||
preserveDate: swOn("swDate"),
|
preserveDate: swOn("swDate"),
|
||||||
organize: swOn("swOrganize")
|
organize: swOn("swOrganize"),
|
||||||
|
videoModel: $("videoModel").value,
|
||||||
|
processVideos: swOn("swVideo")
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2213,7 +2222,8 @@ function connect(){
|
|||||||
const sel=$("model");
|
const sel=$("model");
|
||||||
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
|
const visionModels=(mj.models||[]).filter(m=>m.vision && m.name!=="qwen3.5:4b-mlx"); // mlx build: vision is broken
|
||||||
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
|
[["model","qwen3.5:9b"],["routerModel","glm-ocr:latest"],
|
||||||
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"]].forEach(([id,pref])=>{
|
["shotModel","qwen3.5:4b"],["photoModel","qwen3.5:4b"],
|
||||||
|
["videoModel","minicpm-v4.6:1b"]].forEach(([id,pref])=>{
|
||||||
const s=$(id);
|
const s=$(id);
|
||||||
visionModels.forEach(m=>{
|
visionModels.forEach(m=>{
|
||||||
const o=document.createElement("option");
|
const o=document.createElement("option");
|
||||||
@@ -2225,6 +2235,9 @@ function connect(){
|
|||||||
if(!sel.children.length){
|
if(!sel.children.length){
|
||||||
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
|
addLog({level:"error",msg:"no vision-capable ollama models found — pull one, e.g. `ollama pull qwen3.5:4b`",ts:"--:--:--"});
|
||||||
}
|
}
|
||||||
|
if($("videoModel") && ![...$("videoModel").options].some(o=>o.value==="minicpm-v4.6:1b")){
|
||||||
|
addLog({level:"warn",msg:"minicpm-v4.6:1b not found — pull it for video tagging: `ollama pull minicpm-v4.6:1b`",ts:"--:--:--"});
|
||||||
|
}
|
||||||
|
|
||||||
populateCategoryDropdowns();
|
populateCategoryDropdowns();
|
||||||
renderCategoryListEditor();
|
renderCategoryListEditor();
|
||||||
|
|||||||
141
server.py
141
server.py
@@ -886,10 +886,14 @@ def model_thinks(model):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA):
|
def ollama_generate(model, prompt, img_b64, opts, keep_alive, think=None, schema=SCHEMA):
|
||||||
|
"""img_b64 is either a single base64 image string (photos) or a list of
|
||||||
|
them (video: multiple sampled frames in one call, for real multi-frame
|
||||||
|
understanding instead of treating one static frame as a photo)."""
|
||||||
|
images = img_b64 if isinstance(img_b64, list) else [img_b64]
|
||||||
body = {
|
body = {
|
||||||
"model": model,
|
"model": model,
|
||||||
"prompt": prompt,
|
"prompt": prompt,
|
||||||
"images": [img_b64],
|
"images": images,
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"options": opts,
|
"options": opts,
|
||||||
"keep_alive": keep_alive,
|
"keep_alive": keep_alive,
|
||||||
@@ -1096,9 +1100,11 @@ def downscale(path, max_px, tmpdir):
|
|||||||
with open(out, "rb") as f:
|
with open(out, "rb") as f:
|
||||||
return f.read()
|
return f.read()
|
||||||
|
|
||||||
def extract_video_frame(path, tmpdir):
|
def extract_video_frames_b64(path, tmpdir, max_frames=6):
|
||||||
"""Grab one representative frame from a video via ffmpeg (read-only). Returns a jpeg file path."""
|
"""Sample up to max_frames frames spread uniformly across the video's
|
||||||
out = os.path.join(tmpdir, "photon_vframe.jpg")
|
duration and return them as base64 strings, for genuine multi-frame
|
||||||
|
video understanding — the model sees the clip's actual progression,
|
||||||
|
not one static snapshot mistaken for a photo."""
|
||||||
dur = 3.0
|
dur = 3.0
|
||||||
try:
|
try:
|
||||||
pr = subprocess.run(
|
pr = subprocess.run(
|
||||||
@@ -1108,12 +1114,21 @@ def extract_video_frame(path, tmpdir):
|
|||||||
dur = float(pr.stdout.strip())
|
dur = float(pr.stdout.strip())
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
ts = max(0.5, min(dur * 0.3, max(dur - 0.2, 0.5))) if dur > 1 else 0.1
|
n = max(2, min(max_frames, int(dur / 1.5) + 1)) if dur > 0 else 2
|
||||||
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1", "-q:v", "3", out]
|
frames = []
|
||||||
r = subprocess.run(cmd, capture_output=True, timeout=60)
|
for i in range(n):
|
||||||
if r.returncode != 0 or not os.path.exists(out):
|
ts = (dur * i / n) if n > 1 else dur * 0.3
|
||||||
raise RuntimeError(f"ffmpeg frame extraction failed: {r.stderr.decode(errors='replace')[:200]}")
|
ts = max(0.1, min(ts, max(dur - 0.1, 0.1)))
|
||||||
return out
|
out = os.path.join(tmpdir, f"photon_vframe_{i}.jpg")
|
||||||
|
cmd = ["ffmpeg", "-y", "-ss", str(ts), "-i", path, "-frames:v", "1",
|
||||||
|
"-vf", "scale=896:-2", "-q:v", "4", out]
|
||||||
|
r = subprocess.run(cmd, capture_output=True, timeout=30)
|
||||||
|
if r.returncode == 0 and os.path.exists(out):
|
||||||
|
with open(out, "rb") as f:
|
||||||
|
frames.append(base64.b64encode(f.read()).decode())
|
||||||
|
if not frames:
|
||||||
|
raise RuntimeError("ffmpeg could not extract any frames from this video")
|
||||||
|
return frames
|
||||||
|
|
||||||
def video_thumbnail(path, out_path):
|
def video_thumbnail(path, out_path):
|
||||||
"""Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only)."""
|
"""Cheap ffmpeg frame-grab thumbnail for the browsing grid (read-only)."""
|
||||||
@@ -1161,6 +1176,10 @@ def build_prompt(length_key, mode="photo"):
|
|||||||
task = (f"1. This image is a screenshot or document. In ONE sentence of at most "
|
task = (f"1. This image is a screenshot or document. In ONE sentence of at most "
|
||||||
f"{words} words, say what app/website/document it is and what it shows "
|
f"{words} words, say what app/website/document it is and what it shows "
|
||||||
"(the topic, not the exact words). NEVER copy the text verbatim.\n")
|
"(the topic, not the exact words). NEVER copy the text verbatim.\n")
|
||||||
|
elif mode == "video":
|
||||||
|
task = (f"1. These images are frames sampled in order across a short video clip. "
|
||||||
|
f"Describe what happens in the clip in ONE sentence, at most {words} words, "
|
||||||
|
"noting the main subject/action and any motion or change across the frames.\n")
|
||||||
else:
|
else:
|
||||||
task = (f"1. Describe this photo in ONE sentence, at most {words} words. "
|
task = (f"1. Describe this photo in ONE sentence, at most {words} words. "
|
||||||
"Mention the main subject, setting, and any clearly readable text.\n")
|
"Mention the main subject, setting, and any clearly readable text.\n")
|
||||||
@@ -1189,15 +1208,18 @@ def write_metadata(path, desc, category, keep_backup, preserve_date, old_categor
|
|||||||
if not keep_backup:
|
if not keep_backup:
|
||||||
args.append("-overwrite_original") # writes temp file then atomic rename
|
args.append("-overwrite_original") # writes temp file then atomic rename
|
||||||
if is_video:
|
if is_video:
|
||||||
# mp4/mov containers don't carry EXIF/IPTC — use exiftool's generic
|
# mp4/mov containers don't carry EXIF/IPTC — "-Keywords" is a silent
|
||||||
# tag names so it resolves to QuickTime/Keys groups automatically.
|
# no-op on QuickTime files (exiftool has no mapping for it there),
|
||||||
|
# confirmed by testing directly. XMP-dc:Subject is the one that
|
||||||
|
# actually lands (QuickTime containers support an embedded XMP
|
||||||
|
# packet) — same tag family used for photos.
|
||||||
if old_category and old_category != category:
|
if old_category and old_category != category:
|
||||||
args.append(f"-Keywords-={old_category}")
|
args.append(f"-XMP-dc:Subject-={old_category}")
|
||||||
args += [
|
args += [
|
||||||
f"-Keywords-={category}", f"-Keywords-=photon-tagged",
|
f"-XMP-dc:Subject-={category}", "-XMP-dc:Subject-=photon-tagged",
|
||||||
f"-Description={desc}",
|
f"-Description={desc}",
|
||||||
f"-Keywords+={category}",
|
f"-XMP-dc:Subject+={category}",
|
||||||
"-Keywords+=photon-tagged",
|
"-XMP-dc:Subject+=photon-tagged",
|
||||||
]
|
]
|
||||||
else:
|
else:
|
||||||
if old_category and old_category != category:
|
if old_category and old_category != category:
|
||||||
@@ -1311,14 +1333,19 @@ def process_loop(settings):
|
|||||||
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
||||||
organize = bool(settings.get("organize", False))
|
organize = bool(settings.get("organize", False))
|
||||||
integrity = bool(settings.get("integrity", False))
|
integrity = bool(settings.get("integrity", False))
|
||||||
|
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
|
||||||
|
process_videos = bool(settings.get("processVideos", True))
|
||||||
|
|
||||||
photo_prompt = build_prompt(length_key, "photo")
|
photo_prompt = build_prompt(length_key, "photo")
|
||||||
shot_prompt = build_prompt(length_key, "screenshot")
|
shot_prompt = build_prompt(length_key, "screenshot")
|
||||||
|
video_prompt = build_prompt(length_key, "video")
|
||||||
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
||||||
opts = {"temperature": temp, "num_predict": num_predict}
|
opts = {"temperature": temp, "num_predict": num_predict}
|
||||||
involved = {model} if not router else {router_model, shot_model, photo_model}
|
involved = {model} if not router else {router_model, shot_model, photo_model}
|
||||||
if ocr_text:
|
if ocr_text:
|
||||||
involved.add(ocr_model)
|
involved.add(ocr_model)
|
||||||
|
if process_videos:
|
||||||
|
involved.add(video_model)
|
||||||
think_map = {}
|
think_map = {}
|
||||||
for m in involved:
|
for m in involved:
|
||||||
think_map[m] = False if model_thinks(m) else None
|
think_map[m] = False if model_thinks(m) else None
|
||||||
@@ -1336,7 +1363,7 @@ def process_loop(settings):
|
|||||||
log("warn", "DRY RUN — no metadata will be written")
|
log("warn", "DRY RUN — no metadata will be written")
|
||||||
|
|
||||||
with S.lock:
|
with S.lock:
|
||||||
pending = list(S.files)
|
pending = list(S.files) + (list(S.video_files) if process_videos else [])
|
||||||
idx_offset = 0
|
idx_offset = 0
|
||||||
if skip_done:
|
if skip_done:
|
||||||
before = len(pending)
|
before = len(pending)
|
||||||
@@ -1345,16 +1372,20 @@ def process_loop(settings):
|
|||||||
with S.lock:
|
with S.lock:
|
||||||
S.already_done = idx_offset
|
S.already_done = idx_offset
|
||||||
if idx_offset:
|
if idx_offset:
|
||||||
log("info", f"resume: {idx_offset} photos already in journal, skipping them")
|
log("info", f"resume: {idx_offset} photos/videos already in journal, skipping them")
|
||||||
|
|
||||||
total = len(pending) + idx_offset
|
total = len(pending) + idx_offset
|
||||||
|
with S.lock:
|
||||||
|
S.total_images = total # so the progress bar/ETA reflect videos too, not just images
|
||||||
if total == idx_offset:
|
if total == idx_offset:
|
||||||
log("info", "All photos in this folder are already tagged.")
|
log("info", "Everything in this folder is already tagged.")
|
||||||
with S.lock:
|
with S.lock:
|
||||||
S.status = "done"
|
S.status = "done"
|
||||||
push_stats()
|
push_stats()
|
||||||
broadcast("state", {"status": "done"})
|
broadcast("state", {"status": "done"})
|
||||||
return
|
return
|
||||||
|
if process_videos and any(os.path.splitext(p)[1].lower() in VIDEO_EXTS for p in pending):
|
||||||
|
log("info", f"video model :: {video_model} (multi-frame sampling, up to 6 frames per clip)")
|
||||||
|
|
||||||
# Multi-threaded Queues
|
# Multi-threaded Queues
|
||||||
downscale_queue = queue.Queue(maxsize=2)
|
downscale_queue = queue.Queue(maxsize=2)
|
||||||
@@ -1387,15 +1418,21 @@ def process_loop(settings):
|
|||||||
time.sleep(0.2)
|
time.sleep(0.2)
|
||||||
if should_halt():
|
if should_halt():
|
||||||
break
|
break
|
||||||
|
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
||||||
try:
|
try:
|
||||||
t0 = time.time()
|
t0 = time.time()
|
||||||
img_bytes = downscale(path, max_px, tmpdir)
|
if is_video:
|
||||||
b64 = base64.b64encode(img_bytes).decode()
|
b64_list = extract_video_frames_b64(path, tmpdir)
|
||||||
if not put_until_halt(downscale_queue, (i, path, img_bytes, b64, t0)):
|
preview_bytes = base64.b64decode(b64_list[0])
|
||||||
|
else:
|
||||||
|
preview_bytes = downscale(path, max_px, tmpdir)
|
||||||
|
b64_list = [base64.b64encode(preview_bytes).decode()]
|
||||||
|
if not put_until_halt(downscale_queue, (i, path, preview_bytes, b64_list, t0, is_video)):
|
||||||
break
|
break
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log("error", f"Downscaling failed for {os.path.basename(path)} :: {e}")
|
kind = "frame extraction" if is_video else "downscaling"
|
||||||
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time())):
|
log("error", f"{kind.capitalize()} failed for {os.path.basename(path)} :: {e}")
|
||||||
|
if not put_until_halt(downscale_queue, (i, path, None, str(e), time.time(), is_video)):
|
||||||
break
|
break
|
||||||
put_until_halt(downscale_queue, None)
|
put_until_halt(downscale_queue, None)
|
||||||
|
|
||||||
@@ -1417,30 +1454,34 @@ def process_loop(settings):
|
|||||||
if item is None:
|
if item is None:
|
||||||
put_until_halt(write_queue, None)
|
put_until_halt(write_queue, None)
|
||||||
break
|
break
|
||||||
i, path, img_bytes, b64_or_err, t0 = item
|
i, path, preview_bytes, b64_or_err, t0, is_video = item
|
||||||
name = os.path.basename(path)
|
name = os.path.basename(path)
|
||||||
if img_bytes is None:
|
if preview_bytes is None:
|
||||||
if not put_until_halt(write_queue, (i, path, None, None, None, f"Downscale error: {b64_or_err}", t0)):
|
kind = "Frame extraction" if is_video else "Downscale"
|
||||||
|
if not put_until_halt(write_queue, (i, path, None, None, None, f"{kind} error: {b64_or_err}", t0)):
|
||||||
break
|
break
|
||||||
continue
|
continue
|
||||||
with S.lock:
|
with S.lock:
|
||||||
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total}
|
S.current = {"path": path, "name": name, "idx": idx_offset + i + 1, "total": total, "isVideo": is_video}
|
||||||
S.preview_seq += 1
|
S.preview_seq += 1
|
||||||
S.preview = (img_bytes, S.preview_seq)
|
S.preview = (preview_bytes, S.preview_seq)
|
||||||
broadcast("photo_start", S.current)
|
broadcast("photo_start", S.current)
|
||||||
broadcast("preview", {"seq": S.preview_seq})
|
broadcast("preview", {"seq": S.preview_seq})
|
||||||
try:
|
try:
|
||||||
route = None
|
route = None
|
||||||
use_model, use_prompt = model, photo_prompt
|
if is_video:
|
||||||
if router:
|
use_model, use_prompt = video_model, video_prompt
|
||||||
r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
|
else:
|
||||||
{"temperature": 0, "num_predict": 30},
|
use_model, use_prompt = model, photo_prompt
|
||||||
keep_alive, think_map.get(router_model), KIND_SCHEMA)
|
if router:
|
||||||
route = salvage_json(r.get("response", "")).get("kind")
|
r = ollama_generate(router_model, ROUTER_PROMPT, b64_or_err,
|
||||||
if route not in ("screenshot", "photo"):
|
{"temperature": 0, "num_predict": 30},
|
||||||
route = "photo"
|
keep_alive, think_map.get(router_model), KIND_SCHEMA)
|
||||||
use_model = shot_model if route == "screenshot" else photo_model
|
route = salvage_json(r.get("response", "")).get("kind")
|
||||||
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
if route not in ("screenshot", "photo"):
|
||||||
|
route = "photo"
|
||||||
|
use_model = shot_model if route == "screenshot" else photo_model
|
||||||
|
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
||||||
resp = ollama_generate(use_model, use_prompt, b64_or_err,
|
resp = ollama_generate(use_model, use_prompt, b64_or_err,
|
||||||
opts, keep_alive, think_map.get(use_model))
|
opts, keep_alive, think_map.get(use_model))
|
||||||
raw = resp.get("response", "")
|
raw = resp.get("response", "")
|
||||||
@@ -2425,24 +2466,27 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
ocr_model = settings.get("ocrModel") or "glm-ocr:latest"
|
||||||
organize = bool(settings.get("organize", False))
|
organize = bool(settings.get("organize", False))
|
||||||
|
|
||||||
|
video_model = settings.get("videoModel") or "minicpm-v4.6:1b"
|
||||||
photo_prompt = build_prompt(length_key, "photo")
|
photo_prompt = build_prompt(length_key, "photo")
|
||||||
shot_prompt = build_prompt(length_key, "screenshot")
|
shot_prompt = build_prompt(length_key, "screenshot")
|
||||||
|
video_prompt = build_prompt(length_key, "video")
|
||||||
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
num_predict = LENGTH_PRESETS[length_key]["num_predict"]
|
||||||
opts = {"temperature": temp, "num_predict": num_predict}
|
opts = {"temperature": temp, "num_predict": num_predict}
|
||||||
|
|
||||||
think = None
|
|
||||||
if model_thinks(model):
|
|
||||||
think = False
|
|
||||||
|
|
||||||
tmpdir = tempfile.mkdtemp(prefix="photon_redo_")
|
tmpdir = tempfile.mkdtemp(prefix="photon_redo_")
|
||||||
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
is_video = os.path.splitext(path)[1].lower() in VIDEO_EXTS
|
||||||
frame_source = extract_video_frame(path, tmpdir) if is_video else path
|
|
||||||
img = downscale(frame_source, max_px, tmpdir)
|
|
||||||
b64 = base64.b64encode(img).decode()
|
|
||||||
|
|
||||||
route = None
|
route = None
|
||||||
use_model, use_prompt = model, photo_prompt
|
if is_video:
|
||||||
if router:
|
use_model, use_prompt = video_model, video_prompt
|
||||||
|
b64 = extract_video_frames_b64(path, tmpdir)
|
||||||
|
else:
|
||||||
|
use_model, use_prompt = model, photo_prompt
|
||||||
|
img = downscale(path, max_px, tmpdir)
|
||||||
|
b64 = base64.b64encode(img).decode()
|
||||||
|
|
||||||
|
think = False if model_thinks(use_model) else None
|
||||||
|
if not is_video and router:
|
||||||
r = ollama_generate(router_model, ROUTER_PROMPT, b64,
|
r = ollama_generate(router_model, ROUTER_PROMPT, b64,
|
||||||
{"temperature": 0, "num_predict": 30},
|
{"temperature": 0, "num_predict": 30},
|
||||||
keep_alive, think, KIND_SCHEMA)
|
keep_alive, think, KIND_SCHEMA)
|
||||||
@@ -2451,6 +2495,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
route = "photo"
|
route = "photo"
|
||||||
use_model = shot_model if route == "screenshot" else photo_model
|
use_model = shot_model if route == "screenshot" else photo_model
|
||||||
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
use_prompt = shot_prompt if route == "screenshot" else photo_prompt
|
||||||
|
think = False if model_thinks(use_model) else None
|
||||||
resp = ollama_generate(use_model, use_prompt, b64,
|
resp = ollama_generate(use_model, use_prompt, b64,
|
||||||
opts, keep_alive, think)
|
opts, keep_alive, think)
|
||||||
raw = resp.get("response", "")
|
raw = resp.get("response", "")
|
||||||
|
|||||||
Reference in New Issue
Block a user