GPU reroute: GamingPC 3070 (.222) — text qwen3.5:4b, vision minicpm-v4.5:8b; CoT-stripping line parser
This commit is contained in:
26
app.py
26
app.py
@@ -4,9 +4,9 @@ Single-file Flask app. Deployed on Proxmox CT 172."""
|
|||||||
import base64, json, os, random, re, requests
|
import base64, json, os, random, re, requests
|
||||||
from flask import Flask, jsonify, request, render_template_string
|
from flask import Flask, jsonify, request, render_template_string
|
||||||
|
|
||||||
OLLAMA_URL = os.environ.get("LYRA_OLLAMA_URL", "http://10.30.20.69:11434")
|
OLLAMA_URL = os.environ.get("LYRA_OLLAMA_URL", "http://10.30.20.222:11434")
|
||||||
TEXT_MODEL = os.environ.get("LYRA_TEXT_MODEL", "qwen3.5:9b")
|
TEXT_MODEL = os.environ.get("LYRA_TEXT_MODEL", "ornith-1.5:9b-64k")
|
||||||
VISION_MODEL = os.environ.get("LYRA_VISION_MODEL", "minicpm-v4.6:1b")
|
VISION_MODEL = os.environ.get("LYRA_VISION_MODEL", "minicpm-v4.5:8b")
|
||||||
|
|
||||||
app = Flask(__name__)
|
app = Flask(__name__)
|
||||||
|
|
||||||
@@ -106,8 +106,10 @@ SYSTEM_VISION = (
|
|||||||
|
|
||||||
|
|
||||||
def ollama_generate(prompt, model, images=None, timeout=60):
|
def ollama_generate(prompt, model, images=None, timeout=60):
|
||||||
payload = {"model": model, "prompt": prompt, "stream": False, "think": False,
|
payload = {"model": model, "prompt": prompt, "stream": False,
|
||||||
"options": {"temperature": 0.9, "num_predict": 400}}
|
"options": {"temperature": 0.9, "num_predict": 700}}
|
||||||
|
if "qwen3" in model:
|
||||||
|
payload["think"] = False
|
||||||
if images:
|
if images:
|
||||||
payload = {"model": model, "messages": [
|
payload = {"model": model, "messages": [
|
||||||
{"role": "user", "content": prompt, "images": images}],
|
{"role": "user", "content": prompt, "images": images}],
|
||||||
@@ -116,6 +118,9 @@ def ollama_generate(prompt, model, images=None, timeout=60):
|
|||||||
r = requests.post(f"{OLLAMA_URL}/api/chat", json=payload, timeout=timeout)
|
r = requests.post(f"{OLLAMA_URL}/api/chat", json=payload, timeout=timeout)
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
return (r.json().get("message") or {}).get("content", "").strip()
|
return (r.json().get("message") or {}).get("content", "").strip()
|
||||||
|
r = requests.post(f"{OLLAMA_URL}/api/generate", json=payload, timeout=timeout)
|
||||||
|
r.raise_for_status()
|
||||||
|
return r.json().get("response", "").strip()
|
||||||
|
|
||||||
|
|
||||||
@app.get("/health")
|
@app.get("/health")
|
||||||
@@ -130,9 +135,12 @@ def api_lines():
|
|||||||
count = min(int(request.args.get("count", 5)), 10)
|
count = min(int(request.args.get("count", 5)), 10)
|
||||||
fallback = random.sample(LINES.get(cat, LINES["witty"]), min(count, len(LINES.get(cat, []))))
|
fallback = random.sample(LINES.get(cat, LINES["witty"]), min(count, len(LINES.get(cat, []))))
|
||||||
try:
|
try:
|
||||||
out = ollama_generate(SYSTEM_LINE % (count, cat, spice), TEXT_MODEL, timeout=45)
|
out = ollama_generate(SYSTEM_LINE % (count, cat, spice), TEXT_MODEL, timeout=90)
|
||||||
|
out = re.sub(r"<think>.*?</think>", "", out, flags=re.S) # strip CoT blocks
|
||||||
lines = [l.strip().strip('"-•') for l in out.splitlines() if l.strip()]
|
lines = [l.strip().strip('"-•') for l in out.splitlines() if l.strip()]
|
||||||
lines = [l for l in lines if 3 < len(l) < 140][:count]
|
lines = [l for l in lines if 3 < len(l) < 140 and not re.match(
|
||||||
|
r"^(okay|let me|i (will|'ll|'m)|sure|here|the user|we need|first|next|okay so)", l, re.I)]
|
||||||
|
lines = lines[-count:] if len(lines) > count else lines # CoT models put answers last
|
||||||
if not lines:
|
if not lines:
|
||||||
raise ValueError("empty")
|
raise ValueError("empty")
|
||||||
return jsonify(source="ollama", lines=lines)
|
return jsonify(source="ollama", lines=lines)
|
||||||
@@ -502,8 +510,8 @@ function timerReset(){clearInterval(tInt);clearInterval(bInt);tInt=null;$('timer
|
|||||||
function loadSettings(){
|
function loadSettings(){
|
||||||
$('setEngine').value=S.engine||'browser';$('setVoice').value=S.voice||$('setVoice').value;
|
$('setEngine').value=S.engine||'browser';$('setVoice').value=S.voice||$('setVoice').value;
|
||||||
$('setRate').value=S.rate||1;$('setPitch').value=S.pitch||1;
|
$('setRate').value=S.rate||1;$('setPitch').value=S.pitch||1;
|
||||||
$('setTModel').value=S.tmodel||'qwen3.5:9b';$('setVModel').value=S.vmodel||'minicpm-v4.6:latest';
|
$('setTModel').value=S.tmodel||'qwen3.5:4b';$('setVModel').value=S.vmodel||'minicpm-v4.5:8b';
|
||||||
$('setHost').value=S.host||'http://10.30.20.69:11434';$('setAutoEvery').value=S.autoEvery||45;
|
$('setHost').value=S.host||'http://10.30.20.222:11434';$('setAutoEvery').value=S.autoEvery||45;
|
||||||
$('setWarm').value=S.warm||50;$('autoMode').checked=!!S.autoMode;$('autoSpeak').checked=!!S.autoSpeak;
|
$('setWarm').value=S.warm||50;$('autoMode').checked=!!S.autoMode;$('autoSpeak').checked=!!S.autoSpeak;
|
||||||
$('rateV').textContent=$('setRate').value;$('pitchV').textContent=$('setPitch').value;$('aeV').textContent=$('setAutoEvery').value;
|
$('rateV').textContent=$('setRate').value;$('pitchV').textContent=$('setPitch').value;$('aeV').textContent=$('setAutoEvery').value;
|
||||||
applyWarm();}
|
applyWarm();}
|
||||||
|
|||||||
Reference in New Issue
Block a user