From 6f67a5a361895198d1ccf3be3723372e82d3fd56 Mon Sep 17 00:00:00 2001 From: drjones Date: Thu, 1 Oct 2026 00:08:28 +0000 Subject: [PATCH] =?UTF-8?q?GPU=20reroute:=20GamingPC=203070=20(.222)=20?= =?UTF-8?q?=E2=80=94=20text=20qwen3.5:4b,=20vision=20minicpm-v4.5:8b;=20Co?= =?UTF-8?q?T-stripping=20line=20parser?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- app.py | 26 +++++++++++++++++--------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/app.py b/app.py index faabee1..acb7f3e 100644 --- a/app.py +++ b/app.py @@ -4,9 +4,9 @@ Single-file Flask app. Deployed on Proxmox CT 172.""" import base64, json, os, random, re, requests from flask import Flask, jsonify, request, render_template_string -OLLAMA_URL = os.environ.get("LYRA_OLLAMA_URL", "http://10.30.20.69:11434") -TEXT_MODEL = os.environ.get("LYRA_TEXT_MODEL", "qwen3.5:9b") -VISION_MODEL = os.environ.get("LYRA_VISION_MODEL", "minicpm-v4.6:1b") +OLLAMA_URL = os.environ.get("LYRA_OLLAMA_URL", "http://10.30.20.222:11434") +TEXT_MODEL = os.environ.get("LYRA_TEXT_MODEL", "ornith-1.5:9b-64k") +VISION_MODEL = os.environ.get("LYRA_VISION_MODEL", "minicpm-v4.5:8b") app = Flask(__name__) @@ -106,8 +106,10 @@ SYSTEM_VISION = ( def ollama_generate(prompt, model, images=None, timeout=60): - payload = {"model": model, "prompt": prompt, "stream": False, "think": False, - "options": {"temperature": 0.9, "num_predict": 400}} + payload = {"model": model, "prompt": prompt, "stream": False, + "options": {"temperature": 0.9, "num_predict": 700}} + if "qwen3" in model: + payload["think"] = False if images: payload = {"model": model, "messages": [ {"role": "user", "content": prompt, "images": images}], @@ -116,6 +118,9 @@ def ollama_generate(prompt, model, images=None, timeout=60): r = requests.post(f"{OLLAMA_URL}/api/chat", json=payload, timeout=timeout) r.raise_for_status() return (r.json().get("message") or {}).get("content", "").strip() + r = requests.post(f"{OLLAMA_URL}/api/generate", json=payload, timeout=timeout) + r.raise_for_status() + return r.json().get("response", "").strip() @app.get("/health") @@ -130,9 +135,12 @@ def api_lines(): count = min(int(request.args.get("count", 5)), 10) fallback = random.sample(LINES.get(cat, LINES["witty"]), min(count, len(LINES.get(cat, [])))) try: - out = ollama_generate(SYSTEM_LINE % (count, cat, spice), TEXT_MODEL, timeout=45) + out = ollama_generate(SYSTEM_LINE % (count, cat, spice), TEXT_MODEL, timeout=90) + out = re.sub(r".*?", "", out, flags=re.S) # strip CoT blocks lines = [l.strip().strip('"-•') for l in out.splitlines() if l.strip()] - lines = [l for l in lines if 3 < len(l) < 140][:count] + lines = [l for l in lines if 3 < len(l) < 140 and not re.match( + r"^(okay|let me|i (will|'ll|'m)|sure|here|the user|we need|first|next|okay so)", l, re.I)] + lines = lines[-count:] if len(lines) > count else lines # CoT models put answers last if not lines: raise ValueError("empty") return jsonify(source="ollama", lines=lines) @@ -502,8 +510,8 @@ function timerReset(){clearInterval(tInt);clearInterval(bInt);tInt=null;$('timer function loadSettings(){ $('setEngine').value=S.engine||'browser';$('setVoice').value=S.voice||$('setVoice').value; $('setRate').value=S.rate||1;$('setPitch').value=S.pitch||1; - $('setTModel').value=S.tmodel||'qwen3.5:9b';$('setVModel').value=S.vmodel||'minicpm-v4.6:latest'; - $('setHost').value=S.host||'http://10.30.20.69:11434';$('setAutoEvery').value=S.autoEvery||45; + $('setTModel').value=S.tmodel||'qwen3.5:4b';$('setVModel').value=S.vmodel||'minicpm-v4.5:8b'; + $('setHost').value=S.host||'http://10.30.20.222:11434';$('setAutoEvery').value=S.autoEvery||45; $('setWarm').value=S.warm||50;$('autoMode').checked=!!S.autoMode;$('autoSpeak').checked=!!S.autoSpeak; $('rateV').textContent=$('setRate').value;$('pitchV').textContent=$('setPitch').value;$('aeV').textContent=$('setAutoEvery').value; applyWarm();}