diff --git a/app.py b/app.py index ea260af..d843927 100644 --- a/app.py +++ b/app.py @@ -68,10 +68,12 @@ def api_page(): @app.route("/health") def health(): + # No generation probe here: a tiny probe prompt on the shared resident instance + # is the vornith corruption trigger. Model presence via /api/tags (no GPU work). try: - r = http.post(f"{CFG['ollama_url']}/api/generate", stream=True, timeout=4, - json={"model": CFG["model"], "prompt": "ping", "stream": True, "options": {"num_predict": 1}}) - llm = r.status_code == 200 + r = http.get(f"{CFG['ollama_url']}/api/tags", timeout=4) + models = [m.get("name") for m in r.json().get("models", [])] + llm = CFG["model"] in models except Exception: llm = False return jsonify(status="ok", llm=llm, model=CFG["model"], **core.get_stats()) @@ -122,6 +124,14 @@ def chat(): return Response(generate(), mimetype="text/event-stream", headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"}) +@app.route("/api/warm") +def api_warm(): + """Heartbeat target for the systemd timer; also a manual health-probe for the model lane.""" + healthy, sample = core.warm_model() + if not healthy and "CUDA" not in sample: + core.unload_model() # corrupt instance: drop it; next warm reloads clean + return jsonify(healthy=healthy, sample=sample, model=CFG["model"]) + @app.route("/api/remaining") def remaining(): ip = core.client_ip(request) diff --git a/config.example.json b/config.example.json index 7ece1f2..ac808fc 100644 --- a/config.example.json +++ b/config.example.json @@ -1,8 +1,9 @@ { - "model": "ornith-1.5:9b-64k", + "model": "vornith:latest", "ollama_url": "http://10.30.20.29:11434", "num_ctx": 8192, - "rag_k": 6, + "rag_k": 2, + "num_predict": 1100, "port": 8012, "mcp_port": 8012, "base_url": "https://draco.thetempleofdoom.com", diff --git a/deploy/draco-warm.service b/deploy/draco-warm.service new file mode 100644 index 0000000..c185dfa --- /dev/null +++ b/deploy/draco-warm.service @@ -0,0 +1,6 @@ +[Unit] +Description=DRACO model keep-warm ping + +[Service] +Type=oneshot +ExecStart=/usr/bin/curl -s -m 130 http://127.0.0.1:8012/api/warm diff --git a/deploy/draco-warm.timer b/deploy/draco-warm.timer new file mode 100644 index 0000000..3f1ecdf --- /dev/null +++ b/deploy/draco-warm.timer @@ -0,0 +1,9 @@ +[Unit] +Description=DRACO keep model warm every 4 minutes + +[Timer] +OnBootSec=60 +OnUnitActiveSec=240 + +[Install] +WantedBy=timers.target diff --git a/deploy/draco.service b/deploy/draco.service index 2d06128..bdf7da6 100644 --- a/deploy/draco.service +++ b/deploy/draco.service @@ -5,7 +5,7 @@ After=network.target [Service] Type=simple WorkingDirectory=/opt/draco -ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 app:app +ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 --graceful-timeout 10 app:app Restart=always RestartSec=5 Environment=PYTHONUNBUFFERED=1 diff --git a/draco_core.py b/draco_core.py index b5a3fbd..d88342e 100644 --- a/draco_core.py +++ b/draco_core.py @@ -20,6 +20,7 @@ CREATE TABLE IF NOT EXISTS users ( api_key VARCHAR(64) UNIQUE NOT NULL, credits INTEGER DEFAULT 0, free_used INTEGER DEFAULT 0, + total_calls INTEGER DEFAULT 0, is_admin INTEGER DEFAULT 0, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP, last_used_at TIMESTAMP @@ -58,6 +59,11 @@ def db(): def init_db(): c = db() c.executescript(SCHEMA) + try: # in-place migration for DBs created before total_calls existed + c.execute("SELECT total_calls FROM users LIMIT 1") + except sqlite3.OperationalError: + c.execute("ALTER TABLE users ADD COLUMN total_calls INTEGER DEFAULT 0") + c.commit() c.execute("INSERT OR IGNORE INTO users (email, api_key, credits, is_admin) VALUES (?,?,?,1)", ("admin@draco.local", CFG["admin_key"], 999999)) c.commit() @@ -127,12 +133,17 @@ SYSTEM = ( "Answer directly with no reasoning preamble and no meta commentary." ) -def build_prompt(question, k=6): +def build_prompt(question, k=None): + """RAG prompt. PROMPT BUDGET RULE: all prompts must land in the same size class + (~600 tok). vornith (linear-attn+MTP hybrid) corrupts when one resident instance + receives mixed tiny/long prompts; uniform bounded prompts are proven stable.""" + k = k or CFG.get("rag_k", 2) + k = min(k, 3) # hard cap: 3 x 1200-char excerpts ~= clean-zone prompt hits = search_library(question, k=k) if hits: blocks = [] for n, h in enumerate(hits, 1): - blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1400]}") + blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1200]}") ctx = "\n\n".join(blocks) prompt = f"{SYSTEM}\n\nBOOK EXCERPTS:\n{ctx}\n\nQUESTION: {question}\n\nANSWER (cite [n]):" else: @@ -314,6 +325,9 @@ def stream_ollama(prompt): """Yield (channel, piece) tuples: channel 'think' or 'answer'.""" r = http.post(f"{CFG['ollama_url']}/api/generate", json={"model": CFG["model"], "prompt": prompt, "stream": True, + # RESIDENT model: cold-load prefill >1k tokens CUDA-crashes on this + # hybrid arch. Corruption (????? output) is handled by detection + + # auto-reload in the app layer instead. "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100), "num_ctx": CFG.get("num_ctx", 8192)}}, timeout=(5, None), stream=True) @@ -328,24 +342,55 @@ def stream_ollama(prompt): yield channel, piece def ask_ollama(prompt): - """Non-streaming RAG answer (reasoning stripped, tagged or not).""" - r = http.post(f"{CFG['ollama_url']}/api/generate", - json={"model": CFG["model"], "prompt": prompt, "stream": False, - "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100), - "num_ctx": CFG.get("num_ctx", 8192)}}, - timeout=180) - r.raise_for_status() - _, answer = strip_cot(r.json().get("response", "")) - if degenerate(answer): - unload_model() - raise RuntimeError("model returned degenerate output โ€” model unloaded, retry") - return answer + """Non-streaming RAG answer with auto-recovery: on degenerate/CUDA failure, + unload the model, reload fresh, retry once before surfacing an error.""" + last_err = None + for attempt in (1, 2): + try: + r = http.post(f"{CFG['ollama_url']}/api/generate", + json={"model": CFG["model"], "prompt": prompt, "stream": False, + "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100), + "num_ctx": CFG.get("num_ctx", 8192)}}, + timeout=180) + r.raise_for_status() + raw = r.json().get("response", "") + if raw and degenerate(raw): + last_err = RuntimeError("degenerate output") + else: + _, answer = strip_cot(raw) + return answer + except http.HTTPError as e: + last_err = e + except Exception as e: + last_err = e + unload_model() # force clean reload for the next attempt + time.sleep(2) + raise RuntimeError(f"model unstable after retry: {last_err}") def degenerate(s): """vornith VRAM/session corruption signature: run of '?' chars.""" s = s.strip() return len(s) >= 40 and s.count("?") / len(s) > 0.4 +def warm_model(): + """Budget-sized heartbeat: keeps the model resident AND exercised at the canonical + prompt size. Tiny prompts (<50 tok) on the shared instance are the corruption + trigger, so the ping itself must be RAG-sized.""" + try: + excerpt = ("The utility of a uniform prompt budget is that the model never " + "encounters a context-length distribution shift between requests. " * 9) + prompt = (f"{SYSTEM}\n\nBOOK EXCERPTS:\n[1] Warmup Excerpt (maintenance)\n{excerpt}" + "\n\nQUESTION: Reply with exactly: ok\n\nANSWER (cite [n]):") + r = http.post(f"{CFG['ollama_url']}/api/generate", + json={"model": CFG["model"], "prompt": prompt, + "stream": False, "options": {"num_predict": 4}}, + timeout=120) + sample = r.json().get("response", "") if r.status_code == 200 else "" + healthy = r.status_code == 200 and not degenerate(sample) + return healthy, sample[:120] + except Exception as e: + return False, str(e)[:120] + def unload_model(): """Drop the model from VRAM so the next request reloads clean.""" try: diff --git a/pages.py b/pages.py index 44802c1..a7d082e 100644 --- a/pages.py +++ b/pages.py @@ -140,33 +140,40 @@ async function ask(text){ add('you','user').textContent=text; const body=add('draco','assistant');body.innerHTML=''; const srcs=document.createElement('div');srcs.className='srcs';srcs.style.display='none'; - try{ - const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'}, - body:JSON.stringify({q:text})}); - if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){} - body.textContent='โš  '+m;busy=false;go.disabled=false;return;} - const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='',thinkEl=null; - while(true){const{value,done}=await reader.read();if(done)break; - buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||''; - for(const line of lines){if(!line.startsWith('data: '))continue; - const ev=JSON.parse(line.slice(6)); - if(ev.type==='sources'&&ev.sources.length){ - srcs.style.display='block'; - srcs.textContent='๐Ÿ“š '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' ยท '); - }else if(ev.type==='token'&&ev.channel==='think'){ - if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking'; - thinkEl.innerHTML="โ—ˆ thinking "; - body.parentNode.insertBefore(thinkEl,body);} - thinkEl.querySelector('.tbody').textContent+=ev.text; - log.scrollTop=log.scrollHeight; - }else if(ev.type==='token'){acc+=ev.text;body.innerHTML=''; - body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor'; - log.scrollTop=log.scrollHeight; - }else if(ev.type==='error'){acc+='\nโš  '+ev.text;} - }} - }catch(e){acc+='\nโš  connection lost';} - body.innerHTML='';body.appendChild(document.createTextNode(acc||'(no output)')); + for(let attempt=1;attempt<=3;attempt++){ + let hadTokens=false,hadError=null,acc=''; + try{ + const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'}, + body:JSON.stringify({q:text})}); + if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){} + body.textContent='โš  '+m;busy=false;go.disabled=false;return;} + const reader=r.body.getReader(),dec=new TextDecoder();let buf='',thinkEl=null; + while(true){const{value,done}=await reader.read();if(done)break; + buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||''; + for(const line of lines){if(!line.startsWith('data: '))continue; + const ev=JSON.parse(line.slice(6)); + if(ev.type==='sources'&&ev.sources.length){ + srcs.style.display='block'; + srcs.textContent='๐Ÿ“š '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' ยท '); + }else if(ev.type==='token'&&ev.channel==='think'){ + if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking'; + thinkEl.innerHTML="โ—ˆ thinking "; + body.parentNode.insertBefore(thinkEl,body);} + thinkEl.querySelector('.tbody').textContent+=ev.text; + log.scrollTop=log.scrollHeight; + }else if(ev.type==='token'){acc+=ev.text;hadTokens=true;body.innerHTML=''; + body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor'; + log.scrollTop=log.scrollHeight; + }else if(ev.type==='error'){hadError=ev.text;} + }} + }catch(e){hadError='connection lost';} + if(hadTokens&&!hadError){break;} + if(attempt<3){body.innerHTML='';await new Promise(r=>setTimeout(r,1500));} + else if(hadError){body.textContent='โš  '+hadError;} + else break; + } const cur=body.querySelector('.cursor');if(cur)cur.remove(); + if(!body.textContent.trim())body.textContent='(no output)'; body.appendChild(srcs); const left=document.getElementById('left'); fetch('/api/remaining').then(r=>r.json()).then(d=>left.textContent=d.remaining+' free questions left today').catch(()=>{});