diff --git a/app.py b/app.py
index ea260af..d843927 100644
--- a/app.py
+++ b/app.py
@@ -68,10 +68,12 @@ def api_page():
@app.route("/health")
def health():
+ # No generation probe here: a tiny probe prompt on the shared resident instance
+ # is the vornith corruption trigger. Model presence via /api/tags (no GPU work).
try:
- r = http.post(f"{CFG['ollama_url']}/api/generate", stream=True, timeout=4,
- json={"model": CFG["model"], "prompt": "ping", "stream": True, "options": {"num_predict": 1}})
- llm = r.status_code == 200
+ r = http.get(f"{CFG['ollama_url']}/api/tags", timeout=4)
+ models = [m.get("name") for m in r.json().get("models", [])]
+ llm = CFG["model"] in models
except Exception:
llm = False
return jsonify(status="ok", llm=llm, model=CFG["model"], **core.get_stats())
@@ -122,6 +124,14 @@ def chat():
return Response(generate(), mimetype="text/event-stream",
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
+@app.route("/api/warm")
+def api_warm():
+ """Heartbeat target for the systemd timer; also a manual health-probe for the model lane."""
+ healthy, sample = core.warm_model()
+ if not healthy and "CUDA" not in sample:
+ core.unload_model() # corrupt instance: drop it; next warm reloads clean
+ return jsonify(healthy=healthy, sample=sample, model=CFG["model"])
+
@app.route("/api/remaining")
def remaining():
ip = core.client_ip(request)
diff --git a/config.example.json b/config.example.json
index 7ece1f2..ac808fc 100644
--- a/config.example.json
+++ b/config.example.json
@@ -1,8 +1,9 @@
{
- "model": "ornith-1.5:9b-64k",
+ "model": "vornith:latest",
"ollama_url": "http://10.30.20.29:11434",
"num_ctx": 8192,
- "rag_k": 6,
+ "rag_k": 2,
+ "num_predict": 1100,
"port": 8012,
"mcp_port": 8012,
"base_url": "https://draco.thetempleofdoom.com",
diff --git a/deploy/draco-warm.service b/deploy/draco-warm.service
new file mode 100644
index 0000000..c185dfa
--- /dev/null
+++ b/deploy/draco-warm.service
@@ -0,0 +1,6 @@
+[Unit]
+Description=DRACO model keep-warm ping
+
+[Service]
+Type=oneshot
+ExecStart=/usr/bin/curl -s -m 130 http://127.0.0.1:8012/api/warm
diff --git a/deploy/draco-warm.timer b/deploy/draco-warm.timer
new file mode 100644
index 0000000..3f1ecdf
--- /dev/null
+++ b/deploy/draco-warm.timer
@@ -0,0 +1,9 @@
+[Unit]
+Description=DRACO keep model warm every 4 minutes
+
+[Timer]
+OnBootSec=60
+OnUnitActiveSec=240
+
+[Install]
+WantedBy=timers.target
diff --git a/deploy/draco.service b/deploy/draco.service
index 2d06128..bdf7da6 100644
--- a/deploy/draco.service
+++ b/deploy/draco.service
@@ -5,7 +5,7 @@ After=network.target
[Service]
Type=simple
WorkingDirectory=/opt/draco
-ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 app:app
+ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 --graceful-timeout 10 app:app
Restart=always
RestartSec=5
Environment=PYTHONUNBUFFERED=1
diff --git a/draco_core.py b/draco_core.py
index b5a3fbd..d88342e 100644
--- a/draco_core.py
+++ b/draco_core.py
@@ -20,6 +20,7 @@ CREATE TABLE IF NOT EXISTS users (
api_key VARCHAR(64) UNIQUE NOT NULL,
credits INTEGER DEFAULT 0,
free_used INTEGER DEFAULT 0,
+ total_calls INTEGER DEFAULT 0,
is_admin INTEGER DEFAULT 0,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
last_used_at TIMESTAMP
@@ -58,6 +59,11 @@ def db():
def init_db():
c = db()
c.executescript(SCHEMA)
+ try: # in-place migration for DBs created before total_calls existed
+ c.execute("SELECT total_calls FROM users LIMIT 1")
+ except sqlite3.OperationalError:
+ c.execute("ALTER TABLE users ADD COLUMN total_calls INTEGER DEFAULT 0")
+ c.commit()
c.execute("INSERT OR IGNORE INTO users (email, api_key, credits, is_admin) VALUES (?,?,?,1)",
("admin@draco.local", CFG["admin_key"], 999999))
c.commit()
@@ -127,12 +133,17 @@ SYSTEM = (
"Answer directly with no reasoning preamble and no meta commentary."
)
-def build_prompt(question, k=6):
+def build_prompt(question, k=None):
+ """RAG prompt. PROMPT BUDGET RULE: all prompts must land in the same size class
+ (~600 tok). vornith (linear-attn+MTP hybrid) corrupts when one resident instance
+ receives mixed tiny/long prompts; uniform bounded prompts are proven stable."""
+ k = k or CFG.get("rag_k", 2)
+ k = min(k, 3) # hard cap: 3 x 1200-char excerpts ~= clean-zone prompt
hits = search_library(question, k=k)
if hits:
blocks = []
for n, h in enumerate(hits, 1):
- blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1400]}")
+ blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1200]}")
ctx = "\n\n".join(blocks)
prompt = f"{SYSTEM}\n\nBOOK EXCERPTS:\n{ctx}\n\nQUESTION: {question}\n\nANSWER (cite [n]):"
else:
@@ -314,6 +325,9 @@ def stream_ollama(prompt):
"""Yield (channel, piece) tuples: channel 'think' or 'answer'."""
r = http.post(f"{CFG['ollama_url']}/api/generate",
json={"model": CFG["model"], "prompt": prompt, "stream": True,
+ # RESIDENT model: cold-load prefill >1k tokens CUDA-crashes on this
+ # hybrid arch. Corruption (????? output) is handled by detection +
+ # auto-reload in the app layer instead.
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
"num_ctx": CFG.get("num_ctx", 8192)}},
timeout=(5, None), stream=True)
@@ -328,24 +342,55 @@ def stream_ollama(prompt):
yield channel, piece
def ask_ollama(prompt):
- """Non-streaming RAG answer (reasoning stripped, tagged or not)."""
- r = http.post(f"{CFG['ollama_url']}/api/generate",
- json={"model": CFG["model"], "prompt": prompt, "stream": False,
- "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
- "num_ctx": CFG.get("num_ctx", 8192)}},
- timeout=180)
- r.raise_for_status()
- _, answer = strip_cot(r.json().get("response", ""))
- if degenerate(answer):
- unload_model()
- raise RuntimeError("model returned degenerate output โ model unloaded, retry")
- return answer
+ """Non-streaming RAG answer with auto-recovery: on degenerate/CUDA failure,
+ unload the model, reload fresh, retry once before surfacing an error."""
+ last_err = None
+ for attempt in (1, 2):
+ try:
+ r = http.post(f"{CFG['ollama_url']}/api/generate",
+ json={"model": CFG["model"], "prompt": prompt, "stream": False,
+ "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
+ "num_ctx": CFG.get("num_ctx", 8192)}},
+ timeout=180)
+ r.raise_for_status()
+ raw = r.json().get("response", "")
+ if raw and degenerate(raw):
+ last_err = RuntimeError("degenerate output")
+ else:
+ _, answer = strip_cot(raw)
+ return answer
+ except http.HTTPError as e:
+ last_err = e
+ except Exception as e:
+ last_err = e
+ unload_model() # force clean reload for the next attempt
+ time.sleep(2)
+ raise RuntimeError(f"model unstable after retry: {last_err}")
def degenerate(s):
"""vornith VRAM/session corruption signature: run of '?' chars."""
s = s.strip()
return len(s) >= 40 and s.count("?") / len(s) > 0.4
+def warm_model():
+ """Budget-sized heartbeat: keeps the model resident AND exercised at the canonical
+ prompt size. Tiny prompts (<50 tok) on the shared instance are the corruption
+ trigger, so the ping itself must be RAG-sized."""
+ try:
+ excerpt = ("The utility of a uniform prompt budget is that the model never "
+ "encounters a context-length distribution shift between requests. " * 9)
+ prompt = (f"{SYSTEM}\n\nBOOK EXCERPTS:\n[1] Warmup Excerpt (maintenance)\n{excerpt}"
+ "\n\nQUESTION: Reply with exactly: ok\n\nANSWER (cite [n]):")
+ r = http.post(f"{CFG['ollama_url']}/api/generate",
+ json={"model": CFG["model"], "prompt": prompt,
+ "stream": False, "options": {"num_predict": 4}},
+ timeout=120)
+ sample = r.json().get("response", "") if r.status_code == 200 else ""
+ healthy = r.status_code == 200 and not degenerate(sample)
+ return healthy, sample[:120]
+ except Exception as e:
+ return False, str(e)[:120]
+
def unload_model():
"""Drop the model from VRAM so the next request reloads clean."""
try:
diff --git a/pages.py b/pages.py
index 44802c1..a7d082e 100644
--- a/pages.py
+++ b/pages.py
@@ -140,33 +140,40 @@ async function ask(text){
add('you','user').textContent=text;
const body=add('draco','assistant');body.innerHTML='';
const srcs=document.createElement('div');srcs.className='srcs';srcs.style.display='none';
- try{
- const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
- body:JSON.stringify({q:text})});
- if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
- body.textContent='โ '+m;busy=false;go.disabled=false;return;}
- const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='',thinkEl=null;
- while(true){const{value,done}=await reader.read();if(done)break;
- buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
- for(const line of lines){if(!line.startsWith('data: '))continue;
- const ev=JSON.parse(line.slice(6));
- if(ev.type==='sources'&&ev.sources.length){
- srcs.style.display='block';
- srcs.textContent='๐ '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' ยท ');
- }else if(ev.type==='token'&&ev.channel==='think'){
- if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking';
- thinkEl.innerHTML="โ thinking ";
- body.parentNode.insertBefore(thinkEl,body);}
- thinkEl.querySelector('.tbody').textContent+=ev.text;
- log.scrollTop=log.scrollHeight;
- }else if(ev.type==='token'){acc+=ev.text;body.innerHTML='';
- body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
- log.scrollTop=log.scrollHeight;
- }else if(ev.type==='error'){acc+='\nโ '+ev.text;}
- }}
- }catch(e){acc+='\nโ connection lost';}
- body.innerHTML='';body.appendChild(document.createTextNode(acc||'(no output)'));
+ for(let attempt=1;attempt<=3;attempt++){
+ let hadTokens=false,hadError=null,acc='';
+ try{
+ const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
+ body:JSON.stringify({q:text})});
+ if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
+ body.textContent='โ '+m;busy=false;go.disabled=false;return;}
+ const reader=r.body.getReader(),dec=new TextDecoder();let buf='',thinkEl=null;
+ while(true){const{value,done}=await reader.read();if(done)break;
+ buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
+ for(const line of lines){if(!line.startsWith('data: '))continue;
+ const ev=JSON.parse(line.slice(6));
+ if(ev.type==='sources'&&ev.sources.length){
+ srcs.style.display='block';
+ srcs.textContent='๐ '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' ยท ');
+ }else if(ev.type==='token'&&ev.channel==='think'){
+ if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking';
+ thinkEl.innerHTML="โ thinking ";
+ body.parentNode.insertBefore(thinkEl,body);}
+ thinkEl.querySelector('.tbody').textContent+=ev.text;
+ log.scrollTop=log.scrollHeight;
+ }else if(ev.type==='token'){acc+=ev.text;hadTokens=true;body.innerHTML='';
+ body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
+ log.scrollTop=log.scrollHeight;
+ }else if(ev.type==='error'){hadError=ev.text;}
+ }}
+ }catch(e){hadError='connection lost';}
+ if(hadTokens&&!hadError){break;}
+ if(attempt<3){body.innerHTML='';await new Promise(r=>setTimeout(r,1500));}
+ else if(hadError){body.textContent='โ '+hadError;}
+ else break;
+ }
const cur=body.querySelector('.cursor');if(cur)cur.remove();
+ if(!body.textContent.trim())body.textContent='(no output)';
body.appendChild(srcs);
const left=document.getElementById('left');
fetch('/api/remaining').then(r=>r.json()).then(d=>left.textContent=d.remaining+' free questions left today').catch(()=>{});