vornith stability: auto-retry+reload, k=2 prompt budget, budget-sized warm ping, web auto-retry; revert keep_alive
This commit is contained in:
16
app.py
16
app.py
@@ -68,10 +68,12 @@ def api_page():
|
|||||||
|
|
||||||
@app.route("/health")
|
@app.route("/health")
|
||||||
def health():
|
def health():
|
||||||
|
# No generation probe here: a tiny probe prompt on the shared resident instance
|
||||||
|
# is the vornith corruption trigger. Model presence via /api/tags (no GPU work).
|
||||||
try:
|
try:
|
||||||
r = http.post(f"{CFG['ollama_url']}/api/generate", stream=True, timeout=4,
|
r = http.get(f"{CFG['ollama_url']}/api/tags", timeout=4)
|
||||||
json={"model": CFG["model"], "prompt": "ping", "stream": True, "options": {"num_predict": 1}})
|
models = [m.get("name") for m in r.json().get("models", [])]
|
||||||
llm = r.status_code == 200
|
llm = CFG["model"] in models
|
||||||
except Exception:
|
except Exception:
|
||||||
llm = False
|
llm = False
|
||||||
return jsonify(status="ok", llm=llm, model=CFG["model"], **core.get_stats())
|
return jsonify(status="ok", llm=llm, model=CFG["model"], **core.get_stats())
|
||||||
@@ -122,6 +124,14 @@ def chat():
|
|||||||
return Response(generate(), mimetype="text/event-stream",
|
return Response(generate(), mimetype="text/event-stream",
|
||||||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|
||||||
|
|
||||||
|
@app.route("/api/warm")
|
||||||
|
def api_warm():
|
||||||
|
"""Heartbeat target for the systemd timer; also a manual health-probe for the model lane."""
|
||||||
|
healthy, sample = core.warm_model()
|
||||||
|
if not healthy and "CUDA" not in sample:
|
||||||
|
core.unload_model() # corrupt instance: drop it; next warm reloads clean
|
||||||
|
return jsonify(healthy=healthy, sample=sample, model=CFG["model"])
|
||||||
|
|
||||||
@app.route("/api/remaining")
|
@app.route("/api/remaining")
|
||||||
def remaining():
|
def remaining():
|
||||||
ip = core.client_ip(request)
|
ip = core.client_ip(request)
|
||||||
|
|||||||
@@ -1,8 +1,9 @@
|
|||||||
{
|
{
|
||||||
"model": "ornith-1.5:9b-64k",
|
"model": "vornith:latest",
|
||||||
"ollama_url": "http://10.30.20.29:11434",
|
"ollama_url": "http://10.30.20.29:11434",
|
||||||
"num_ctx": 8192,
|
"num_ctx": 8192,
|
||||||
"rag_k": 6,
|
"rag_k": 2,
|
||||||
|
"num_predict": 1100,
|
||||||
"port": 8012,
|
"port": 8012,
|
||||||
"mcp_port": 8012,
|
"mcp_port": 8012,
|
||||||
"base_url": "https://draco.thetempleofdoom.com",
|
"base_url": "https://draco.thetempleofdoom.com",
|
||||||
|
|||||||
6
deploy/draco-warm.service
Normal file
6
deploy/draco-warm.service
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=DRACO model keep-warm ping
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/bin/curl -s -m 130 http://127.0.0.1:8012/api/warm
|
||||||
9
deploy/draco-warm.timer
Normal file
9
deploy/draco-warm.timer
Normal file
@@ -0,0 +1,9 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=DRACO keep model warm every 4 minutes
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnBootSec=60
|
||||||
|
OnUnitActiveSec=240
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -5,7 +5,7 @@ After=network.target
|
|||||||
[Service]
|
[Service]
|
||||||
Type=simple
|
Type=simple
|
||||||
WorkingDirectory=/opt/draco
|
WorkingDirectory=/opt/draco
|
||||||
ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 app:app
|
ExecStart=/opt/draco/venv/bin/gunicorn -w 2 --threads 8 -b 127.0.0.1:8012 --timeout 300 --graceful-timeout 10 app:app
|
||||||
Restart=always
|
Restart=always
|
||||||
RestartSec=5
|
RestartSec=5
|
||||||
Environment=PYTHONUNBUFFERED=1
|
Environment=PYTHONUNBUFFERED=1
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ CREATE TABLE IF NOT EXISTS users (
|
|||||||
api_key VARCHAR(64) UNIQUE NOT NULL,
|
api_key VARCHAR(64) UNIQUE NOT NULL,
|
||||||
credits INTEGER DEFAULT 0,
|
credits INTEGER DEFAULT 0,
|
||||||
free_used INTEGER DEFAULT 0,
|
free_used INTEGER DEFAULT 0,
|
||||||
|
total_calls INTEGER DEFAULT 0,
|
||||||
is_admin INTEGER DEFAULT 0,
|
is_admin INTEGER DEFAULT 0,
|
||||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||||
last_used_at TIMESTAMP
|
last_used_at TIMESTAMP
|
||||||
@@ -58,6 +59,11 @@ def db():
|
|||||||
def init_db():
|
def init_db():
|
||||||
c = db()
|
c = db()
|
||||||
c.executescript(SCHEMA)
|
c.executescript(SCHEMA)
|
||||||
|
try: # in-place migration for DBs created before total_calls existed
|
||||||
|
c.execute("SELECT total_calls FROM users LIMIT 1")
|
||||||
|
except sqlite3.OperationalError:
|
||||||
|
c.execute("ALTER TABLE users ADD COLUMN total_calls INTEGER DEFAULT 0")
|
||||||
|
c.commit()
|
||||||
c.execute("INSERT OR IGNORE INTO users (email, api_key, credits, is_admin) VALUES (?,?,?,1)",
|
c.execute("INSERT OR IGNORE INTO users (email, api_key, credits, is_admin) VALUES (?,?,?,1)",
|
||||||
("admin@draco.local", CFG["admin_key"], 999999))
|
("admin@draco.local", CFG["admin_key"], 999999))
|
||||||
c.commit()
|
c.commit()
|
||||||
@@ -127,12 +133,17 @@ SYSTEM = (
|
|||||||
"Answer directly with no reasoning preamble and no meta commentary."
|
"Answer directly with no reasoning preamble and no meta commentary."
|
||||||
)
|
)
|
||||||
|
|
||||||
def build_prompt(question, k=6):
|
def build_prompt(question, k=None):
|
||||||
|
"""RAG prompt. PROMPT BUDGET RULE: all prompts must land in the same size class
|
||||||
|
(~600 tok). vornith (linear-attn+MTP hybrid) corrupts when one resident instance
|
||||||
|
receives mixed tiny/long prompts; uniform bounded prompts are proven stable."""
|
||||||
|
k = k or CFG.get("rag_k", 2)
|
||||||
|
k = min(k, 3) # hard cap: 3 x 1200-char excerpts ~= clean-zone prompt
|
||||||
hits = search_library(question, k=k)
|
hits = search_library(question, k=k)
|
||||||
if hits:
|
if hits:
|
||||||
blocks = []
|
blocks = []
|
||||||
for n, h in enumerate(hits, 1):
|
for n, h in enumerate(hits, 1):
|
||||||
blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1400]}")
|
blocks.append(f"[{n}] {h['title']} ({h['category']})\n{h['text'][:1200]}")
|
||||||
ctx = "\n\n".join(blocks)
|
ctx = "\n\n".join(blocks)
|
||||||
prompt = f"{SYSTEM}\n\nBOOK EXCERPTS:\n{ctx}\n\nQUESTION: {question}\n\nANSWER (cite [n]):"
|
prompt = f"{SYSTEM}\n\nBOOK EXCERPTS:\n{ctx}\n\nQUESTION: {question}\n\nANSWER (cite [n]):"
|
||||||
else:
|
else:
|
||||||
@@ -314,6 +325,9 @@ def stream_ollama(prompt):
|
|||||||
"""Yield (channel, piece) tuples: channel 'think' or 'answer'."""
|
"""Yield (channel, piece) tuples: channel 'think' or 'answer'."""
|
||||||
r = http.post(f"{CFG['ollama_url']}/api/generate",
|
r = http.post(f"{CFG['ollama_url']}/api/generate",
|
||||||
json={"model": CFG["model"], "prompt": prompt, "stream": True,
|
json={"model": CFG["model"], "prompt": prompt, "stream": True,
|
||||||
|
# RESIDENT model: cold-load prefill >1k tokens CUDA-crashes on this
|
||||||
|
# hybrid arch. Corruption (????? output) is handled by detection +
|
||||||
|
# auto-reload in the app layer instead.
|
||||||
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
|
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
|
||||||
"num_ctx": CFG.get("num_ctx", 8192)}},
|
"num_ctx": CFG.get("num_ctx", 8192)}},
|
||||||
timeout=(5, None), stream=True)
|
timeout=(5, None), stream=True)
|
||||||
@@ -328,24 +342,55 @@ def stream_ollama(prompt):
|
|||||||
yield channel, piece
|
yield channel, piece
|
||||||
|
|
||||||
def ask_ollama(prompt):
|
def ask_ollama(prompt):
|
||||||
"""Non-streaming RAG answer (reasoning stripped, tagged or not)."""
|
"""Non-streaming RAG answer with auto-recovery: on degenerate/CUDA failure,
|
||||||
|
unload the model, reload fresh, retry once before surfacing an error."""
|
||||||
|
last_err = None
|
||||||
|
for attempt in (1, 2):
|
||||||
|
try:
|
||||||
r = http.post(f"{CFG['ollama_url']}/api/generate",
|
r = http.post(f"{CFG['ollama_url']}/api/generate",
|
||||||
json={"model": CFG["model"], "prompt": prompt, "stream": False,
|
json={"model": CFG["model"], "prompt": prompt, "stream": False,
|
||||||
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
|
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
|
||||||
"num_ctx": CFG.get("num_ctx", 8192)}},
|
"num_ctx": CFG.get("num_ctx", 8192)}},
|
||||||
timeout=180)
|
timeout=180)
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
_, answer = strip_cot(r.json().get("response", ""))
|
raw = r.json().get("response", "")
|
||||||
if degenerate(answer):
|
if raw and degenerate(raw):
|
||||||
unload_model()
|
last_err = RuntimeError("degenerate output")
|
||||||
raise RuntimeError("model returned degenerate output — model unloaded, retry")
|
else:
|
||||||
|
_, answer = strip_cot(raw)
|
||||||
return answer
|
return answer
|
||||||
|
except http.HTTPError as e:
|
||||||
|
last_err = e
|
||||||
|
except Exception as e:
|
||||||
|
last_err = e
|
||||||
|
unload_model() # force clean reload for the next attempt
|
||||||
|
time.sleep(2)
|
||||||
|
raise RuntimeError(f"model unstable after retry: {last_err}")
|
||||||
|
|
||||||
def degenerate(s):
|
def degenerate(s):
|
||||||
"""vornith VRAM/session corruption signature: run of '?' chars."""
|
"""vornith VRAM/session corruption signature: run of '?' chars."""
|
||||||
s = s.strip()
|
s = s.strip()
|
||||||
return len(s) >= 40 and s.count("?") / len(s) > 0.4
|
return len(s) >= 40 and s.count("?") / len(s) > 0.4
|
||||||
|
|
||||||
|
def warm_model():
|
||||||
|
"""Budget-sized heartbeat: keeps the model resident AND exercised at the canonical
|
||||||
|
prompt size. Tiny prompts (<50 tok) on the shared instance are the corruption
|
||||||
|
trigger, so the ping itself must be RAG-sized."""
|
||||||
|
try:
|
||||||
|
excerpt = ("The utility of a uniform prompt budget is that the model never "
|
||||||
|
"encounters a context-length distribution shift between requests. " * 9)
|
||||||
|
prompt = (f"{SYSTEM}\n\nBOOK EXCERPTS:\n[1] Warmup Excerpt (maintenance)\n{excerpt}"
|
||||||
|
"\n\nQUESTION: Reply with exactly: ok\n\nANSWER (cite [n]):")
|
||||||
|
r = http.post(f"{CFG['ollama_url']}/api/generate",
|
||||||
|
json={"model": CFG["model"], "prompt": prompt,
|
||||||
|
"stream": False, "options": {"num_predict": 4}},
|
||||||
|
timeout=120)
|
||||||
|
sample = r.json().get("response", "") if r.status_code == 200 else ""
|
||||||
|
healthy = r.status_code == 200 and not degenerate(sample)
|
||||||
|
return healthy, sample[:120]
|
||||||
|
except Exception as e:
|
||||||
|
return False, str(e)[:120]
|
||||||
|
|
||||||
def unload_model():
|
def unload_model():
|
||||||
"""Drop the model from VRAM so the next request reloads clean."""
|
"""Drop the model from VRAM so the next request reloads clean."""
|
||||||
try:
|
try:
|
||||||
|
|||||||
17
pages.py
17
pages.py
@@ -140,12 +140,14 @@ async function ask(text){
|
|||||||
add('you','user').textContent=text;
|
add('you','user').textContent=text;
|
||||||
const body=add('draco','assistant');body.innerHTML='<span class=cursor></span>';
|
const body=add('draco','assistant');body.innerHTML='<span class=cursor></span>';
|
||||||
const srcs=document.createElement('div');srcs.className='srcs';srcs.style.display='none';
|
const srcs=document.createElement('div');srcs.className='srcs';srcs.style.display='none';
|
||||||
|
for(let attempt=1;attempt<=3;attempt++){
|
||||||
|
let hadTokens=false,hadError=null,acc='';
|
||||||
try{
|
try{
|
||||||
const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
|
const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
|
||||||
body:JSON.stringify({q:text})});
|
body:JSON.stringify({q:text})});
|
||||||
if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
|
if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
|
||||||
body.textContent='⚠ '+m;busy=false;go.disabled=false;return;}
|
body.textContent='⚠ '+m;busy=false;go.disabled=false;return;}
|
||||||
const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='',thinkEl=null;
|
const reader=r.body.getReader(),dec=new TextDecoder();let buf='',thinkEl=null;
|
||||||
while(true){const{value,done}=await reader.read();if(done)break;
|
while(true){const{value,done}=await reader.read();if(done)break;
|
||||||
buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
|
buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
|
||||||
for(const line of lines){if(!line.startsWith('data: '))continue;
|
for(const line of lines){if(!line.startsWith('data: '))continue;
|
||||||
@@ -159,14 +161,19 @@ async function ask(text){
|
|||||||
body.parentNode.insertBefore(thinkEl,body);}
|
body.parentNode.insertBefore(thinkEl,body);}
|
||||||
thinkEl.querySelector('.tbody').textContent+=ev.text;
|
thinkEl.querySelector('.tbody').textContent+=ev.text;
|
||||||
log.scrollTop=log.scrollHeight;
|
log.scrollTop=log.scrollHeight;
|
||||||
}else if(ev.type==='token'){acc+=ev.text;body.innerHTML='';
|
}else if(ev.type==='token'){acc+=ev.text;hadTokens=true;body.innerHTML='';
|
||||||
body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
|
body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
|
||||||
log.scrollTop=log.scrollHeight;
|
log.scrollTop=log.scrollHeight;
|
||||||
}else if(ev.type==='error'){acc+='\n⚠ '+ev.text;}
|
}else if(ev.type==='error'){hadError=ev.text;}
|
||||||
}}
|
}}
|
||||||
}catch(e){acc+='\n⚠ connection lost';}
|
}catch(e){hadError='connection lost';}
|
||||||
body.innerHTML='';body.appendChild(document.createTextNode(acc||'(no output)'));
|
if(hadTokens&&!hadError){break;}
|
||||||
|
if(attempt<3){body.innerHTML='<span class=cursor></span>';await new Promise(r=>setTimeout(r,1500));}
|
||||||
|
else if(hadError){body.textContent='⚠ '+hadError;}
|
||||||
|
else break;
|
||||||
|
}
|
||||||
const cur=body.querySelector('.cursor');if(cur)cur.remove();
|
const cur=body.querySelector('.cursor');if(cur)cur.remove();
|
||||||
|
if(!body.textContent.trim())body.textContent='(no output)';
|
||||||
body.appendChild(srcs);
|
body.appendChild(srcs);
|
||||||
const left=document.getElementById('left');
|
const left=document.getElementById('left');
|
||||||
fetch('/api/remaining').then(r=>r.json()).then(d=>left.textContent=d.remaining+' free questions left today').catch(()=>{});
|
fetch('/api/remaining').then(r=>r.json()).then(d=>left.textContent=d.remaining+' free questions left today').catch(()=>{});
|
||||||
|
|||||||
Reference in New Issue
Block a user