DRACO v1: book-forged code oracle — RAG over 500+ books, REST API, MCP, BTCPay, full agent SEO kit

This commit is contained in:
drjones
2026-10-01 17:25:05 -07:00
commit e5d85a912e
13 changed files with 1604 additions and 0 deletions

77
build_index.py Normal file
View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
"""draco index build — chunk all extracted book text and build a BM25 index.
Run on the CT after ingest.py: python3 build_index.py [--books /opt/books]
Output: /opt/books/index/chunks.json + index.pkl (book_count.json = stats)
"""
import argparse, json, os, pickle, re, time
BOOKS_DIR = "/opt/books"
INDEX_DIR = os.path.join(BOOKS_DIR, "index")
CHUNK_TOKENS, CHUNK_OVERLAP = 300, 60
def chunk_text(text, size=CHUNK_TOKENS, overlap=CHUNK_OVERLAP):
words = text.split()
if len(words) <= size:
return [" ".join(words)] if words else []
out = []
step = size - overlap
for i in range(0, len(words), step):
c = words[i:i + size]
if len(c) >= 40:
out.append(" ".join(c))
if i + size >= len(words):
break
return out
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--books", default=BOOKS_DIR)
a = ap.parse_args()
books_dir = a.books
index_dir = os.path.join(books_dir, "index")
os.makedirs(index_dir, exist_ok=True)
from rank_bm25 import BM25Okapi
from nltk.stem import SnowballStemmer
t0 = time.time()
manifest = json.load(open(os.path.join(books_dir, "manifest.json")))
usable = [m for m in manifest if m.get("txt")]
print(f"index: {len(usable)} usable books from manifest", flush=True)
stemmer = SnowballStemmer("english")
stop = set("""a an and are as at be by for from has have he her his how i in is it its of on or she that the this to was were what when where which who will with you your our we they them not no but can could should would may might must do does did done if then than into over under about after before between during through against without within""".split())
def toks(s):
return [stemmer.stem(w) for w in re.findall(r"[a-z0-9_#+.-]{2,}", s.lower()) if w not in stop]
chunk_records, texts = [], []
CAP = 400 # max chunks indexed per book (≈480 pages) — keeps the fit inside 8GB
for m in usable:
try:
raw = open(os.path.join(books_dir, "text", m["txt"]), encoding="utf-8", errors="replace").read()
except OSError:
continue
for j, ch in enumerate(chunk_text(raw)[:CAP]):
chunk_records.append({"book": m["id"], "chunk": j})
texts.append(ch)
print(f"index: {len(texts)} chunks from {len(usable)} books", flush=True)
print("index: tokenizing...", flush=True)
corpus = [toks(t) for t in texts]
print("index: fitting BM25...", flush=True)
bm25 = BM25Okapi(corpus)
with open(os.path.join(index_dir, "chunks.json"), "w") as f:
json.dump(chunk_records, f)
with open(os.path.join(index_dir, "index.pkl"), "wb") as f:
pickle.dump({"bm25": bm25, "texts": texts}, f, protocol=4)
stats = {"books": len(usable), "chunks": len(texts),
"chars": sum(m["chars"] for m in usable), "built": time.strftime("%Y-%m-%dT%H:%M:%S")}
with open(os.path.join(books_dir, "book_count.json"), "w") as f:
json.dump(stats, f)
print(f"index DONE: {stats} in {time.time()-t0:.0f}s", flush=True)
if __name__ == "__main__":
main()