#!/usr/bin/env python3
# COOKIE VAULT v5 — clean UI, session-1 chrome launch bridge, LLM fallback parse
import os, re, json, sqlite3, time, datetime, threading, subprocess, urllib.request
from flask import Flask, request, render_template_string, redirect, url_for, jsonify, Response
APP_DIR = r"C:\cookievault"
DB_PATH = os.path.join(APP_DIR, "cookies.db")
WATCH_DIR = os.path.join(APP_DIR, "incoming")
SESSION_DIR = os.path.join(APP_DIR, "sessions")
WORK_DRIVE = r"D:\\"
BRIDGE = "http://127.0.0.1:5067/launch"
OLLAMA_HOSTS = ["10.30.20.222", "10.30.20.29"]
LLM_MODEL = "ornith-1.5:9b-64k"
os.makedirs(WATCH_DIR, exist_ok=True)
os.makedirs(SESSION_DIR, exist_ok=True)
app = Flask(__name__)
DOMAIN_RE = re.compile(r"^\.?([a-z0-9]([a-z0-9-]*[a-z0-9])?\.)+[a-z]{2,}$", re.I)
def db():
c = sqlite3.connect(DB_PATH, timeout=30)
c.execute("""CREATE TABLE IF NOT EXISTS cookies(
id INTEGER PRIMARY KEY AUTOINCREMENT,
domain TEXT, flag TEXT, path TEXT, secure TEXT,
expiry INTEGER, name TEXT, value TEXT,
source_file TEXT, imported_at INTEGER,
UNIQUE(domain, path, name, source_file))""")
c.execute("CREATE INDEX IF NOT EXISTS ix_dom ON cookies(domain)")
return c
def utc(ts):
try: return datetime.datetime.utcfromtimestamp(int(float(ts))).strftime("%Y-%m-%d")
except Exception: return "session"
def _safe_expiry(val):
try:
iv = int(float(val))
return iv if iv > 0 else 0
except (TypeError, ValueError):
return 0
def parse_netscape(text):
out = []
for line in text.splitlines():
line = line.strip()
if not line or line.startswith("# ") or line == "#HttpOnly_":
continue
if line.startswith("#HttpOnly_"):
line = line[len("#HttpOnly_"):]
elif line.startswith("#"):
continue
parts = line.split("\t") if "\t" in line else line.split()
if len(parts) < 7:
continue
domain, flag, path, secure, expiry, name, value = parts[:7]
if len(parts) > 7:
value = "\t".join(parts[6:])
out.append((domain, flag, path, secure, expiry, name, value))
return out
def sniff_netscape(text):
"""Deterministic: does the first data line look like 7-col cookie format?"""
seen = 0
for line in text.splitlines():
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t") if "\t" in line else line.split()
if len(parts) >= 7 and parts[4].isdigit():
return True
seen += 1
if seen > 5:
return False
return False
def llm_classify(text):
"""Fallback: ask local ornith whether the file holds cookie data. Returns True/False/None(fail)."""
prompt = ("You are a file classifier. Does this text contain browser cookie data in Netscape "
"format (7 tab-separated fields: domain, TRUE/FALSE, path, TRUE/FALSE, numeric expiry, name, value)? "
"Answer exactly one word: YES or NO.\n\n" + text[:2000])
for host in OLLAMA_HOSTS:
try:
req = urllib.request.Request(f"http://{host}:11434/api/generate",
data=json.dumps({"model": LLM_MODEL, "prompt": prompt, "stream": False, "think": False}).encode(),
headers={"Content-Type": "application/json"})
with urllib.request.urlopen(req, timeout=45) as r:
ans = json.load(r)["response"].strip().upper()
if "YES" in ans[:6]: return True
if "NO" in ans[:6]: return False
if ans and "?" not in ans[:6]: return False
except Exception:
continue
return None
def import_file(path, use_llm_fallback=True):
fname = os.path.basename(path)
with open(path, "r", encoding="utf-8", errors="replace") as f:
text = f.read()
if not sniff_netscape(text):
if not (use_llm_fallback and llm_classify(text)):
return 0, 0, None # not a cookie file
cookies = parse_netscape(text)
clean = []
for ck in cookies:
if not ck[5] or not ck[6]:
continue
if not re.match(r"^[0-9]+$", (ck[4] or "0").strip()):
continue
if not DOMAIN_RE.match(ck[0] or ""):
continue
clean.append(ck)
cookies = clean
if not cookies:
return 0, 0, None
c = db()
before = c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0]
for ck in cookies:
c.execute("INSERT OR IGNORE INTO cookies(domain,flag,path,secure,expiry,name,value,source_file,imported_at) VALUES(?,?,?,?,?,?,?,?,?)",
(*ck, fname, int(time.time())))
c.commit()
after = c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0]
c.close()
return len(cookies), after - before, after
# ---------------- UI ----------------
STYLE = """"""
HOME_PAGE = STYLE + r"""
Scans this folder + subfolders. Any file with cookie data gets imported — any filename, any format (LLM-assisted detection).
"""
PAGE_SIZE = 40
@app.route("/", methods=["GET"])
def index():
q = request.args.get("q", "").strip()
try: page = max(1, int(request.args.get("page", 1)))
except ValueError: page = 1
like = f"%{q}%" if q else "%"
c = db()
if q:
total = c.execute("SELECT COUNT(DISTINCT domain) FROM cookies WHERE domain LIKE ?", (like,)).fetchone()[0]
domains = c.execute("""SELECT domain, COUNT(*), MAX(CASE WHEN expiry GLOB '[0-9]*' THEN CAST(expiry AS INTEGER) ELSE 0 END)
FROM cookies WHERE domain LIKE ? GROUP BY domain
HAVING COUNT(*) >= 2 AND domain GLOB '*.*.*' OR (domain LIKE ? AND COUNT(*) >= 1)
ORDER BY COUNT(*) DESC, domain LIMIT ? OFFSET ?""",
(like, like, PAGE_SIZE, (page-1)*PAGE_SIZE)).fetchall()
else:
total = c.execute("SELECT COUNT(DISTINCT domain) FROM cookies").fetchone()[0]
domains = c.execute("""SELECT domain, COUNT(*), MAX(CASE WHEN expiry GLOB '[0-9]*' THEN CAST(expiry AS INTEGER) ELSE 0 END)
FROM cookies GROUP BY domain
HAVING COUNT(*) >= 3 AND domain GLOB '*.*.*'
ORDER BY COUNT(*) DESC, domain LIMIT ? OFFSET ?""",
(PAGE_SIZE, (page-1)*PAGE_SIZE)).fetchall()
stats = (c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0],
c.execute("SELECT COUNT(DISTINCT domain) FROM cookies").fetchone()[0],
c.execute("SELECT COUNT(DISTINCT source_file) FROM cookies").fetchone()[0])
c.close()
return render_template_string(HOME_PAGE, domains=domains, stats=stats, q=q, page=page,
has_more=(page*PAGE_SIZE) < total, msg=request.args.get("msg"), utc=utc)
@app.route("/import_page")
def import_page():
return render_template_string(IMPORT_PAGE, msg=request.args.get("msg"))
@app.route("/import", methods=["POST"])
def do_import():
msg, errs = [], []
for f in request.files.getlist("files"):
try:
fname = os.path.basename(f.filename or "").strip()
fname = re.sub(r"[^A-Za-z0-9._\- ]", "_", fname)
if not fname or fname == ".":
data = f.read()
if not data.strip(): continue
fname = "unnamed_%d.txt" % (int(time.time()*1000) % 10**9)
path = os.path.join(WATCH_DIR, fname)
with open(path, "wb") as fh: fh.write(data)
else:
path = os.path.join(WATCH_DIR, fname)
f.save(path)
n, new, total = import_file(path)
msg.append(f"{fname}: {new} new" if total is not None else f"{fname}: no cookie data found")
except Exception as e:
errs.append(f"{getattr(f,'filename','?')}: {e}")
paste = request.form.get("paste", "").strip()
if paste:
tmp = os.path.join(WATCH_DIR, "_pasted_%d.txt" % int(time.time()))
with open(tmp, "w", encoding="utf-8") as fh: fh.write(paste)
n, new, total = import_file(tmp)
msg.append(f"pasted: {new} new" if total is not None else "pasted: no cookie data detected")
msg += ["%s" % e for e in errs]
return redirect(url_for("index", msg=" · ".join(msg) or "nothing imported"))
@app.route("/scan", methods=["POST"])
def scan():
msg = []
for fn in os.listdir(WATCH_DIR):
if fn.lower().endswith((".txt", ".json")) and not fn.startswith("_pasted"):
n, new, total = import_file(os.path.join(WATCH_DIR, fn))
msg.append(f"{fn}: {new} new")
return redirect(url_for("index", msg=" · ".join(msg) or "nothing new"))
@app.route("/browse")
def browse():
sub = request.args.get("path", "").strip()
base = os.path.abspath(WORK_DRIVE)
target = os.path.abspath(os.path.join(base, sub.lstrip("/\\"))) if sub else base
if not target.startswith(base) or not os.path.isdir(target):
return redirect(url_for("index", msg="invalid folder"))
entries = []
for name in sorted(os.listdir(target)):
if os.path.isdir(os.path.join(target, name)) and name not in ("$RECYCLE.BIN", "System Volume Information"):
entries.append(name)
rel = os.path.relpath(target, base)
parent = os.path.dirname(rel) if rel != "." else None
return render_template_string(BROWSE_PAGE, entries=entries, rel=rel, parent=parent)
@app.route("/scan_folder", methods=["POST"])
def scan_folder():
sub = request.form.get("path", "").strip()
base = os.path.abspath(WORK_DRIVE)
target = os.path.abspath(os.path.join(base, sub.lstrip("/\\"))) if sub else base
if not target.startswith(base) or not os.path.isdir(target):
return redirect(url_for("index", msg="invalid folder"))
return _smart_scan(target)
@app.route("/scan_work", methods=["POST"])
def scan_work():
if not os.path.isdir(WORK_DRIVE):
return redirect(url_for("index", msg="Work drive (D:\\) not plugged in"))
return _smart_scan(WORK_DRIVE)
def _smart_scan(root):
found = []
for rootpath, dirs, files in os.walk(root):
dirs[:] = [d for d in dirs if d not in ("$RECYCLE.BIN", "System Volume Information", "node_modules")]
for fn in files:
if fn.lower().endswith((".txt", ".json", ".log")):
found.append(os.path.join(rootpath, fn))
if not found:
return redirect(url_for("index", msg="no text files found in that folder"))
msg, errs = [], []
new_total, imported_files, llm_used = 0, 0, 0
for p in found:
try:
with open(p, "r", encoding="utf-8", errors="replace") as f:
text = f.read(65536)
if not sniff_netscape(text):
# deterministic miss -> try LLM only for small-ish files (cost control)
if os.path.getsize(p) > 200_000:
continue
verdict = llm_classify(text)
if verdict is None:
continue
llm_used += 1
if not verdict:
continue
with open(p, "r", encoding="utf-8", errors="replace") as f:
pass # full read happens in import_file
n, new, total = import_file(p)
if total is None:
continue
imported_files += 1
new_total += new
try: shown = os.path.relpath(p, root)
except ValueError: shown = p
msg.append(f"{shown}: {new} new")
except Exception as e:
errs.append(f"{p}: {e}")
if not imported_files and not errs:
return redirect(url_for("index", msg=f"scanned {len(found)} files — no cookie data found (LLM checked {llm_used} ambiguous ones)"))
msg += ["%s" % e for e in errs]
return redirect(url_for("index", msg=f"{imported_files} cookie files ({new_total} new cookies, LLM-assisted on {llm_used}) · " + " · ".join(msg[:20])))
# ---------------- LOGIN (session-1 bridge) ----------------
@app.route("/sessions")
def sessions():
domain = request.args.get("domain", "").strip()
if not domain:
return redirect(url_for("index"))
stem = domain.strip(".")
c = db()
sess = c.execute("""SELECT source_file, COUNT(*) as n,
MAX(CASE WHEN expiry GLOB '[0-9]*' AND CAST(expiry AS INTEGER) > strftime('%s','now') THEN 1 ELSE 0 END) as has_fresh
FROM cookies WHERE domain LIKE ? GROUP BY source_file ORDER BY n DESC LIMIT 300""",
("%" + stem + "%",)).fetchall()
total = c.execute("SELECT COUNT(*) FROM cookies WHERE domain LIKE ?", ("%" + stem + "%",)).fetchone()[0]
c.close()
return render_template_string(SESSIONS_PAGE, domain=domain, sess=sess, total=total)
SESSIONS_PAGE = STYLE + r"""Sessions — COOKIE VAULT