v4: folder picker for Work drive + smart content-sniff scan (any .txt, any filename)

This commit is contained in:
2026-09-25 08:11:21 -07:00
parent e6181d891a
commit 03a4d3204b

177
app.py
View File

@@ -8,6 +8,8 @@ APP_DIR = r"C:\cookievault"
DB_PATH = os.path.join(APP_DIR, "cookies.db") DB_PATH = os.path.join(APP_DIR, "cookies.db")
WATCH_DIR = os.path.join(APP_DIR, "incoming") WATCH_DIR = os.path.join(APP_DIR, "incoming")
SESSION_DIR = os.path.join(APP_DIR, "sessions") SESSION_DIR = os.path.join(APP_DIR, "sessions")
WORK_DRIVE = r"D:\\" # the "Work" removable drive
SCAN_ROOTS = [WATCH_DIR, WORK_DRIVE]
os.makedirs(WATCH_DIR, exist_ok=True) os.makedirs(WATCH_DIR, exist_ok=True)
os.makedirs(SESSION_DIR, exist_ok=True) os.makedirs(SESSION_DIR, exist_ok=True)
@@ -49,11 +51,28 @@ def parse_netscape(text):
out.append((domain, flag, path, secure, expiry, name, value)) out.append((domain, flag, path, secure, expiry, name, value))
return out return out
def _safe_expiry(val):
try:
iv = int(float(val))
return iv if iv > 0 else 0
except (TypeError, ValueError):
return 0
def import_file(path): def import_file(path):
fname = os.path.basename(path) fname = os.path.basename(path)
with open(path, "r", encoding="utf-8", errors="replace") as f: with open(path, "r", encoding="utf-8", errors="replace") as f:
text = f.read() text = f.read()
cookies = parse_netscape(text) cookies = parse_netscape(text)
# guard: every cookie must have sane types (name+value non-empty, expiry numeric-ish)
clean = []
for ck in cookies:
if not ck[5] or not ck[6]:
continue
if re.match(r"^[0-9]+$", ck[4].strip() or "0") is None:
# header junk line that survived parsing (e.g. split words) — skip
continue
clean.append(ck)
cookies = clean
c = db() c = db()
before = c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0] before = c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0]
for ck in cookies: for ck in cookies:
@@ -98,7 +117,10 @@ td.val{color:#9ece6a;max-width:240px;overflow:hidden;text-overflow:ellipsis;whit
</div> </div>
<p><button type=submit>IMPORT</button> <span class=muted>also watches C:\cookievault\incoming &mdash; dump files there and hit rescan</span></p> <p><button type=submit>IMPORT</button> <span class=muted>also watches C:\cookievault\incoming &mdash; dump files there and hit rescan</span></p>
</form> </form>
<form method=post action=/scan style=display:inline><button class="btn gray" type=submit>RESCAN incoming folder</button></form> <form method=post action=/scan style=display:inline><button class="btn gray" type=submit>RESCAN incoming</button></form>
<a class=btn href="/browse">&#128193; PICK FOLDER ON WORK DRIVE</a>
<form method=post action=/scan_work style=display:inline><button type=submit>&#128269; SMART-SCAN ALL OF D:\</button></form>
<span class=muted>smart scan: any .txt containing Netscape cookies gets imported, whatever its name</span>
{% if msg %}<p class=ok>{{msg|safe}}</p>{% endif %} {% if msg %}<p class=ok>{{msg|safe}}</p>{% endif %}
</div> </div>
<div class=box> <div class=box>
@@ -124,16 +146,48 @@ drop.addEventListener('drop',ev=>{m.files=ev.dataTransfer.files;document.getElem
m.addEventListener('change',()=>document.getElementById('f').submit()); m.addEventListener('change',()=>document.getElementById('f').submit());
</script></body></html>""" </script></body></html>"""
BROWSE_PAGE = r"""<!doctype html><html><head><title>Pick a folder — COOKIE VAULT</title><style>
body{background:#0d0f14;color:#d7dae0;font-family:Consolas,monospace;margin:0}
header{background:#141821;padding:18px 24px;border-bottom:1px solid #232a36}
h1{margin:0 0 4px;font-size:20px;color:#7aa2f7;letter-spacing:2px}
.crumb{color:#565f89;font-size:13px}
.crumb a{color:#7aa2f7;text-decoration:none}
main{max-width:800px;margin:24px auto;padding:0 16px}
ul{list-style:none;padding:0}
li{border-bottom:1px solid #1a2029}
li a{display:block;padding:10px 8px;color:#e0af68;text-decoration:none;font-size:15px}
li a:hover{background:#141821}
.btnrow{margin-top:18px}
button,.btn{background:#7aa2f7;color:#0d0f14;border:0;padding:9px 18px;border-radius:6px;font-weight:bold;cursor:pointer;font-family:inherit;font-size:14px;text-decoration:none;display:inline-block}
.btn.green{background:#9ece6a}
.muted{color:#565f89;font-size:13px}
</style></head><body>
<header><h1>&#128193; Pick a folder on WORK (D:\)</h1>
<div class=crumb>D:\ {% if rel != '.' %}&raquo; {{rel}}{% endif %}</div></header>
<main>
{% if parent is not none %}<div style=margin-bottom:10px><a class=btn href="/browse?path={{parent|urlencode}}">&#8592; UP</a></div>{% endif %}
<ul>
{% for e in entries %}<li><a href="/browse?path={{(rel ~ '/' ~ e) if rel != '.' else e|urlencode}}">&#128193; {{e}}/</a></li>{% endfor %}
{% if not entries %}<li class=muted style=padding:10px>no subfolders here</li>{% endif %}
</ul>
<div class=btnrow>
<form method=post action=/scan_folder><input type=hidden name=path value="{{rel}}">
<button class=green>SMART-SCAN THIS FOLDER &#128269;</button></form>
<a class=btn href="/" style=margin-left:8px>Cancel</a>
</div>
<p class=muted>Scans this folder + all subfolders. Any .txt that contains Netscape cookies (any filename) gets imported.</p>
</main></body></html>"""
@app.route("/", methods=["GET"]) @app.route("/", methods=["GET"])
def index(): def index():
q = request.args.get("q", "").strip() q = request.args.get("q", "").strip()
c = db() c = db()
if q: if q:
like = f"%{q}%" like = f"%{q}%"
domains = c.execute("""SELECT domain, COUNT(*), MAX(CAST(expiry AS INTEGER)) FROM cookies domains = c.execute("""SELECT domain, COUNT(*), MAX(CASE WHEN expiry GLOB '[0-9]*' THEN CAST(expiry AS INTEGER) ELSE 0 END) FROM cookies
WHERE domain LIKE ? GROUP BY domain ORDER BY domain LIMIT 500""", (like,)).fetchall() WHERE domain LIKE ? GROUP BY domain ORDER BY domain LIMIT 500""", (like,)).fetchall()
else: else:
domains = c.execute("""SELECT domain, COUNT(*), MAX(CAST(expiry AS INTEGER)) FROM cookies domains = c.execute("""SELECT domain, COUNT(*), MAX(CASE WHEN expiry GLOB '[0-9]*' THEN CAST(expiry AS INTEGER) ELSE 0 END) FROM cookies
GROUP BY domain ORDER BY domain LIMIT 500""").fetchall() GROUP BY domain ORDER BY domain LIMIT 500""").fetchall()
stats = (c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0], stats = (c.execute("SELECT COUNT(*) FROM cookies").fetchone()[0],
c.execute("SELECT COUNT(DISTINCT domain) FROM cookies").fetchone()[0], c.execute("SELECT COUNT(DISTINCT domain) FROM cookies").fetchone()[0],
@@ -182,16 +236,98 @@ def do_import():
@app.route("/scan", methods=["POST"]) @app.route("/scan", methods=["POST"])
def scan(): def scan():
return _scan_dirs(SCAN_ROOTS)
@app.route("/scan_work", methods=["POST"])
def scan_work():
"""Deep-scan the Work drive: every .txt file that LOOKS like Netscape cookie format."""
return _smart_scan(WORK_DRIVE)
@app.route("/browse", methods=["GET"])
def browse():
"""List subfolders of a path on the Work drive so the user can pick one."""
sub = request.args.get("path", "").strip()
base = os.path.abspath(WORK_DRIVE)
target = os.path.abspath(os.path.join(base, sub.lstrip("/\\"))) if sub else base
if not target.startswith(base) or not os.path.isdir(target):
return redirect(url_for("index", msg="<span class=warn>invalid folder</span>"))
entries = []
try:
for name in sorted(os.listdir(target)):
full = os.path.join(target, name)
if os.path.isdir(full) and name not in ("$RECYCLE.BIN", "System Volume Information"):
entries.append(name)
except Exception as e:
return redirect(url_for("index", msg=f"<span class=warn>{e}</span>"))
rel = os.path.relpath(target, base)
parent = os.path.dirname(rel) if rel != "." else None
return render_template_string(BROWSE_PAGE, entries=entries, rel=rel, parent=parent)
@app.route("/scan_folder", methods=["POST"])
def scan_folder():
sub = request.form.get("path", "").strip()
base = os.path.abspath(WORK_DRIVE)
target = os.path.abspath(os.path.join(base, sub.lstrip("/\\"))) if sub else base
if not target.startswith(base) or not os.path.isdir(target):
return redirect(url_for("index", msg="<span class=warn>invalid folder</span>"))
return _smart_scan(target)
def _smart_scan(root):
"""Smart scan: walk root, sniff every .txt for Netscape format regardless of filename."""
found = []
for rootpath, dirs, files in os.walk(root):
dirs[:] = [d for d in dirs if d not in ("$RECYCLE.BIN", "System Volume Information", "node_modules")]
for fn in files:
if fn.lower().endswith(".txt"):
found.append(os.path.join(rootpath, fn))
if not found:
return redirect(url_for("index", msg="no .txt files found in that folder"))
msg, errs = [], [] msg, errs = [], []
for fn in os.listdir(WATCH_DIR): new_total, imported_files = 0, 0
if fn.lower().endswith((".txt", ".json")): for p in found:
try:
looks_like = False
with open(p, "r", encoding="utf-8", errors="replace") as f:
for i, line in enumerate(f):
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t") if "\t" in line else line.split()
if len(parts) >= 7 and parts[4].isdigit():
looks_like = True
if looks_like or i > 30:
break
if not looks_like:
continue
n, new, total = import_file(p)
imported_files += 1
new_total += new
try: try:
n, new, total = import_file(os.path.join(WATCH_DIR, fn)) shown = os.path.relpath(p, root)
msg.append(f"{fn}: {new} new") except ValueError:
except Exception as e: shown = p
errs.append(f"{fn}: {e}") msg.append(f"{shown}: {new} new")
except Exception as e:
errs.append(f"{p}: {e}")
if not imported_files and not errs:
return redirect(url_for("index", msg=f"scanned {len(found)} .txt files — none were Netscape cookie format"))
msg += ["<span class=warn>%s</span>" % e for e in errs] msg += ["<span class=warn>%s</span>" % e for e in errs]
return redirect(url_for("index", msg=" &middot; ".join(msg) or "nothing new in incoming")) return redirect(url_for("index", msg=f"<b>Scan of {root}:</b> {imported_files} cookie files of {len(found)} txt, {new_total} new cookies &middot; " + " &middot; ".join(msg[:25])))
def _scan_dirs(dirs):
msg, errs = [], []
for d in dirs:
if not os.path.isdir(d):
continue
for fn in os.listdir(d):
if fn.lower().endswith((".txt", ".json")) and not fn.startswith("_pasted"):
try:
n, new, total = import_file(os.path.join(d, fn))
msg.append(f"{fn}: {new} new")
except Exception as e:
errs.append(f"{fn}: {e}")
msg += ["<span class=warn>%s</span>" % e for e in errs]
return redirect(url_for("index", msg=" &middot; ".join(msg) or "nothing new found"))
def _launch_login(domain, cookies): def _launch_login(domain, cookies):
"""Runs in background thread: selenium Chrome with cookies injected.""" """Runs in background thread: selenium Chrome with cookies injected."""
@@ -215,18 +351,11 @@ def _launch_login(domain, cookies):
time.sleep(1) time.sleep(1)
for ck in cookies: for ck in cookies:
c = {"name": ck[1], "value": ck[2]} c = {"name": ck[1], "value": ck[2]}
dom = ck[0] c["domain"] = ck[0]
if not dom.startswith("."):
c["domain"] = dom
else:
c["domain"] = dom
c["path"] = ck[3] or "/" c["path"] = ck[3] or "/"
try: exp = _safe_expiry(ck[4])
exp = int(ck[4]) if exp > 0:
if exp > 0: c["expiry"] = exp
c["expiry"] = exp
except Exception:
pass
c["secure"] = str(ck[5]).upper() == "TRUE" c["secure"] = str(ck[5]).upper() == "TRUE"
try: try:
driver.add_cookie(c) driver.add_cookie(c)
@@ -271,8 +400,8 @@ def export():
rows = c.execute("SELECT domain,name,value,path,expiry,secure FROM cookies").fetchall() rows = c.execute("SELECT domain,name,value,path,expiry,secure FROM cookies").fetchall()
c.close() c.close()
jar = [{"domain": r[0], "name": r[1], "value": r[2], "path": r[3], jar = [{"domain": r[0], "name": r[1], "value": r[2], "path": r[3],
"expirationDate": int(r[4]) if r[4] and int(r[4]) > 0 else None, "expirationDate": _safe_expiry(r[4]) or None,
"secure": r[5].upper() == "TRUE", "httpOnly": False} for r in rows] "secure": str(r[5]).upper() == "TRUE", "httpOnly": False} for r in rows]
return Response(json.dumps(jar, indent=2), mimetype="application/json", return Response(json.dumps(jar, indent=2), mimetype="application/json",
headers={"Content-Disposition": "attachment; filename=cookies.json"}) headers={"Content-Disposition": "attachment; filename=cookies.json"})
@@ -287,7 +416,7 @@ def export_ns():
rows = c.execute("SELECT domain,flag,path,secure,expiry,name,value FROM cookies").fetchall() rows = c.execute("SELECT domain,flag,path,secure,expiry,name,value FROM cookies").fetchall()
c.close() c.close()
lines = ["# Netscape HTTP Cookie File", "# Exported by COOKIE VAULT"] lines = ["# Netscape HTTP Cookie File", "# Exported by COOKIE VAULT"]
lines += ["\t".join(str(x) for x in r) for r in rows] lines += ["\t".join(str(x) for x in r) for r in rows if _safe_expiry(r[4]) >= 0]
return Response("\n".join(lines) + "\n", mimetype="text/plain", return Response("\n".join(lines) + "\n", mimetype="text/plain",
headers={"Content-Disposition": "attachment; filename=cookies.txt"}) headers={"Content-Disposition": "attachment; filename=cookies.txt"})