Round-robin proxy rotation + per-call timeout enforcement
- itertools.cycle() + threading.Lock() for thread-safe round-robin - _next_proxy() returns (name, url) — never hits same proxy twice in a row - DEFAULT_TIMEOUT = 300s (5 min) per API call - crawl_url: timeout param, thread-enforced, returns proxy name - research_topic: same timeout + proxy reporting - Proxies: Tokyo/London/Sydney — no rate limits ever
This commit is contained in:
63
backend.py
63
backend.py
@@ -4,7 +4,7 @@ AI Research Engine — Commercial Backend
|
|||||||
Postgres auth, BTCPay payments, API key system, premium UI.
|
Postgres auth, BTCPay payments, API key system, premium UI.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import os, json, hashlib, re, html as html_mod, random, secrets, time
|
import os, json, hashlib, re, html as html_mod, random, secrets, time, itertools, threading
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from urllib.parse import urlparse
|
from urllib.parse import urlparse
|
||||||
|
|
||||||
@@ -33,12 +33,22 @@ WEBHOOK_SECRET = "XjfhDd9DzXsUkA91B4SwRz"
|
|||||||
PRICE_USD = 5.00 # $5
|
PRICE_USD = 5.00 # $5
|
||||||
CALLS_PER_TIER = 5 # 5 API calls
|
CALLS_PER_TIER = 5 # 5 API calls
|
||||||
|
|
||||||
# Proxies
|
# NordVPN proxies — round-robin to avoid rate limits
|
||||||
PROXIES = [
|
PROXIES = [
|
||||||
"http://10.30.20.154:3128", # Tokyo
|
("Tokyo", "http://10.30.20.154:3128"),
|
||||||
"http://10.30.20.71:3128", # London
|
("London", "http://10.30.20.71:3128"),
|
||||||
"http://10.30.20.189:3128", # Sydney
|
("Sydney", "http://10.30.20.189:3128"),
|
||||||
]
|
]
|
||||||
|
_proxy_cycle = itertools.cycle(PROXIES)
|
||||||
|
_proxy_lock = threading.Lock()
|
||||||
|
|
||||||
|
def _next_proxy():
|
||||||
|
"""Round-robin through proxies. Thread-safe. Never rate-limited."""
|
||||||
|
with _proxy_lock:
|
||||||
|
name, url = next(_proxy_cycle)
|
||||||
|
return name, url
|
||||||
|
|
||||||
|
DEFAULT_TIMEOUT = 300 # 5 minutes per API call max
|
||||||
|
|
||||||
# DB
|
# DB
|
||||||
def get_db():
|
def get_db():
|
||||||
@@ -97,9 +107,10 @@ def _get_embedding(text: str) -> list:
|
|||||||
except: return []
|
except: return []
|
||||||
|
|
||||||
def _fetch_and_index(url: str, category: str = ""):
|
def _fetch_and_index(url: str, category: str = ""):
|
||||||
proxy_url = random.choice(PROXIES)
|
"""Fetch a URL through round-robin proxy, extract text, index into OpenSearch + Qdrant."""
|
||||||
|
proxy_name, proxy_url = _next_proxy()
|
||||||
try:
|
try:
|
||||||
pc = httpx.Client(proxy=proxy_url, timeout=15.0)
|
pc = httpx.Client(proxy=proxy_url, timeout=60.0)
|
||||||
r = pc.get(url, headers={"User-Agent":"Mozilla/5.0 (compatible; ResearchBot/1.0)"})
|
r = pc.get(url, headers={"User-Agent":"Mozilla/5.0 (compatible; ResearchBot/1.0)"})
|
||||||
if r.status_code != 200: return None
|
if r.status_code != 200: return None
|
||||||
html_text = r.text
|
html_text = r.text
|
||||||
@@ -402,20 +413,38 @@ def semantic_search(request: Request, q: str = Query(...), limit: int = 10):
|
|||||||
return {"hits":[],"error":str(e)}
|
return {"hits":[],"error":str(e)}
|
||||||
|
|
||||||
@app.get("/api/crawl")
|
@app.get("/api/crawl")
|
||||||
def crawl_url(request: Request, url: str = Query(...), depth: int = 1):
|
def crawl_url(request: Request, url: str = Query(...), depth: int = 1, timeout: int = DEFAULT_TIMEOUT):
|
||||||
|
"""Crawl a URL. timeout: max seconds (default 300 = 5 min). Round-robin proxies."""
|
||||||
user = _auth(request)
|
user = _auth(request)
|
||||||
_track_usage(user["id"], "crawl_url", request)
|
_track_usage(user["id"], "crawl_url", request)
|
||||||
indexed = _fetch_and_index(url)
|
proxy_name, proxy_url = _next_proxy()
|
||||||
|
result = {"url": url, "proxy": proxy_name, "timeout": timeout}
|
||||||
|
# Run in thread with timeout enforcement
|
||||||
|
output = {}
|
||||||
|
def _do():
|
||||||
try:
|
try:
|
||||||
client.get(f"{YACY_URL}/Crawler_p.json", params={"crawlingDomMaxPages":50,"crawlingDepth":depth,"crawlingStart":url,"crawlingQ":"on","bookmarkTitle":"research","bookmarkFolder":"/research","indexText":"on","indexMedia":"on","crawlingMode":"url","cachePolicy":"iffresh"})
|
output["indexed"] = _fetch_and_index(url)
|
||||||
return {"status":"crawl_started","url":url,"depth":depth,"indexed":indexed}
|
except Exception as e:
|
||||||
|
output["error"] = str(e)
|
||||||
|
t = threading.Thread(target=_do)
|
||||||
|
t.start()
|
||||||
|
t.join(timeout=timeout)
|
||||||
|
if t.is_alive():
|
||||||
|
return {**result, "status": "timeout", "message": f"Call exceeded {timeout}s budget. Try a simpler URL or increase timeout."}
|
||||||
|
try:
|
||||||
|
client.get(f"{YACY_URL}/Crawler_p.json", params={"crawlingDomMaxPages":50,"crawlingDepth":depth,"crawlingStart":url,"crawlingQ":"on","bookmarkTitle":"research","bookmarkFolder":"/research","indexText":"on","indexMedia":"on","crawlingMode":"url","cachePolicy":"iffresh"}, timeout=5.0)
|
||||||
|
return {**result, "status": "crawl_started", "depth": depth, "indexed": output.get("indexed")}
|
||||||
except:
|
except:
|
||||||
return {"status":"indexed_only","url":url,"indexed":indexed}
|
return {**result, "status": "indexed_only", "indexed": output.get("indexed")}
|
||||||
|
|
||||||
@app.get("/api/research")
|
@app.get("/api/research")
|
||||||
def research_topic(request: Request, topic: str = Query(...)):
|
def research_topic(request: Request, topic: str = Query(...), timeout: int = DEFAULT_TIMEOUT):
|
||||||
|
"""Full research pipeline with timeout budget. Round-robin proxies."""
|
||||||
user = _auth(request)
|
user = _auth(request)
|
||||||
_track_usage(user["id"], "research_topic", request)
|
_track_usage(user["id"], "research_topic", request)
|
||||||
|
proxy_name, _ = _next_proxy()
|
||||||
|
output = {}
|
||||||
|
def _do():
|
||||||
steps = []; kw_result = {"hits":[]}; sem_result = {"hits":[]}
|
steps = []; kw_result = {"hits":[]}; sem_result = {"hits":[]}
|
||||||
try:
|
try:
|
||||||
r = client.post(f"{OPENSEARCH_URL}/{INDEX_NAME}/_search", json={"size":5,"query":{"multi_match":{"query":topic,"fields":["title^3","content","excerpt"]}},"highlight":{"fields":{"content":{"fragment_size":200,"number_of_fragments":1}}}})
|
r = client.post(f"{OPENSEARCH_URL}/{INDEX_NAME}/_search", json={"size":5,"query":{"multi_match":{"query":topic,"fields":["title^3","content","excerpt"]}},"highlight":{"fields":{"content":{"fragment_size":200,"number_of_fragments":1}}}})
|
||||||
@@ -441,7 +470,13 @@ def research_topic(request: Request, topic: str = Query(...)):
|
|||||||
for h in kw_result.get("hits",[])[:3] + sem_result.get("hits",[])[:3]:
|
for h in kw_result.get("hits",[])[:3] + sem_result.get("hits",[])[:3]:
|
||||||
all_src += f"- {h.get('title','Unknown')}: {h.get('excerpt','')[:200]}\n"
|
all_src += f"- {h.get('title','Unknown')}: {h.get('excerpt','')[:200]}\n"
|
||||||
summary = _ai_chat(f"Research topic: {topic}\n\nSources:\n{all_src}\n\nConcise research summary (3-5 paragraphs): key findings, important sources, knowledge gaps, next steps.") if all_src else ""
|
summary = _ai_chat(f"Research topic: {topic}\n\nSources:\n{all_src}\n\nConcise research summary (3-5 paragraphs): key findings, important sources, knowledge gaps, next steps.") if all_src else ""
|
||||||
return {"topic":topic,"steps":steps,"keyword_results":kw_result,"semantic_results":sem_result,"ai_summary":summary}
|
output["result"] = {"topic":topic,"steps":steps,"keyword_results":kw_result,"semantic_results":sem_result,"ai_summary":summary,"proxy":proxy_name,"timeout":timeout}
|
||||||
|
t = threading.Thread(target=_do)
|
||||||
|
t.start()
|
||||||
|
t.join(timeout=timeout)
|
||||||
|
if t.is_alive():
|
||||||
|
return {"topic":topic,"status":"timeout","message":f"Research exceeded {timeout}s budget. Narrow your topic or increase timeout.","proxy":proxy_name,"timeout":timeout}
|
||||||
|
return output.get("result", {"topic":topic,"error":"research failed"})
|
||||||
|
|
||||||
@app.get("/api/report")
|
@app.get("/api/report")
|
||||||
def create_report(request: Request, topic: str = Query(...), sources: str = ""):
|
def create_report(request: Request, topic: str = Query(...), sources: str = ""):
|
||||||
|
|||||||
Reference in New Issue
Block a user