88 lines
3.6 KiB
Python
88 lines
3.6 KiB
Python
"""Screenshot & Vision Analysis — takes screenshots and classifies with local LLM."""
|
|
import base64, json, subprocess, tempfile, os, urllib.request, time
|
|
|
|
OLLAMA_URL = "http://10.30.20.186:11434"
|
|
VISION_MODEL = "minicpm-v4.5:8b" # Best local vision model available
|
|
|
|
def take_screenshot(url, width=1280, height=900):
|
|
"""Take a screenshot of a URL using Playwright."""
|
|
try:
|
|
from playwright.sync_api import sync_playwright
|
|
with sync_playwright() as p:
|
|
browser = p.chromium.launch(headless=True)
|
|
page = browser.new_page(viewport={'width': width, 'height': height})
|
|
page.goto(url, wait_until='networkidle', timeout=15000)
|
|
time.sleep(1) # Let page settle
|
|
screenshot = page.screenshot(full_page=False)
|
|
browser.close()
|
|
return screenshot
|
|
except ImportError:
|
|
# Fallback: use macos screencapture via Safari (less reliable)
|
|
return None
|
|
except Exception as e:
|
|
print(f"Screenshot error for {url}: {e}")
|
|
return None
|
|
|
|
def screenshot_to_base64(screenshot_bytes):
|
|
"""Convert screenshot bytes to base64 for Ollama."""
|
|
return base64.b64encode(screenshot_bytes).decode('utf-8')
|
|
|
|
def analyze_screenshot(url, screenshot_bytes, processor_name):
|
|
"""Analyze a screenshot using local vision model to describe the site."""
|
|
if not screenshot_bytes:
|
|
return generate_placeholder_description(url, processor_name)
|
|
|
|
img_b64 = screenshot_to_base64(screenshot_bytes)
|
|
|
|
prompt = f"""Look at this screenshot of a website. The site likely uses {processor_name} as a payment processor.
|
|
Describe this site in 2-3 sentences covering:
|
|
1. What type of site is this? (e-commerce, gambling, SaaS, adult, CBD, firearms, subscription, crypto, etc.)
|
|
2. What does it sell or offer?
|
|
3. What's the overall vibe/quality level?
|
|
|
|
Respond with just the description, no prefixes."""
|
|
|
|
try:
|
|
req = urllib.request.Request(
|
|
f"{OLLAMA_URL}/api/generate",
|
|
data=json.dumps({
|
|
"model": VISION_MODEL,
|
|
"prompt": prompt,
|
|
"images": [img_b64],
|
|
"stream": False,
|
|
"options": {"temperature": 0.3, "num_predict": 150}
|
|
}).encode(),
|
|
headers={"Content-Type": "application/json"}
|
|
)
|
|
resp = urllib.request.urlopen(req, timeout=30)
|
|
data = json.loads(resp.read())
|
|
return data.get('response', '').strip()
|
|
except Exception as e:
|
|
print(f"Vision analysis error: {e}")
|
|
return generate_placeholder_description(url, processor_name)
|
|
|
|
def generate_placeholder_description(url, processor_name):
|
|
"""Generate a description when vision isn't available."""
|
|
domain = urllib.parse.urlparse(url).netloc.replace('www.', '')
|
|
return f"Website at {domain}. Uses {processor_name} for payment processing. Visit the site to learn more about their products and services."
|
|
|
|
def batch_analyze(sites, processor_name, max_screenshots=5):
|
|
"""Analyze multiple sites with screenshots."""
|
|
results = []
|
|
for i, site in enumerate(sites[:max_screenshots]):
|
|
url = site.get('url', '')
|
|
if not url:
|
|
continue
|
|
|
|
print(f" Screenshot {i+1}/{min(len(sites), max_screenshots)}: {site.get('domain','?')}")
|
|
screenshot = take_screenshot(url)
|
|
description = analyze_screenshot(url, screenshot, processor_name)
|
|
|
|
results.append({
|
|
**site,
|
|
'description': description,
|
|
'screenshot_available': screenshot is not None,
|
|
'screenshot_base64': screenshot_to_base64(screenshot) if screenshot else None
|
|
})
|
|
return results
|