Snapshot: full project state

This commit is contained in:
2026-10-06 23:43:21 -07:00
commit 1551408fc5
37 changed files with 2827 additions and 0 deletions

424
scraper.py Normal file
View File

@@ -0,0 +1,424 @@
# Procyon — job discovery. RSS/JSON feeds + optional proxy + manual add.
# Uses his OWN proxy infra (optional), not an anti-bot evasion rig.
import json
import re
import time
import urllib.parse
import urllib.request
import xml.etree.ElementTree as ET
import db
UA = 'Procyon/1.0 (personal job-search assistant)'
def _open(url, timeout=20):
req = urllib.request.Request(url, headers={'User-Agent': UA})
proxy = db.get_setting('proxy_url', '')
if proxy:
opener = urllib.request.build_opener(
urllib.request.ProxyHandler({'http': proxy, 'https': proxy}))
else:
opener = urllib.request.build_opener()
with opener.open(req, timeout=timeout) as r:
return r.read()
def fetch_rss(url):
"""Parse an RSS/Atom feed into job dicts."""
data = _open(url)
root = ET.fromstring(data)
items = []
for item in root.iter('item'):
def _t(tag):
el = item.find(tag)
return el.text.strip() if el is not None and el.text else ''
title = _t('title')
link = _t('link')
desc = _t('description')
items.append({'title': title, 'url': link, 'description': desc,
'company': '', 'location': '', 'source': url})
return items
def fetch_remoteok():
"""RemoteOK JSON API — no key required. Filters to fresh listings: RemoteOK's job
pages 302-redirect to the homepage once a posting expires (~30 days), so stale
entries would give broken 'Open Application' links."""
data = json.loads(_open('https://remoteok.com/api').decode('utf-8'))
out = []
now = time.time()
STALE_AFTER = 30 * 86400 # 30 days
for j in data[1:]:
if not isinstance(j, dict) or not j.get('position'):
continue
epoch = j.get('epoch')
if epoch:
try:
if now - float(epoch) > STALE_AFTER:
continue
except (TypeError, ValueError):
pass
out.append({
'title': j.get('position'),
'company': j.get('company'),
'location': j.get('location', 'Remote'),
'url': j.get('url'),
'description': (j.get('description') or '')[:4000],
'source': 'remoteok',
'posted_at': j.get('date'),
})
return out
def fetch_remotive():
"""Remotive API — remote jobs JSON, no key. https://remotive.com/api/remote-jobs"""
data = json.loads(_open('https://remotive.com/api/remote-jobs').decode('utf-8'))
out = []
for j in data.get('jobs', []):
out.append({
'title': j.get('title'),
'company': j.get('company_name'),
'location': j.get('candidate_required_location') or 'Remote',
'url': j.get('url'),
'description': (j.get('description') or '')[:4000],
'source': 'remotive',
'posted_at': j.get('publication_date'),
})
return out
def fetch_wwr():
"""We Work Remotely — remote programming jobs RSS (no key)."""
data = _open('https://weworkremotely.com/categories/remote-programming-jobs.rss')
root = ET.fromstring(data)
items = []
for item in root.iter('item'):
def _t(tag):
el = item.find(tag)
return el.text.strip() if el is not None and el.text else ''
items.append({'title': _t('title'), 'url': _t('link'),
'description': _t('description'), 'company': '',
'location': 'Remote', 'source': 'wwr'})
return items
# ---- ATS board APIs (Greenhouse / Lever) ----
# The headless-agent discovery path: most tech companies (Seattle + remote) post through
# Greenhouse/Lever, which expose clean JSON with no bot wall. Sweep their boards directly.
ATS_COMPANIES = [
# Seattle / PNW
'zillow', 'redfin', 'remitly', 'outreach', 'smartsheet', 'expedia', 'offerup',
'highspot', 'rover', 'allenai', 'f5', 'convoy',
# Remote-friendly tech
'stripe', 'airbnb', 'doordash', 'datadog', 'snowflake', 'gitlab', 'figma',
'notion', 'openai', 'anthropic', 'discord', 'reddit', 'robinhood', 'coinbase',
'block', 'vercel', 'cloudflare', 'mongodb', 'elastic', 'confluent', 'twilio',
'dropbox', 'asana', 'atlassian', 'postman', 'github', 'hashicorp',
]
def _strip_html(s):
return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', s or '')).strip()
def fetch_greenhouse(slug):
"""Greenhouse board JSON (no key). Returns list of job dicts."""
url = f'https://boards-api.greenhouse.io/v1/boards/{slug}/jobs'
data = json.loads(_open(url).decode('utf-8'))
out = []
for j in data.get('jobs', []):
loc = j.get('location') or {}
out.append({
'title': j.get('title'),
'company': j.get('company_name') or slug,
'location': loc.get('name') if isinstance(loc, dict) else str(loc or 'Remote'),
'url': j.get('absolute_url'),
'description': _strip_html(j.get('content') or '')[:4000],
'source': 'greenhouse',
})
return out
def fetch_lever(slug):
"""Lever board JSON (no key). Returns list of job dicts."""
url = f'https://api.lever.co/v0/postings/{slug}?mode=json'
data = json.loads(_open(url).decode('utf-8'))
if not isinstance(data, list):
return []
out = []
for j in data:
cats = j.get('categories') or {}
out.append({
'title': j.get('text'),
'company': slug,
'location': cats.get('location') or 'Remote',
'url': j.get('hostedUrl'),
'description': _strip_html(j.get('descriptionPlain') or j.get('description') or '')[:4000],
'source': 'lever',
})
return out
# Locations worth keeping: WFH (remote) + Seattle-area. Everything else (onsite SF/NY/etc) is noise.
_REMOTE_TERMS = ('remote', 'seattle', 'anywhere', 'worldwide', 'work from home', 'wfh',
'united states', 'north america', 'americas')
def _is_relevant_location(loc):
l = (loc or '').lower()
return any(t in l for t in _REMOTE_TERMS)
def fetch_ats():
"""Sweep the ATS_COMPANIES boards across Greenhouse + Lever. Returns list of job dicts,
filtered to remote / Seattle / US-remote locations."""
out = []
for slug in ATS_COMPANIES:
for fn, label in ((fetch_greenhouse, 'greenhouse'), (fetch_lever, 'lever')):
try:
jobs = fn(slug)
for j in jobs:
if _is_relevant_location(j.get('location')):
out.append(j)
except Exception:
continue
return out
def indeed_feed(query, location):
"""Indeed RSS endpoint (public, returns XML)."""
q = urllib.parse.quote_plus(query)
l = urllib.parse.quote_plus(location)
return f'https://www.indeed.com/rss?q={q}&l={l}'
def ingest(items):
added = 0
for it in items:
if not it.get('title') or not it.get('url'):
continue
jid = db.upsert_job(
it.get('title'), it.get('company') or '', it.get('location') or '',
it.get('url'), it.get('source', 'manual'), it.get('description') or '',
it.get('posted_at'),
)
if jid:
added += 1
return added
def discover(query='software engineer', location='seattle wa', sources=None):
"""Pull from configured sources. Returns count added."""
added = 0
# RemoteOK (remote)
try:
added += ingest(fetch_remoteok())
except Exception as e:
print(f'remoteok failed: {e}')
# Remotive (remote)
try:
added += ingest(fetch_remotive())
except Exception as e:
print(f'remotive failed: {e}')
# We Work Remotely (remote)
try:
added += ingest(fetch_wwr())
except Exception as e:
print(f'wwr failed: {e}')
# ATS boards (Greenhouse + Lever) — Seattle + remote tech companies
try:
added += ingest(fetch_ats())
except Exception as e:
print(f'ats failed: {e}')
# Indeed RSS (dead — left in try/except, non-functional)
try:
added += ingest(fetch_rss(indeed_feed(query, location)))
except Exception as e:
print(f'indeed rss failed: {e}')
# Firecrawl search (self-hosted, /v1/search works; scrape is bot-walled)
if db.get_setting('firecrawl_enabled', '1') == '1':
added += firecrawl_discover()
# configured extra feeds
feeds = db.get_setting('rss_feeds', '')
for f in [x.strip() for x in feeds.splitlines() if x.strip()]:
try:
added += ingest(fetch_rss(f))
except Exception as e:
print(f'feed {f} failed: {e}')
return added
# ---- Firecrawl (self-hosted) ----
def firecrawl_search(query, limit=5):
"""Self-hosted Firecrawl /v1/search. Returns list of {url, title, description}."""
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
payload = json.dumps({'query': query, 'limit': limit}).encode('utf-8')
req = urllib.request.Request(f'{base}/v1/search', data=payload,
headers={'Content-Type': 'application/json', 'User-Agent': UA})
with urllib.request.urlopen(req, timeout=30) as r:
data = json.loads(r.read().decode('utf-8'))
return data.get('data', []) if data.get('success') else []
def research_company(job):
"""Best-effort company research: a short factual blurb to weave into the email.
Uses self-hosted Firecrawl search for the company name. Returns a string of
1-2 snippet(s), or '' if nothing relevant is found. Never raises."""
company = (job.get('company') or '').strip()
if not company:
return ''
try:
results = firecrawl_search(f'{company} about', limit=3)
except Exception:
return ''
facts = []
for r in results:
title = (r.get('title') or '').strip()
desc = (r.get('description') or '').strip()
blob = f'{title} {desc}'.lower()
if desc and company.lower() in blob:
facts.append(f'{title}: {desc}'[:500])
return '\n'.join(facts[:2])
# ---- contact email scraping (for auto-send on approve) ----
_EMAIL_RE = re.compile(r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}')
_HARD_REJECT = ('noreply', 'no-reply', 'donotreply', 'privacy', 'abuse', 'legal', 'arb',
'demand', 'press', 'media', 'security@', 'support', 'example', 'yourdomain',
'email.com', 'sentry', 'wixpress', 'domain.com', 'squarespace', 'cloudflare',
'compliance', 'trust', 'notice', 'copyright', 'dpo', 'gdpr', 'unsubscribe')
_RECRUITING_HINT = ('career', 'recruit', 'hiring', 'jobs', 'talent', 'people', 'hr@',
'work', 'join', 'apply', 'hello', 'team', 'contact', 'info')
def _email_rank(e):
"""Score an email: -1 = reject (legal/abuse/footer junk), higher = more recruiting-ish."""
el = e.lower()
if any(w in el for w in _HARD_REJECT):
return -1
score = 0
for h in _RECRUITING_HINT:
if h in el:
score += 1
return score
def _emails_from_url(url):
try:
raw = _open(url, timeout=12)
text = raw.decode('utf-8', errors='replace')
found = {}
for e in _EMAIL_RE.findall(text):
e = e.rstrip('.,;:>').lower()
rank = _email_rank(e)
if rank > 0 and (e not in found or rank > found[e]):
found[e] = rank
return sorted(found, key=lambda x: -found[x])
except Exception:
return []
def _company_domains(job_url, company):
"""Derive candidate company domains to check for a contact email."""
doms = []
slug = None
m = re.match(r'https?://(?:job-boards|boards)\.greenhouse\.io/([a-z0-9-]+)', job_url or '')
if not m:
m = re.match(r'https?://jobs\.lever\.co/([a-z0-9-]+)', job_url or '')
if m:
slug = m.group(1)
if slug:
doms += [f'https://{slug}.com', f'https://www.{slug}.com']
if company:
cslug = re.sub(r'[^a-z0-9]', '', (company or '').lower())
if cslug and cslug not in (slug or ''):
doms += [f'https://{cslug}.com', f'https://www.{cslug}.com']
return doms
def scrape_contact_email(job):
"""Best-effort: find a real contact email for the job/company (for auto-send).
Checks the posting first, then the company's /contact, /careers, /about, /jobs, and
homepage. Returns an email string or ''. Never raises."""
url = job.get('url') or ''
company = job.get('company') or ''
if not url and not company:
return ''
if url:
emails = _emails_from_url(url)
if emails:
return emails[0]
for base in _company_domains(url, company):
for path in ('/contact', '/careers', '/about', '/jobs', '/contact-us', '/'):
emails = _emails_from_url(f'{base}{path}')
if emails:
return emails[0]
return ''
def firecrawl_discover():
"""Run configured search queries -> job leads. Returns count added."""
added = 0
queries = [q.strip() for q in db.get_setting('firecrawl_queries', '').splitlines() if q.strip()]
for q in queries:
try:
for res in firecrawl_search(q, limit=5):
if not res.get('url') or not res.get('title'):
continue
if _is_garbage_title(res.get('title')):
continue
# only keep plausible individual postings / listings
db.upsert_job(
res.get('title'), '', 'Remote/Web', res.get('url'), 'firecrawl',
res.get('description', '') or '', None,
)
added += 1
except Exception as e:
print(f'firecrawl query {q!r} failed: {e}')
return added
_GARBAGE_TITLE_MARKERS = ('jobsradar', 'linkedin', '1,000+', 'open roles', 'greater ',
'remote jobs', 'jobs in', 'job board', 'indeed', 'glassdoor',
'ziprecruiter', 'simplyhired', ' careers', 'search jobs')
def _is_garbage_title(t):
"""Firecrawl search returns search-result / aggregator pages, not individual postings.
Skip titles that look like a results page rather than a single role."""
tl = (t or '').lower()
return any(m in tl for m in _GARBAGE_TITLE_MARKERS)
def firecrawl_scrape(url):
"""Scrape a URL to markdown via self-hosted Firecrawl. Returns text or None."""
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
payload = json.dumps({'url': url, 'formats': ['markdown']}).encode('utf-8')
req = urllib.request.Request(f'{base}/v1/scrape', data=payload,
headers={'Content-Type': 'application/json', 'User-Agent': UA})
with urllib.request.urlopen(req, timeout=20) as r:
data = json.loads(r.read().decode('utf-8'))
return (data.get('data') or {}).get('markdown') or None
def enrich_description(url):
"""Get a fuller job description for a URL. Firecrawl first, HTTP fallback."""
try:
md = firecrawl_scrape(url)
if md:
return md
except Exception:
pass
try:
raw = _open(url, timeout=15)
# strip tags crudely
text = re.sub(r'<[^>]+>', ' ', raw.decode('utf-8', errors='replace'))
return re.sub(r'\s+', ' ', text)[:4000]
except Exception:
return None