Snapshot: full project state
This commit is contained in:
424
scraper.py
Normal file
424
scraper.py
Normal file
@@ -0,0 +1,424 @@
|
||||
# Procyon — job discovery. RSS/JSON feeds + optional proxy + manual add.
|
||||
# Uses his OWN proxy infra (optional), not an anti-bot evasion rig.
|
||||
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
import db
|
||||
|
||||
UA = 'Procyon/1.0 (personal job-search assistant)'
|
||||
|
||||
|
||||
def _open(url, timeout=20):
|
||||
req = urllib.request.Request(url, headers={'User-Agent': UA})
|
||||
proxy = db.get_setting('proxy_url', '')
|
||||
if proxy:
|
||||
opener = urllib.request.build_opener(
|
||||
urllib.request.ProxyHandler({'http': proxy, 'https': proxy}))
|
||||
else:
|
||||
opener = urllib.request.build_opener()
|
||||
with opener.open(req, timeout=timeout) as r:
|
||||
return r.read()
|
||||
|
||||
|
||||
def fetch_rss(url):
|
||||
"""Parse an RSS/Atom feed into job dicts."""
|
||||
data = _open(url)
|
||||
root = ET.fromstring(data)
|
||||
items = []
|
||||
for item in root.iter('item'):
|
||||
def _t(tag):
|
||||
el = item.find(tag)
|
||||
return el.text.strip() if el is not None and el.text else ''
|
||||
title = _t('title')
|
||||
link = _t('link')
|
||||
desc = _t('description')
|
||||
items.append({'title': title, 'url': link, 'description': desc,
|
||||
'company': '', 'location': '', 'source': url})
|
||||
return items
|
||||
|
||||
|
||||
def fetch_remoteok():
|
||||
"""RemoteOK JSON API — no key required. Filters to fresh listings: RemoteOK's job
|
||||
pages 302-redirect to the homepage once a posting expires (~30 days), so stale
|
||||
entries would give broken 'Open Application' links."""
|
||||
data = json.loads(_open('https://remoteok.com/api').decode('utf-8'))
|
||||
out = []
|
||||
now = time.time()
|
||||
STALE_AFTER = 30 * 86400 # 30 days
|
||||
for j in data[1:]:
|
||||
if not isinstance(j, dict) or not j.get('position'):
|
||||
continue
|
||||
epoch = j.get('epoch')
|
||||
if epoch:
|
||||
try:
|
||||
if now - float(epoch) > STALE_AFTER:
|
||||
continue
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
out.append({
|
||||
'title': j.get('position'),
|
||||
'company': j.get('company'),
|
||||
'location': j.get('location', 'Remote'),
|
||||
'url': j.get('url'),
|
||||
'description': (j.get('description') or '')[:4000],
|
||||
'source': 'remoteok',
|
||||
'posted_at': j.get('date'),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def fetch_remotive():
|
||||
"""Remotive API — remote jobs JSON, no key. https://remotive.com/api/remote-jobs"""
|
||||
data = json.loads(_open('https://remotive.com/api/remote-jobs').decode('utf-8'))
|
||||
out = []
|
||||
for j in data.get('jobs', []):
|
||||
out.append({
|
||||
'title': j.get('title'),
|
||||
'company': j.get('company_name'),
|
||||
'location': j.get('candidate_required_location') or 'Remote',
|
||||
'url': j.get('url'),
|
||||
'description': (j.get('description') or '')[:4000],
|
||||
'source': 'remotive',
|
||||
'posted_at': j.get('publication_date'),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def fetch_wwr():
|
||||
"""We Work Remotely — remote programming jobs RSS (no key)."""
|
||||
data = _open('https://weworkremotely.com/categories/remote-programming-jobs.rss')
|
||||
root = ET.fromstring(data)
|
||||
items = []
|
||||
for item in root.iter('item'):
|
||||
def _t(tag):
|
||||
el = item.find(tag)
|
||||
return el.text.strip() if el is not None and el.text else ''
|
||||
items.append({'title': _t('title'), 'url': _t('link'),
|
||||
'description': _t('description'), 'company': '',
|
||||
'location': 'Remote', 'source': 'wwr'})
|
||||
return items
|
||||
|
||||
|
||||
# ---- ATS board APIs (Greenhouse / Lever) ----
|
||||
# The headless-agent discovery path: most tech companies (Seattle + remote) post through
|
||||
# Greenhouse/Lever, which expose clean JSON with no bot wall. Sweep their boards directly.
|
||||
|
||||
ATS_COMPANIES = [
|
||||
# Seattle / PNW
|
||||
'zillow', 'redfin', 'remitly', 'outreach', 'smartsheet', 'expedia', 'offerup',
|
||||
'highspot', 'rover', 'allenai', 'f5', 'convoy',
|
||||
# Remote-friendly tech
|
||||
'stripe', 'airbnb', 'doordash', 'datadog', 'snowflake', 'gitlab', 'figma',
|
||||
'notion', 'openai', 'anthropic', 'discord', 'reddit', 'robinhood', 'coinbase',
|
||||
'block', 'vercel', 'cloudflare', 'mongodb', 'elastic', 'confluent', 'twilio',
|
||||
'dropbox', 'asana', 'atlassian', 'postman', 'github', 'hashicorp',
|
||||
]
|
||||
|
||||
|
||||
def _strip_html(s):
|
||||
return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', s or '')).strip()
|
||||
|
||||
|
||||
def fetch_greenhouse(slug):
|
||||
"""Greenhouse board JSON (no key). Returns list of job dicts."""
|
||||
url = f'https://boards-api.greenhouse.io/v1/boards/{slug}/jobs'
|
||||
data = json.loads(_open(url).decode('utf-8'))
|
||||
out = []
|
||||
for j in data.get('jobs', []):
|
||||
loc = j.get('location') or {}
|
||||
out.append({
|
||||
'title': j.get('title'),
|
||||
'company': j.get('company_name') or slug,
|
||||
'location': loc.get('name') if isinstance(loc, dict) else str(loc or 'Remote'),
|
||||
'url': j.get('absolute_url'),
|
||||
'description': _strip_html(j.get('content') or '')[:4000],
|
||||
'source': 'greenhouse',
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def fetch_lever(slug):
|
||||
"""Lever board JSON (no key). Returns list of job dicts."""
|
||||
url = f'https://api.lever.co/v0/postings/{slug}?mode=json'
|
||||
data = json.loads(_open(url).decode('utf-8'))
|
||||
if not isinstance(data, list):
|
||||
return []
|
||||
out = []
|
||||
for j in data:
|
||||
cats = j.get('categories') or {}
|
||||
out.append({
|
||||
'title': j.get('text'),
|
||||
'company': slug,
|
||||
'location': cats.get('location') or 'Remote',
|
||||
'url': j.get('hostedUrl'),
|
||||
'description': _strip_html(j.get('descriptionPlain') or j.get('description') or '')[:4000],
|
||||
'source': 'lever',
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
# Locations worth keeping: WFH (remote) + Seattle-area. Everything else (onsite SF/NY/etc) is noise.
|
||||
_REMOTE_TERMS = ('remote', 'seattle', 'anywhere', 'worldwide', 'work from home', 'wfh',
|
||||
'united states', 'north america', 'americas')
|
||||
|
||||
|
||||
def _is_relevant_location(loc):
|
||||
l = (loc or '').lower()
|
||||
return any(t in l for t in _REMOTE_TERMS)
|
||||
|
||||
|
||||
def fetch_ats():
|
||||
"""Sweep the ATS_COMPANIES boards across Greenhouse + Lever. Returns list of job dicts,
|
||||
filtered to remote / Seattle / US-remote locations."""
|
||||
out = []
|
||||
for slug in ATS_COMPANIES:
|
||||
for fn, label in ((fetch_greenhouse, 'greenhouse'), (fetch_lever, 'lever')):
|
||||
try:
|
||||
jobs = fn(slug)
|
||||
for j in jobs:
|
||||
if _is_relevant_location(j.get('location')):
|
||||
out.append(j)
|
||||
except Exception:
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def indeed_feed(query, location):
|
||||
"""Indeed RSS endpoint (public, returns XML)."""
|
||||
q = urllib.parse.quote_plus(query)
|
||||
l = urllib.parse.quote_plus(location)
|
||||
return f'https://www.indeed.com/rss?q={q}&l={l}'
|
||||
|
||||
|
||||
def ingest(items):
|
||||
added = 0
|
||||
for it in items:
|
||||
if not it.get('title') or not it.get('url'):
|
||||
continue
|
||||
jid = db.upsert_job(
|
||||
it.get('title'), it.get('company') or '', it.get('location') or '',
|
||||
it.get('url'), it.get('source', 'manual'), it.get('description') or '',
|
||||
it.get('posted_at'),
|
||||
)
|
||||
if jid:
|
||||
added += 1
|
||||
return added
|
||||
|
||||
|
||||
def discover(query='software engineer', location='seattle wa', sources=None):
|
||||
"""Pull from configured sources. Returns count added."""
|
||||
added = 0
|
||||
# RemoteOK (remote)
|
||||
try:
|
||||
added += ingest(fetch_remoteok())
|
||||
except Exception as e:
|
||||
print(f'remoteok failed: {e}')
|
||||
# Remotive (remote)
|
||||
try:
|
||||
added += ingest(fetch_remotive())
|
||||
except Exception as e:
|
||||
print(f'remotive failed: {e}')
|
||||
# We Work Remotely (remote)
|
||||
try:
|
||||
added += ingest(fetch_wwr())
|
||||
except Exception as e:
|
||||
print(f'wwr failed: {e}')
|
||||
# ATS boards (Greenhouse + Lever) — Seattle + remote tech companies
|
||||
try:
|
||||
added += ingest(fetch_ats())
|
||||
except Exception as e:
|
||||
print(f'ats failed: {e}')
|
||||
# Indeed RSS (dead — left in try/except, non-functional)
|
||||
try:
|
||||
added += ingest(fetch_rss(indeed_feed(query, location)))
|
||||
except Exception as e:
|
||||
print(f'indeed rss failed: {e}')
|
||||
# Firecrawl search (self-hosted, /v1/search works; scrape is bot-walled)
|
||||
if db.get_setting('firecrawl_enabled', '1') == '1':
|
||||
added += firecrawl_discover()
|
||||
# configured extra feeds
|
||||
feeds = db.get_setting('rss_feeds', '')
|
||||
for f in [x.strip() for x in feeds.splitlines() if x.strip()]:
|
||||
try:
|
||||
added += ingest(fetch_rss(f))
|
||||
except Exception as e:
|
||||
print(f'feed {f} failed: {e}')
|
||||
return added
|
||||
|
||||
|
||||
# ---- Firecrawl (self-hosted) ----
|
||||
|
||||
def firecrawl_search(query, limit=5):
|
||||
"""Self-hosted Firecrawl /v1/search. Returns list of {url, title, description}."""
|
||||
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
|
||||
payload = json.dumps({'query': query, 'limit': limit}).encode('utf-8')
|
||||
req = urllib.request.Request(f'{base}/v1/search', data=payload,
|
||||
headers={'Content-Type': 'application/json', 'User-Agent': UA})
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
data = json.loads(r.read().decode('utf-8'))
|
||||
return data.get('data', []) if data.get('success') else []
|
||||
|
||||
|
||||
def research_company(job):
|
||||
"""Best-effort company research: a short factual blurb to weave into the email.
|
||||
|
||||
Uses self-hosted Firecrawl search for the company name. Returns a string of
|
||||
1-2 snippet(s), or '' if nothing relevant is found. Never raises."""
|
||||
company = (job.get('company') or '').strip()
|
||||
if not company:
|
||||
return ''
|
||||
try:
|
||||
results = firecrawl_search(f'{company} about', limit=3)
|
||||
except Exception:
|
||||
return ''
|
||||
facts = []
|
||||
for r in results:
|
||||
title = (r.get('title') or '').strip()
|
||||
desc = (r.get('description') or '').strip()
|
||||
blob = f'{title} {desc}'.lower()
|
||||
if desc and company.lower() in blob:
|
||||
facts.append(f'{title}: {desc}'[:500])
|
||||
return '\n'.join(facts[:2])
|
||||
|
||||
|
||||
# ---- contact email scraping (for auto-send on approve) ----
|
||||
|
||||
_EMAIL_RE = re.compile(r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}')
|
||||
_HARD_REJECT = ('noreply', 'no-reply', 'donotreply', 'privacy', 'abuse', 'legal', 'arb',
|
||||
'demand', 'press', 'media', 'security@', 'support', 'example', 'yourdomain',
|
||||
'email.com', 'sentry', 'wixpress', 'domain.com', 'squarespace', 'cloudflare',
|
||||
'compliance', 'trust', 'notice', 'copyright', 'dpo', 'gdpr', 'unsubscribe')
|
||||
_RECRUITING_HINT = ('career', 'recruit', 'hiring', 'jobs', 'talent', 'people', 'hr@',
|
||||
'work', 'join', 'apply', 'hello', 'team', 'contact', 'info')
|
||||
|
||||
|
||||
def _email_rank(e):
|
||||
"""Score an email: -1 = reject (legal/abuse/footer junk), higher = more recruiting-ish."""
|
||||
el = e.lower()
|
||||
if any(w in el for w in _HARD_REJECT):
|
||||
return -1
|
||||
score = 0
|
||||
for h in _RECRUITING_HINT:
|
||||
if h in el:
|
||||
score += 1
|
||||
return score
|
||||
|
||||
|
||||
def _emails_from_url(url):
|
||||
try:
|
||||
raw = _open(url, timeout=12)
|
||||
text = raw.decode('utf-8', errors='replace')
|
||||
found = {}
|
||||
for e in _EMAIL_RE.findall(text):
|
||||
e = e.rstrip('.,;:>').lower()
|
||||
rank = _email_rank(e)
|
||||
if rank > 0 and (e not in found or rank > found[e]):
|
||||
found[e] = rank
|
||||
return sorted(found, key=lambda x: -found[x])
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def _company_domains(job_url, company):
|
||||
"""Derive candidate company domains to check for a contact email."""
|
||||
doms = []
|
||||
slug = None
|
||||
m = re.match(r'https?://(?:job-boards|boards)\.greenhouse\.io/([a-z0-9-]+)', job_url or '')
|
||||
if not m:
|
||||
m = re.match(r'https?://jobs\.lever\.co/([a-z0-9-]+)', job_url or '')
|
||||
if m:
|
||||
slug = m.group(1)
|
||||
if slug:
|
||||
doms += [f'https://{slug}.com', f'https://www.{slug}.com']
|
||||
if company:
|
||||
cslug = re.sub(r'[^a-z0-9]', '', (company or '').lower())
|
||||
if cslug and cslug not in (slug or ''):
|
||||
doms += [f'https://{cslug}.com', f'https://www.{cslug}.com']
|
||||
return doms
|
||||
|
||||
|
||||
def scrape_contact_email(job):
|
||||
"""Best-effort: find a real contact email for the job/company (for auto-send).
|
||||
Checks the posting first, then the company's /contact, /careers, /about, /jobs, and
|
||||
homepage. Returns an email string or ''. Never raises."""
|
||||
url = job.get('url') or ''
|
||||
company = job.get('company') or ''
|
||||
if not url and not company:
|
||||
return ''
|
||||
if url:
|
||||
emails = _emails_from_url(url)
|
||||
if emails:
|
||||
return emails[0]
|
||||
for base in _company_domains(url, company):
|
||||
for path in ('/contact', '/careers', '/about', '/jobs', '/contact-us', '/'):
|
||||
emails = _emails_from_url(f'{base}{path}')
|
||||
if emails:
|
||||
return emails[0]
|
||||
return ''
|
||||
|
||||
|
||||
def firecrawl_discover():
|
||||
"""Run configured search queries -> job leads. Returns count added."""
|
||||
added = 0
|
||||
queries = [q.strip() for q in db.get_setting('firecrawl_queries', '').splitlines() if q.strip()]
|
||||
for q in queries:
|
||||
try:
|
||||
for res in firecrawl_search(q, limit=5):
|
||||
if not res.get('url') or not res.get('title'):
|
||||
continue
|
||||
if _is_garbage_title(res.get('title')):
|
||||
continue
|
||||
# only keep plausible individual postings / listings
|
||||
db.upsert_job(
|
||||
res.get('title'), '', 'Remote/Web', res.get('url'), 'firecrawl',
|
||||
res.get('description', '') or '', None,
|
||||
)
|
||||
added += 1
|
||||
except Exception as e:
|
||||
print(f'firecrawl query {q!r} failed: {e}')
|
||||
return added
|
||||
|
||||
|
||||
_GARBAGE_TITLE_MARKERS = ('jobsradar', 'linkedin', '1,000+', 'open roles', 'greater ',
|
||||
'remote jobs', 'jobs in', 'job board', 'indeed', 'glassdoor',
|
||||
'ziprecruiter', 'simplyhired', ' careers', 'search jobs')
|
||||
|
||||
|
||||
def _is_garbage_title(t):
|
||||
"""Firecrawl search returns search-result / aggregator pages, not individual postings.
|
||||
Skip titles that look like a results page rather than a single role."""
|
||||
tl = (t or '').lower()
|
||||
return any(m in tl for m in _GARBAGE_TITLE_MARKERS)
|
||||
|
||||
|
||||
def firecrawl_scrape(url):
|
||||
"""Scrape a URL to markdown via self-hosted Firecrawl. Returns text or None."""
|
||||
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
|
||||
payload = json.dumps({'url': url, 'formats': ['markdown']}).encode('utf-8')
|
||||
req = urllib.request.Request(f'{base}/v1/scrape', data=payload,
|
||||
headers={'Content-Type': 'application/json', 'User-Agent': UA})
|
||||
with urllib.request.urlopen(req, timeout=20) as r:
|
||||
data = json.loads(r.read().decode('utf-8'))
|
||||
return (data.get('data') or {}).get('markdown') or None
|
||||
|
||||
|
||||
def enrich_description(url):
|
||||
"""Get a fuller job description for a URL. Firecrawl first, HTTP fallback."""
|
||||
try:
|
||||
md = firecrawl_scrape(url)
|
||||
if md:
|
||||
return md
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
raw = _open(url, timeout=15)
|
||||
# strip tags crudely
|
||||
text = re.sub(r'<[^>]+>', ' ', raw.decode('utf-8', errors='replace'))
|
||||
return re.sub(r'\s+', ' ', text)[:4000]
|
||||
except Exception:
|
||||
return None
|
||||
Reference in New Issue
Block a user