425 lines
15 KiB
Python
425 lines
15 KiB
Python
# Procyon — job discovery. RSS/JSON feeds + optional proxy + manual add.
|
|
# Uses his OWN proxy infra (optional), not an anti-bot evasion rig.
|
|
|
|
import json
|
|
import re
|
|
import time
|
|
import urllib.parse
|
|
import urllib.request
|
|
import xml.etree.ElementTree as ET
|
|
|
|
import db
|
|
|
|
UA = 'Procyon/1.0 (personal job-search assistant)'
|
|
|
|
|
|
def _open(url, timeout=20):
|
|
req = urllib.request.Request(url, headers={'User-Agent': UA})
|
|
proxy = db.get_setting('proxy_url', '')
|
|
if proxy:
|
|
opener = urllib.request.build_opener(
|
|
urllib.request.ProxyHandler({'http': proxy, 'https': proxy}))
|
|
else:
|
|
opener = urllib.request.build_opener()
|
|
with opener.open(req, timeout=timeout) as r:
|
|
return r.read()
|
|
|
|
|
|
def fetch_rss(url):
|
|
"""Parse an RSS/Atom feed into job dicts."""
|
|
data = _open(url)
|
|
root = ET.fromstring(data)
|
|
items = []
|
|
for item in root.iter('item'):
|
|
def _t(tag):
|
|
el = item.find(tag)
|
|
return el.text.strip() if el is not None and el.text else ''
|
|
title = _t('title')
|
|
link = _t('link')
|
|
desc = _t('description')
|
|
items.append({'title': title, 'url': link, 'description': desc,
|
|
'company': '', 'location': '', 'source': url})
|
|
return items
|
|
|
|
|
|
def fetch_remoteok():
|
|
"""RemoteOK JSON API — no key required. Filters to fresh listings: RemoteOK's job
|
|
pages 302-redirect to the homepage once a posting expires (~30 days), so stale
|
|
entries would give broken 'Open Application' links."""
|
|
data = json.loads(_open('https://remoteok.com/api').decode('utf-8'))
|
|
out = []
|
|
now = time.time()
|
|
STALE_AFTER = 30 * 86400 # 30 days
|
|
for j in data[1:]:
|
|
if not isinstance(j, dict) or not j.get('position'):
|
|
continue
|
|
epoch = j.get('epoch')
|
|
if epoch:
|
|
try:
|
|
if now - float(epoch) > STALE_AFTER:
|
|
continue
|
|
except (TypeError, ValueError):
|
|
pass
|
|
out.append({
|
|
'title': j.get('position'),
|
|
'company': j.get('company'),
|
|
'location': j.get('location', 'Remote'),
|
|
'url': j.get('url'),
|
|
'description': (j.get('description') or '')[:4000],
|
|
'source': 'remoteok',
|
|
'posted_at': j.get('date'),
|
|
})
|
|
return out
|
|
|
|
|
|
def fetch_remotive():
|
|
"""Remotive API — remote jobs JSON, no key. https://remotive.com/api/remote-jobs"""
|
|
data = json.loads(_open('https://remotive.com/api/remote-jobs').decode('utf-8'))
|
|
out = []
|
|
for j in data.get('jobs', []):
|
|
out.append({
|
|
'title': j.get('title'),
|
|
'company': j.get('company_name'),
|
|
'location': j.get('candidate_required_location') or 'Remote',
|
|
'url': j.get('url'),
|
|
'description': (j.get('description') or '')[:4000],
|
|
'source': 'remotive',
|
|
'posted_at': j.get('publication_date'),
|
|
})
|
|
return out
|
|
|
|
|
|
def fetch_wwr():
|
|
"""We Work Remotely — remote programming jobs RSS (no key)."""
|
|
data = _open('https://weworkremotely.com/categories/remote-programming-jobs.rss')
|
|
root = ET.fromstring(data)
|
|
items = []
|
|
for item in root.iter('item'):
|
|
def _t(tag):
|
|
el = item.find(tag)
|
|
return el.text.strip() if el is not None and el.text else ''
|
|
items.append({'title': _t('title'), 'url': _t('link'),
|
|
'description': _t('description'), 'company': '',
|
|
'location': 'Remote', 'source': 'wwr'})
|
|
return items
|
|
|
|
|
|
# ---- ATS board APIs (Greenhouse / Lever) ----
|
|
# The headless-agent discovery path: most tech companies (Seattle + remote) post through
|
|
# Greenhouse/Lever, which expose clean JSON with no bot wall. Sweep their boards directly.
|
|
|
|
ATS_COMPANIES = [
|
|
# Seattle / PNW
|
|
'zillow', 'redfin', 'remitly', 'outreach', 'smartsheet', 'expedia', 'offerup',
|
|
'highspot', 'rover', 'allenai', 'f5', 'convoy',
|
|
# Remote-friendly tech
|
|
'stripe', 'airbnb', 'doordash', 'datadog', 'snowflake', 'gitlab', 'figma',
|
|
'notion', 'openai', 'anthropic', 'discord', 'reddit', 'robinhood', 'coinbase',
|
|
'block', 'vercel', 'cloudflare', 'mongodb', 'elastic', 'confluent', 'twilio',
|
|
'dropbox', 'asana', 'atlassian', 'postman', 'github', 'hashicorp',
|
|
]
|
|
|
|
|
|
def _strip_html(s):
|
|
return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', s or '')).strip()
|
|
|
|
|
|
def fetch_greenhouse(slug):
|
|
"""Greenhouse board JSON (no key). Returns list of job dicts."""
|
|
url = f'https://boards-api.greenhouse.io/v1/boards/{slug}/jobs'
|
|
data = json.loads(_open(url).decode('utf-8'))
|
|
out = []
|
|
for j in data.get('jobs', []):
|
|
loc = j.get('location') or {}
|
|
out.append({
|
|
'title': j.get('title'),
|
|
'company': j.get('company_name') or slug,
|
|
'location': loc.get('name') if isinstance(loc, dict) else str(loc or 'Remote'),
|
|
'url': j.get('absolute_url'),
|
|
'description': _strip_html(j.get('content') or '')[:4000],
|
|
'source': 'greenhouse',
|
|
})
|
|
return out
|
|
|
|
|
|
def fetch_lever(slug):
|
|
"""Lever board JSON (no key). Returns list of job dicts."""
|
|
url = f'https://api.lever.co/v0/postings/{slug}?mode=json'
|
|
data = json.loads(_open(url).decode('utf-8'))
|
|
if not isinstance(data, list):
|
|
return []
|
|
out = []
|
|
for j in data:
|
|
cats = j.get('categories') or {}
|
|
out.append({
|
|
'title': j.get('text'),
|
|
'company': slug,
|
|
'location': cats.get('location') or 'Remote',
|
|
'url': j.get('hostedUrl'),
|
|
'description': _strip_html(j.get('descriptionPlain') or j.get('description') or '')[:4000],
|
|
'source': 'lever',
|
|
})
|
|
return out
|
|
|
|
|
|
# Locations worth keeping: WFH (remote) + Seattle-area. Everything else (onsite SF/NY/etc) is noise.
|
|
_REMOTE_TERMS = ('remote', 'seattle', 'anywhere', 'worldwide', 'work from home', 'wfh',
|
|
'united states', 'north america', 'americas')
|
|
|
|
|
|
def _is_relevant_location(loc):
|
|
l = (loc or '').lower()
|
|
return any(t in l for t in _REMOTE_TERMS)
|
|
|
|
|
|
def fetch_ats():
|
|
"""Sweep the ATS_COMPANIES boards across Greenhouse + Lever. Returns list of job dicts,
|
|
filtered to remote / Seattle / US-remote locations."""
|
|
out = []
|
|
for slug in ATS_COMPANIES:
|
|
for fn, label in ((fetch_greenhouse, 'greenhouse'), (fetch_lever, 'lever')):
|
|
try:
|
|
jobs = fn(slug)
|
|
for j in jobs:
|
|
if _is_relevant_location(j.get('location')):
|
|
out.append(j)
|
|
except Exception:
|
|
continue
|
|
return out
|
|
|
|
|
|
def indeed_feed(query, location):
|
|
"""Indeed RSS endpoint (public, returns XML)."""
|
|
q = urllib.parse.quote_plus(query)
|
|
l = urllib.parse.quote_plus(location)
|
|
return f'https://www.indeed.com/rss?q={q}&l={l}'
|
|
|
|
|
|
def ingest(items):
|
|
added = 0
|
|
for it in items:
|
|
if not it.get('title') or not it.get('url'):
|
|
continue
|
|
jid = db.upsert_job(
|
|
it.get('title'), it.get('company') or '', it.get('location') or '',
|
|
it.get('url'), it.get('source', 'manual'), it.get('description') or '',
|
|
it.get('posted_at'),
|
|
)
|
|
if jid:
|
|
added += 1
|
|
return added
|
|
|
|
|
|
def discover(query='software engineer', location='seattle wa', sources=None):
|
|
"""Pull from configured sources. Returns count added."""
|
|
added = 0
|
|
# RemoteOK (remote)
|
|
try:
|
|
added += ingest(fetch_remoteok())
|
|
except Exception as e:
|
|
print(f'remoteok failed: {e}')
|
|
# Remotive (remote)
|
|
try:
|
|
added += ingest(fetch_remotive())
|
|
except Exception as e:
|
|
print(f'remotive failed: {e}')
|
|
# We Work Remotely (remote)
|
|
try:
|
|
added += ingest(fetch_wwr())
|
|
except Exception as e:
|
|
print(f'wwr failed: {e}')
|
|
# ATS boards (Greenhouse + Lever) — Seattle + remote tech companies
|
|
try:
|
|
added += ingest(fetch_ats())
|
|
except Exception as e:
|
|
print(f'ats failed: {e}')
|
|
# Indeed RSS (dead — left in try/except, non-functional)
|
|
try:
|
|
added += ingest(fetch_rss(indeed_feed(query, location)))
|
|
except Exception as e:
|
|
print(f'indeed rss failed: {e}')
|
|
# Firecrawl search (self-hosted, /v1/search works; scrape is bot-walled)
|
|
if db.get_setting('firecrawl_enabled', '1') == '1':
|
|
added += firecrawl_discover()
|
|
# configured extra feeds
|
|
feeds = db.get_setting('rss_feeds', '')
|
|
for f in [x.strip() for x in feeds.splitlines() if x.strip()]:
|
|
try:
|
|
added += ingest(fetch_rss(f))
|
|
except Exception as e:
|
|
print(f'feed {f} failed: {e}')
|
|
return added
|
|
|
|
|
|
# ---- Firecrawl (self-hosted) ----
|
|
|
|
def firecrawl_search(query, limit=5):
|
|
"""Self-hosted Firecrawl /v1/search. Returns list of {url, title, description}."""
|
|
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
|
|
payload = json.dumps({'query': query, 'limit': limit}).encode('utf-8')
|
|
req = urllib.request.Request(f'{base}/v1/search', data=payload,
|
|
headers={'Content-Type': 'application/json', 'User-Agent': UA})
|
|
with urllib.request.urlopen(req, timeout=30) as r:
|
|
data = json.loads(r.read().decode('utf-8'))
|
|
return data.get('data', []) if data.get('success') else []
|
|
|
|
|
|
def research_company(job):
|
|
"""Best-effort company research: a short factual blurb to weave into the email.
|
|
|
|
Uses self-hosted Firecrawl search for the company name. Returns a string of
|
|
1-2 snippet(s), or '' if nothing relevant is found. Never raises."""
|
|
company = (job.get('company') or '').strip()
|
|
if not company:
|
|
return ''
|
|
try:
|
|
results = firecrawl_search(f'{company} about', limit=3)
|
|
except Exception:
|
|
return ''
|
|
facts = []
|
|
for r in results:
|
|
title = (r.get('title') or '').strip()
|
|
desc = (r.get('description') or '').strip()
|
|
blob = f'{title} {desc}'.lower()
|
|
if desc and company.lower() in blob:
|
|
facts.append(f'{title}: {desc}'[:500])
|
|
return '\n'.join(facts[:2])
|
|
|
|
|
|
# ---- contact email scraping (for auto-send on approve) ----
|
|
|
|
_EMAIL_RE = re.compile(r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}')
|
|
_HARD_REJECT = ('noreply', 'no-reply', 'donotreply', 'privacy', 'abuse', 'legal', 'arb',
|
|
'demand', 'press', 'media', 'security@', 'support', 'example', 'yourdomain',
|
|
'email.com', 'sentry', 'wixpress', 'domain.com', 'squarespace', 'cloudflare',
|
|
'compliance', 'trust', 'notice', 'copyright', 'dpo', 'gdpr', 'unsubscribe')
|
|
_RECRUITING_HINT = ('career', 'recruit', 'hiring', 'jobs', 'talent', 'people', 'hr@',
|
|
'work', 'join', 'apply', 'hello', 'team', 'contact', 'info')
|
|
|
|
|
|
def _email_rank(e):
|
|
"""Score an email: -1 = reject (legal/abuse/footer junk), higher = more recruiting-ish."""
|
|
el = e.lower()
|
|
if any(w in el for w in _HARD_REJECT):
|
|
return -1
|
|
score = 0
|
|
for h in _RECRUITING_HINT:
|
|
if h in el:
|
|
score += 1
|
|
return score
|
|
|
|
|
|
def _emails_from_url(url):
|
|
try:
|
|
raw = _open(url, timeout=12)
|
|
text = raw.decode('utf-8', errors='replace')
|
|
found = {}
|
|
for e in _EMAIL_RE.findall(text):
|
|
e = e.rstrip('.,;:>').lower()
|
|
rank = _email_rank(e)
|
|
if rank > 0 and (e not in found or rank > found[e]):
|
|
found[e] = rank
|
|
return sorted(found, key=lambda x: -found[x])
|
|
except Exception:
|
|
return []
|
|
|
|
|
|
def _company_domains(job_url, company):
|
|
"""Derive candidate company domains to check for a contact email."""
|
|
doms = []
|
|
slug = None
|
|
m = re.match(r'https?://(?:job-boards|boards)\.greenhouse\.io/([a-z0-9-]+)', job_url or '')
|
|
if not m:
|
|
m = re.match(r'https?://jobs\.lever\.co/([a-z0-9-]+)', job_url or '')
|
|
if m:
|
|
slug = m.group(1)
|
|
if slug:
|
|
doms += [f'https://{slug}.com', f'https://www.{slug}.com']
|
|
if company:
|
|
cslug = re.sub(r'[^a-z0-9]', '', (company or '').lower())
|
|
if cslug and cslug not in (slug or ''):
|
|
doms += [f'https://{cslug}.com', f'https://www.{cslug}.com']
|
|
return doms
|
|
|
|
|
|
def scrape_contact_email(job):
|
|
"""Best-effort: find a real contact email for the job/company (for auto-send).
|
|
Checks the posting first, then the company's /contact, /careers, /about, /jobs, and
|
|
homepage. Returns an email string or ''. Never raises."""
|
|
url = job.get('url') or ''
|
|
company = job.get('company') or ''
|
|
if not url and not company:
|
|
return ''
|
|
if url:
|
|
emails = _emails_from_url(url)
|
|
if emails:
|
|
return emails[0]
|
|
for base in _company_domains(url, company):
|
|
for path in ('/contact', '/careers', '/about', '/jobs', '/contact-us', '/'):
|
|
emails = _emails_from_url(f'{base}{path}')
|
|
if emails:
|
|
return emails[0]
|
|
return ''
|
|
|
|
|
|
def firecrawl_discover():
|
|
"""Run configured search queries -> job leads. Returns count added."""
|
|
added = 0
|
|
queries = [q.strip() for q in db.get_setting('firecrawl_queries', '').splitlines() if q.strip()]
|
|
for q in queries:
|
|
try:
|
|
for res in firecrawl_search(q, limit=5):
|
|
if not res.get('url') or not res.get('title'):
|
|
continue
|
|
if _is_garbage_title(res.get('title')):
|
|
continue
|
|
# only keep plausible individual postings / listings
|
|
db.upsert_job(
|
|
res.get('title'), '', 'Remote/Web', res.get('url'), 'firecrawl',
|
|
res.get('description', '') or '', None,
|
|
)
|
|
added += 1
|
|
except Exception as e:
|
|
print(f'firecrawl query {q!r} failed: {e}')
|
|
return added
|
|
|
|
|
|
_GARBAGE_TITLE_MARKERS = ('jobsradar', 'linkedin', '1,000+', 'open roles', 'greater ',
|
|
'remote jobs', 'jobs in', 'job board', 'indeed', 'glassdoor',
|
|
'ziprecruiter', 'simplyhired', ' careers', 'search jobs')
|
|
|
|
|
|
def _is_garbage_title(t):
|
|
"""Firecrawl search returns search-result / aggregator pages, not individual postings.
|
|
Skip titles that look like a results page rather than a single role."""
|
|
tl = (t or '').lower()
|
|
return any(m in tl for m in _GARBAGE_TITLE_MARKERS)
|
|
|
|
|
|
def firecrawl_scrape(url):
|
|
"""Scrape a URL to markdown via self-hosted Firecrawl. Returns text or None."""
|
|
base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/')
|
|
payload = json.dumps({'url': url, 'formats': ['markdown']}).encode('utf-8')
|
|
req = urllib.request.Request(f'{base}/v1/scrape', data=payload,
|
|
headers={'Content-Type': 'application/json', 'User-Agent': UA})
|
|
with urllib.request.urlopen(req, timeout=20) as r:
|
|
data = json.loads(r.read().decode('utf-8'))
|
|
return (data.get('data') or {}).get('markdown') or None
|
|
|
|
|
|
def enrich_description(url):
|
|
"""Get a fuller job description for a URL. Firecrawl first, HTTP fallback."""
|
|
try:
|
|
md = firecrawl_scrape(url)
|
|
if md:
|
|
return md
|
|
except Exception:
|
|
pass
|
|
try:
|
|
raw = _open(url, timeout=15)
|
|
# strip tags crudely
|
|
text = re.sub(r'<[^>]+>', ' ', raw.decode('utf-8', errors='replace'))
|
|
return re.sub(r'\s+', ' ', text)[:4000]
|
|
except Exception:
|
|
return None
|