# Procyon — job discovery. RSS/JSON feeds + optional proxy + manual add. # Uses his OWN proxy infra (optional), not an anti-bot evasion rig. import json import re import time import urllib.parse import urllib.request import xml.etree.ElementTree as ET import db UA = 'Procyon/1.0 (personal job-search assistant)' def _open(url, timeout=20): req = urllib.request.Request(url, headers={'User-Agent': UA}) proxy = db.get_setting('proxy_url', '') if proxy: opener = urllib.request.build_opener( urllib.request.ProxyHandler({'http': proxy, 'https': proxy})) else: opener = urllib.request.build_opener() with opener.open(req, timeout=timeout) as r: return r.read() def fetch_rss(url): """Parse an RSS/Atom feed into job dicts.""" data = _open(url) root = ET.fromstring(data) items = [] for item in root.iter('item'): def _t(tag): el = item.find(tag) return el.text.strip() if el is not None and el.text else '' title = _t('title') link = _t('link') desc = _t('description') items.append({'title': title, 'url': link, 'description': desc, 'company': '', 'location': '', 'source': url}) return items def fetch_remoteok(): """RemoteOK JSON API — no key required. Filters to fresh listings: RemoteOK's job pages 302-redirect to the homepage once a posting expires (~30 days), so stale entries would give broken 'Open Application' links.""" data = json.loads(_open('https://remoteok.com/api').decode('utf-8')) out = [] now = time.time() STALE_AFTER = 30 * 86400 # 30 days for j in data[1:]: if not isinstance(j, dict) or not j.get('position'): continue epoch = j.get('epoch') if epoch: try: if now - float(epoch) > STALE_AFTER: continue except (TypeError, ValueError): pass out.append({ 'title': j.get('position'), 'company': j.get('company'), 'location': j.get('location', 'Remote'), 'url': j.get('url'), 'description': (j.get('description') or '')[:4000], 'source': 'remoteok', 'posted_at': j.get('date'), }) return out def fetch_remotive(): """Remotive API — remote jobs JSON, no key. https://remotive.com/api/remote-jobs""" data = json.loads(_open('https://remotive.com/api/remote-jobs').decode('utf-8')) out = [] for j in data.get('jobs', []): out.append({ 'title': j.get('title'), 'company': j.get('company_name'), 'location': j.get('candidate_required_location') or 'Remote', 'url': j.get('url'), 'description': (j.get('description') or '')[:4000], 'source': 'remotive', 'posted_at': j.get('publication_date'), }) return out def fetch_wwr(): """We Work Remotely — remote programming jobs RSS (no key).""" data = _open('https://weworkremotely.com/categories/remote-programming-jobs.rss') root = ET.fromstring(data) items = [] for item in root.iter('item'): def _t(tag): el = item.find(tag) return el.text.strip() if el is not None and el.text else '' items.append({'title': _t('title'), 'url': _t('link'), 'description': _t('description'), 'company': '', 'location': 'Remote', 'source': 'wwr'}) return items # ---- ATS board APIs (Greenhouse / Lever) ---- # The headless-agent discovery path: most tech companies (Seattle + remote) post through # Greenhouse/Lever, which expose clean JSON with no bot wall. Sweep their boards directly. ATS_COMPANIES = [ # Seattle / PNW 'zillow', 'redfin', 'remitly', 'outreach', 'smartsheet', 'expedia', 'offerup', 'highspot', 'rover', 'allenai', 'f5', 'convoy', # Remote-friendly tech 'stripe', 'airbnb', 'doordash', 'datadog', 'snowflake', 'gitlab', 'figma', 'notion', 'openai', 'anthropic', 'discord', 'reddit', 'robinhood', 'coinbase', 'block', 'vercel', 'cloudflare', 'mongodb', 'elastic', 'confluent', 'twilio', 'dropbox', 'asana', 'atlassian', 'postman', 'github', 'hashicorp', ] def _strip_html(s): return re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', ' ', s or '')).strip() def fetch_greenhouse(slug): """Greenhouse board JSON (no key). Returns list of job dicts.""" url = f'https://boards-api.greenhouse.io/v1/boards/{slug}/jobs' data = json.loads(_open(url).decode('utf-8')) out = [] for j in data.get('jobs', []): loc = j.get('location') or {} out.append({ 'title': j.get('title'), 'company': j.get('company_name') or slug, 'location': loc.get('name') if isinstance(loc, dict) else str(loc or 'Remote'), 'url': j.get('absolute_url'), 'description': _strip_html(j.get('content') or '')[:4000], 'source': 'greenhouse', }) return out def fetch_lever(slug): """Lever board JSON (no key). Returns list of job dicts.""" url = f'https://api.lever.co/v0/postings/{slug}?mode=json' data = json.loads(_open(url).decode('utf-8')) if not isinstance(data, list): return [] out = [] for j in data: cats = j.get('categories') or {} out.append({ 'title': j.get('text'), 'company': slug, 'location': cats.get('location') or 'Remote', 'url': j.get('hostedUrl'), 'description': _strip_html(j.get('descriptionPlain') or j.get('description') or '')[:4000], 'source': 'lever', }) return out # Locations worth keeping: WFH (remote) + Seattle-area. Everything else (onsite SF/NY/etc) is noise. _REMOTE_TERMS = ('remote', 'seattle', 'anywhere', 'worldwide', 'work from home', 'wfh', 'united states', 'north america', 'americas') def _is_relevant_location(loc): l = (loc or '').lower() return any(t in l for t in _REMOTE_TERMS) def fetch_ats(): """Sweep the ATS_COMPANIES boards across Greenhouse + Lever. Returns list of job dicts, filtered to remote / Seattle / US-remote locations.""" out = [] for slug in ATS_COMPANIES: for fn, label in ((fetch_greenhouse, 'greenhouse'), (fetch_lever, 'lever')): try: jobs = fn(slug) for j in jobs: if _is_relevant_location(j.get('location')): out.append(j) except Exception: continue return out def indeed_feed(query, location): """Indeed RSS endpoint (public, returns XML).""" q = urllib.parse.quote_plus(query) l = urllib.parse.quote_plus(location) return f'https://www.indeed.com/rss?q={q}&l={l}' def ingest(items): added = 0 for it in items: if not it.get('title') or not it.get('url'): continue jid = db.upsert_job( it.get('title'), it.get('company') or '', it.get('location') or '', it.get('url'), it.get('source', 'manual'), it.get('description') or '', it.get('posted_at'), ) if jid: added += 1 return added def discover(query='software engineer', location='seattle wa', sources=None): """Pull from configured sources. Returns count added.""" added = 0 # RemoteOK (remote) try: added += ingest(fetch_remoteok()) except Exception as e: print(f'remoteok failed: {e}') # Remotive (remote) try: added += ingest(fetch_remotive()) except Exception as e: print(f'remotive failed: {e}') # We Work Remotely (remote) try: added += ingest(fetch_wwr()) except Exception as e: print(f'wwr failed: {e}') # ATS boards (Greenhouse + Lever) — Seattle + remote tech companies try: added += ingest(fetch_ats()) except Exception as e: print(f'ats failed: {e}') # Indeed RSS (dead — left in try/except, non-functional) try: added += ingest(fetch_rss(indeed_feed(query, location))) except Exception as e: print(f'indeed rss failed: {e}') # Firecrawl search (self-hosted, /v1/search works; scrape is bot-walled) if db.get_setting('firecrawl_enabled', '1') == '1': added += firecrawl_discover() # configured extra feeds feeds = db.get_setting('rss_feeds', '') for f in [x.strip() for x in feeds.splitlines() if x.strip()]: try: added += ingest(fetch_rss(f)) except Exception as e: print(f'feed {f} failed: {e}') return added # ---- Firecrawl (self-hosted) ---- def firecrawl_search(query, limit=5): """Self-hosted Firecrawl /v1/search. Returns list of {url, title, description}.""" base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/') payload = json.dumps({'query': query, 'limit': limit}).encode('utf-8') req = urllib.request.Request(f'{base}/v1/search', data=payload, headers={'Content-Type': 'application/json', 'User-Agent': UA}) with urllib.request.urlopen(req, timeout=30) as r: data = json.loads(r.read().decode('utf-8')) return data.get('data', []) if data.get('success') else [] def research_company(job): """Best-effort company research: a short factual blurb to weave into the email. Uses self-hosted Firecrawl search for the company name. Returns a string of 1-2 snippet(s), or '' if nothing relevant is found. Never raises.""" company = (job.get('company') or '').strip() if not company: return '' try: results = firecrawl_search(f'{company} about', limit=3) except Exception: return '' facts = [] for r in results: title = (r.get('title') or '').strip() desc = (r.get('description') or '').strip() blob = f'{title} {desc}'.lower() if desc and company.lower() in blob: facts.append(f'{title}: {desc}'[:500]) return '\n'.join(facts[:2]) # ---- contact email scraping (for auto-send on approve) ---- _EMAIL_RE = re.compile(r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}') _HARD_REJECT = ('noreply', 'no-reply', 'donotreply', 'privacy', 'abuse', 'legal', 'arb', 'demand', 'press', 'media', 'security@', 'support', 'example', 'yourdomain', 'email.com', 'sentry', 'wixpress', 'domain.com', 'squarespace', 'cloudflare', 'compliance', 'trust', 'notice', 'copyright', 'dpo', 'gdpr', 'unsubscribe') _RECRUITING_HINT = ('career', 'recruit', 'hiring', 'jobs', 'talent', 'people', 'hr@', 'work', 'join', 'apply', 'hello', 'team', 'contact', 'info') def _email_rank(e): """Score an email: -1 = reject (legal/abuse/footer junk), higher = more recruiting-ish.""" el = e.lower() if any(w in el for w in _HARD_REJECT): return -1 score = 0 for h in _RECRUITING_HINT: if h in el: score += 1 return score def _emails_from_url(url): try: raw = _open(url, timeout=12) text = raw.decode('utf-8', errors='replace') found = {} for e in _EMAIL_RE.findall(text): e = e.rstrip('.,;:>').lower() rank = _email_rank(e) if rank > 0 and (e not in found or rank > found[e]): found[e] = rank return sorted(found, key=lambda x: -found[x]) except Exception: return [] def _company_domains(job_url, company): """Derive candidate company domains to check for a contact email.""" doms = [] slug = None m = re.match(r'https?://(?:job-boards|boards)\.greenhouse\.io/([a-z0-9-]+)', job_url or '') if not m: m = re.match(r'https?://jobs\.lever\.co/([a-z0-9-]+)', job_url or '') if m: slug = m.group(1) if slug: doms += [f'https://{slug}.com', f'https://www.{slug}.com'] if company: cslug = re.sub(r'[^a-z0-9]', '', (company or '').lower()) if cslug and cslug not in (slug or ''): doms += [f'https://{cslug}.com', f'https://www.{cslug}.com'] return doms def scrape_contact_email(job): """Best-effort: find a real contact email for the job/company (for auto-send). Checks the posting first, then the company's /contact, /careers, /about, /jobs, and homepage. Returns an email string or ''. Never raises.""" url = job.get('url') or '' company = job.get('company') or '' if not url and not company: return '' if url: emails = _emails_from_url(url) if emails: return emails[0] for base in _company_domains(url, company): for path in ('/contact', '/careers', '/about', '/jobs', '/contact-us', '/'): emails = _emails_from_url(f'{base}{path}') if emails: return emails[0] return '' def firecrawl_discover(): """Run configured search queries -> job leads. Returns count added.""" added = 0 queries = [q.strip() for q in db.get_setting('firecrawl_queries', '').splitlines() if q.strip()] for q in queries: try: for res in firecrawl_search(q, limit=5): if not res.get('url') or not res.get('title'): continue if _is_garbage_title(res.get('title')): continue # only keep plausible individual postings / listings db.upsert_job( res.get('title'), '', 'Remote/Web', res.get('url'), 'firecrawl', res.get('description', '') or '', None, ) added += 1 except Exception as e: print(f'firecrawl query {q!r} failed: {e}') return added _GARBAGE_TITLE_MARKERS = ('jobsradar', 'linkedin', '1,000+', 'open roles', 'greater ', 'remote jobs', 'jobs in', 'job board', 'indeed', 'glassdoor', 'ziprecruiter', 'simplyhired', ' careers', 'search jobs') def _is_garbage_title(t): """Firecrawl search returns search-result / aggregator pages, not individual postings. Skip titles that look like a results page rather than a single role.""" tl = (t or '').lower() return any(m in tl for m in _GARBAGE_TITLE_MARKERS) def firecrawl_scrape(url): """Scrape a URL to markdown via self-hosted Firecrawl. Returns text or None.""" base = db.get_setting('firecrawl_url', 'http://10.30.20.182:3002').rstrip('/') payload = json.dumps({'url': url, 'formats': ['markdown']}).encode('utf-8') req = urllib.request.Request(f'{base}/v1/scrape', data=payload, headers={'Content-Type': 'application/json', 'User-Agent': UA}) with urllib.request.urlopen(req, timeout=20) as r: data = json.loads(r.read().decode('utf-8')) return (data.get('data') or {}).get('markdown') or None def enrich_description(url): """Get a fuller job description for a URL. Firecrawl first, HTTP fallback.""" try: md = firecrawl_scrape(url) if md: return md except Exception: pass try: raw = _open(url, timeout=15) # strip tags crudely text = re.sub(r'<[^>]+>', ' ', raw.decode('utf-8', errors='replace')) return re.sub(r'\s+', ' ', text)[:4000] except Exception: return None