# Procyon — LLM client (Ollama) with graceful keyword fallback. # NEVER runs inference on the CT: it calls the 24/7 LAN Ollama host over HTTP. # Under saturation, a circuit breaker fast-fails so the pipeline never blocks. import json import time import urllib.request import db SYSTEM = ("You are an expert technical recruiter and resume writer. You help a candidate " "present their REAL experience in the strongest honest light. HARD RULE: never " "invent, exaggerate, or fabricate any employer, job title, date, skill, credential, " "or metric that is not present in the source resume. You may reorder, re-emphasize, " "and rephrase only.") # circuit breaker: if the last LLM call failed, skip LLM for COOLDOWN seconds _last_failure = 0.0 COOLDOWN = 90 def _post(url, payload, timeout): req = urllib.request.Request( url, data=json.dumps(payload).encode('utf-8'), headers={'Content-Type': 'application/json'}, ) with urllib.request.urlopen(req, timeout=timeout) as r: return json.loads(r.read().decode('utf-8')) def _timeout(): return int(db.get_setting('llm_timeout', '45')) def llm_available(): """Quick preflight: is the model host answering at all?""" global _last_failure if time.time() - _last_failure < COOLDOWN: return False base = db.get_setting('llm_base_url', 'http://10.30.20.29:11434').rstrip('/') try: _post(f'{base}/api/tags', {}, 4) return True except Exception: _last_failure = time.time() return False def llm_chat(prompt, system=SYSTEM, model=None): """Return text, or None if the model is unavailable/saturated.""" global _last_failure if time.time() - _last_failure < COOLDOWN: return None # circuit open — fall back immediately base = db.get_setting('llm_base_url', 'http://10.30.20.29:11434').rstrip('/') model = model or db.get_setting('llm_model', 'qwen3.8fast:latest') full = f"{system}\n\n{prompt}" if system else prompt try: data = _post(f'{base}/api/generate', { 'model': model, 'prompt': full, 'stream': False, 'think': False, }, _timeout()) txt = (data.get('response') or '').strip() if not txt: raise ValueError('empty response') _last_failure = 0.0 return txt except Exception: _last_failure = time.time() return None def llm_json(prompt, model=None): txt = llm_chat(prompt + '\n\nRespond with ONLY valid JSON, no markdown fences.', model=model) if not txt: return None txt = txt.strip().lstrip('```json').lstrip('```').rstrip('```').strip() try: return json.loads(txt) except Exception: import re m = re.search(r'\{.*\}', txt, re.DOTALL) if m: try: return json.loads(m.group(0)) except Exception: return None return None # ---- keyword fallback (deterministic, always works) ---- def _skill_keywords(): return [ 'python', 'flask', 'sqlite', 'postgres', 'mysql', 'docker', 'proxmox', 'lxc', 'linux', 'bash', 'nginx', 'systemd', 'networking', 'tcp/ip', 'dns', 'vpn', 'automation', 'ci/cd', 'git', 'gitea', 'rest api', 'api', 'json', 'llm', 'ollama', 'gpu', 'machine learning', 'ai', 'agent', 'mcp', 'javascript', 'html', 'css', 'react', 'node', 'full-stack', 'devops', 'sysadmin', 'monitoring', 'grafana', 'prometheus', 'backup', 'security', 'homelab', 'esp32', 'sensor', 'hardware', 'raspberry pi', 'cryptocurrency', 'bitcoin', 'btcpay', 'payments', 'cloudflare', 'tailscale', 'virtualization', 'kvm', ] def keyword_score(description): """Deterministic fit score 0..1 based on skill overlap.""" if not description: return 0.0, 'No description available.' d = description.lower() hits = [k for k in _skill_keywords() if k in d] score = min(1.0, len(hits) / 14.0) matched = ', '.join(hits[:8]) or 'no direct keyword matches' return round(score, 3), f'Keyword overlap: {len(hits)} skills ({matched}).'