Files
procyon/llm.py
2026-10-06 23:43:21 -07:00

117 lines
4.1 KiB
Python

# Procyon — LLM client (Ollama) with graceful keyword fallback.
# NEVER runs inference on the CT: it calls the 24/7 LAN Ollama host over HTTP.
# Under saturation, a circuit breaker fast-fails so the pipeline never blocks.
import json
import time
import urllib.request
import db
SYSTEM = ("You are an expert technical recruiter and resume writer. You help a candidate "
"present their REAL experience in the strongest honest light. HARD RULE: never "
"invent, exaggerate, or fabricate any employer, job title, date, skill, credential, "
"or metric that is not present in the source resume. You may reorder, re-emphasize, "
"and rephrase only.")
# circuit breaker: if the last LLM call failed, skip LLM for COOLDOWN seconds
_last_failure = 0.0
COOLDOWN = 90
def _post(url, payload, timeout):
req = urllib.request.Request(
url,
data=json.dumps(payload).encode('utf-8'),
headers={'Content-Type': 'application/json'},
)
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read().decode('utf-8'))
def _timeout():
return int(db.get_setting('llm_timeout', '45'))
def llm_available():
"""Quick preflight: is the model host answering at all?"""
global _last_failure
if time.time() - _last_failure < COOLDOWN:
return False
base = db.get_setting('llm_base_url', 'http://10.30.20.29:11434').rstrip('/')
try:
_post(f'{base}/api/tags', {}, 4)
return True
except Exception:
_last_failure = time.time()
return False
def llm_chat(prompt, system=SYSTEM, model=None):
"""Return text, or None if the model is unavailable/saturated."""
global _last_failure
if time.time() - _last_failure < COOLDOWN:
return None # circuit open — fall back immediately
base = db.get_setting('llm_base_url', 'http://10.30.20.29:11434').rstrip('/')
model = model or db.get_setting('llm_model', 'qwen3.8fast:latest')
full = f"{system}\n\n{prompt}" if system else prompt
try:
data = _post(f'{base}/api/generate', {
'model': model,
'prompt': full,
'stream': False,
'think': False,
}, _timeout())
txt = (data.get('response') or '').strip()
if not txt:
raise ValueError('empty response')
_last_failure = 0.0
return txt
except Exception:
_last_failure = time.time()
return None
def llm_json(prompt, model=None):
txt = llm_chat(prompt + '\n\nRespond with ONLY valid JSON, no markdown fences.', model=model)
if not txt:
return None
txt = txt.strip().lstrip('```json').lstrip('```').rstrip('```').strip()
try:
return json.loads(txt)
except Exception:
import re
m = re.search(r'\{.*\}', txt, re.DOTALL)
if m:
try:
return json.loads(m.group(0))
except Exception:
return None
return None
# ---- keyword fallback (deterministic, always works) ----
def _skill_keywords():
return [
'python', 'flask', 'sqlite', 'postgres', 'mysql', 'docker', 'proxmox', 'lxc',
'linux', 'bash', 'nginx', 'systemd', 'networking', 'tcp/ip', 'dns', 'vpn',
'automation', 'ci/cd', 'git', 'gitea', 'rest api', 'api', 'json', 'llm',
'ollama', 'gpu', 'machine learning', 'ai', 'agent', 'mcp', 'javascript',
'html', 'css', 'react', 'node', 'full-stack', 'devops', 'sysadmin',
'monitoring', 'grafana', 'prometheus', 'backup', 'security', 'homelab',
'esp32', 'sensor', 'hardware', 'raspberry pi', 'cryptocurrency', 'bitcoin',
'btcpay', 'payments', 'cloudflare', 'tailscale', 'virtualization', 'kvm',
]
def keyword_score(description):
"""Deterministic fit score 0..1 based on skill overlap."""
if not description:
return 0.0, 'No description available.'
d = description.lower()
hits = [k for k in _skill_keywords() if k in d]
score = min(1.0, len(hits) / 14.0)
matched = ', '.join(hits[:8]) or 'no direct keyword matches'
return round(score, 3), f'Keyword overlap: {len(hits)} skills ({matched}).'