Files
qtalker---/backend/app/llm/service.py
Indiana b9110f45de feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
  passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
  fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
  effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
  with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
2026-07-20 20:13:28 +00:00

162 lines
6.0 KiB
Python

"""SpiritService: the single gateway every spirit-mode LLM call flows through.
All calls are serialized through the bounded LLMQueue (the Ollama box is
CPU-only and shared). Every method has a curated offline fallback so the veil
never visibly tears — if the box is dark, the spirits still whisper."""
import json
import random
import time
from collections.abc import AsyncIterator
import httpx
from app import entities
from app.config import settings
from app.llm import prompts
from app.llm.client import OllamaClient
from app.llm.queue import LLMQueue, QueueFullError
from app.tts.voices import EN_VOICE_IDS, ES_VOICE_IDS
FALLBACK_FRAGMENTS = [
"listen", "below", "stay", "cold", "again", "not alone", "behind you",
"the rain", "wait", "closer", "remember", "still here", "don't go",
"the door", "hush", "almost", "forgive", "the water", "home",
]
FALLBACK_WIRE_WHISPERS = [
"something moves through me that is not yours",
"the pulses quicken when you watch",
"i am the hum between your packets",
"traffic thickens. the others are waking",
"your presence is a warmth in the wire",
"i count your heartbeats in round trips",
]
FALLBACK_REPLIES = [
"The veil is thick tonight. Ask again when the static settles.",
"I heard you. The answer is still forming in the noise.",
"Patience, seeker. Even the dead must gather themselves.",
]
class SpiritBusyError(Exception):
"""The LLM queue is full — too many seekers at once."""
class SpiritService:
def __init__(self, client: OllamaClient | None = None, queue: LLMQueue | None = None):
self._client = client or OllamaClient()
self._queue = queue or LLMQueue(
max_concurrency=settings.llm_max_concurrency,
max_queue_depth=settings.llm_max_queue_depth,
)
self._last_call_at = 0.0
def _touch(self) -> None:
self._last_call_at = time.monotonic()
def ambient_ready(self) -> bool:
"""Ambient whispers yield the box to anything a user asked for."""
return time.monotonic() - self._last_call_at >= settings.llm_cooldown_seconds
async def fragment(self, source: str, anomaly: dict, language: str = "en") -> str:
"""One Ovilus-style word/fragment for an anomaly event."""
async def call() -> str:
return await self._client.generate(
settings.ollama_fast_model,
prompts.fragment_prompt(source, anomaly, language),
system=prompts.fragment_system(
"a spirit box" if source == "radio" else "an EVP recorder", language
),
options={"num_predict": 16, "temperature": 0.95},
)
try:
text = await self._queue.submit(call)
self._touch()
cleaned = " ".join(text.strip().split())[:80]
return cleaned or random.choice(FALLBACK_FRAGMENTS)
except QueueFullError:
raise SpiritBusyError()
except (httpx.HTTPError, KeyError, ValueError):
return random.choice(FALLBACK_FRAGMENTS)
async def wire_whisper(self, telemetry: dict, language: str = "en") -> str:
"""One ambient line from the Wire Ghost about current telemetry."""
async def call() -> str:
return await self._client.generate(
settings.ollama_fast_model,
prompts.wire_prompt(telemetry),
system=prompts.wire_system(language),
options={"num_predict": 40, "temperature": 1.0},
)
try:
text = await self._queue.submit(call)
self._touch()
cleaned = " ".join(text.strip().split())[:140]
return cleaned or random.choice(FALLBACK_WIRE_WHISPERS)
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
return random.choice(FALLBACK_WIRE_WHISPERS)
async def chat_stream(
self,
entity: dict,
question: str,
history: list[dict],
language: str = "en",
) -> AsyncIterator[str]:
"""Streams the spirit's reply token by token. Falls back to a curated
line when the box is unreachable so a séance never dies on screen."""
try:
stream = self._queue.submit_stream(
lambda: self._client.generate_stream(
settings.ollama_chat_model,
prompts.chat_prompt(question, history),
system=prompts.chat_system(entity, language),
options={"num_predict": 140, "temperature": 0.85},
)
)
self._touch()
async for token in stream:
yield token
except QueueFullError:
raise SpiritBusyError()
except (httpx.HTTPError, KeyError, ValueError):
yield random.choice(FALLBACK_REPLIES)
async def mint_profile(
self,
signature: str,
channel: str,
anomalies: list[dict],
language: str = "en",
) -> dict:
"""Invent a full persona for a new signature, normalized to schema."""
summary = json.dumps(anomalies[-10:])[:600]
voice_ids = ES_VOICE_IDS if language == "es" else EN_VOICE_IDS
async def call() -> str:
return await self._client.generate(
settings.ollama_chat_model,
prompts.mint_prompt(signature, channel, summary, voice_ids),
system=prompts.MINT_SYSTEM,
options={"num_predict": 400, "temperature": 0.9},
)
try:
raw = await self._queue.submit(call)
self._touch()
profile = entities.parse_mint_response(raw)
if profile is None:
return entities.fallback_profile(signature)
return entities.normalize_profile(profile, signature)
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
return entities.fallback_profile(signature)
# The app-wide instance; tests monkeypatch this.
spirit_service = SpiritService()