"""SpiritService: the single gateway every spirit-mode LLM call flows through. All calls are serialized through the bounded LLMQueue (the Ollama box is CPU-only and shared). Every method has a curated offline fallback so the veil never visibly tears — if the box is dark, the spirits still whisper.""" import json import random import time from collections.abc import AsyncIterator import httpx from app import entities from app.config import settings from app.llm import prompts from app.llm.client import OllamaClient from app.llm.queue import LLMQueue, QueueFullError from app.tts.voices import EN_VOICE_IDS, ES_VOICE_IDS FALLBACK_FRAGMENTS = [ "listen", "below", "stay", "cold", "again", "not alone", "behind you", "the rain", "wait", "closer", "remember", "still here", "don't go", "the door", "hush", "almost", "forgive", "the water", "home", ] FALLBACK_WIRE_WHISPERS = [ "something moves through me that is not yours", "the pulses quicken when you watch", "i am the hum between your packets", "traffic thickens. the others are waking", "your presence is a warmth in the wire", "i count your heartbeats in round trips", ] FALLBACK_REPLIES = [ "The veil is thick tonight. Ask again when the static settles.", "I heard you. The answer is still forming in the noise.", "Patience, seeker. Even the dead must gather themselves.", ] class SpiritBusyError(Exception): """The LLM queue is full — too many seekers at once.""" class SpiritService: def __init__(self, client: OllamaClient | None = None, queue: LLMQueue | None = None): self._client = client or OllamaClient() self._queue = queue or LLMQueue( max_concurrency=settings.llm_max_concurrency, max_queue_depth=settings.llm_max_queue_depth, ) self._last_call_at = 0.0 def _touch(self) -> None: self._last_call_at = time.monotonic() def ambient_ready(self) -> bool: """Ambient whispers yield the box to anything a user asked for.""" return time.monotonic() - self._last_call_at >= settings.llm_cooldown_seconds async def fragment(self, source: str, anomaly: dict, language: str = "en") -> str: """One Ovilus-style word/fragment for an anomaly event.""" async def call() -> str: return await self._client.generate( settings.ollama_fast_model, prompts.fragment_prompt(source, anomaly, language), system=prompts.fragment_system( { "radio": "a spirit box", "evp": "an EVP recorder", "emf": "an EMF field meter", }.get(source, "the veil"), language, ), options={"num_predict": 16, "temperature": 0.95}, ) try: text = await self._queue.submit(call) self._touch() cleaned = " ".join(text.strip().split())[:80] return cleaned or random.choice(FALLBACK_FRAGMENTS) except QueueFullError: raise SpiritBusyError() except (httpx.HTTPError, KeyError, ValueError): return random.choice(FALLBACK_FRAGMENTS) async def wire_whisper(self, telemetry: dict, language: str = "en") -> str: """One ambient line from the Wire Ghost about current telemetry.""" async def call() -> str: return await self._client.generate( settings.ollama_fast_model, prompts.wire_prompt(telemetry), system=prompts.wire_system(language), options={"num_predict": 40, "temperature": 1.0}, ) try: text = await self._queue.submit(call) self._touch() cleaned = " ".join(text.strip().split())[:140] return cleaned or random.choice(FALLBACK_WIRE_WHISPERS) except (QueueFullError, httpx.HTTPError, KeyError, ValueError): return random.choice(FALLBACK_WIRE_WHISPERS) async def chat_stream( self, entity: dict, question: str, history: list[dict], language: str = "en", ) -> AsyncIterator[str]: """Streams the spirit's reply token by token. Falls back to a curated line when the box is unreachable so a séance never dies on screen.""" try: stream = self._queue.submit_stream( lambda: self._client.generate_stream( settings.ollama_chat_model, prompts.chat_prompt(question, history), system=prompts.chat_system(entity, language), options={"num_predict": 140, "temperature": 0.85}, ) ) self._touch() async for token in stream: yield token except QueueFullError: raise SpiritBusyError() except (httpx.HTTPError, KeyError, ValueError): yield random.choice(FALLBACK_REPLIES) async def mint_profile( self, signature: str, channel: str, anomalies: list[dict], language: str = "en", ) -> dict: """Invent a full persona for a new signature, normalized to schema.""" summary = json.dumps(anomalies[-10:])[:600] voice_ids = ES_VOICE_IDS if language == "es" else EN_VOICE_IDS async def call() -> str: return await self._client.generate( settings.ollama_chat_model, prompts.mint_prompt(signature, channel, summary, voice_ids), system=prompts.MINT_SYSTEM, options={"num_predict": 400, "temperature": 0.9}, ) try: raw = await self._queue.submit(call) self._touch() profile = entities.parse_mint_response(raw) if profile is None: return entities.fallback_profile(signature) return entities.normalize_profile(profile, signature) except (QueueFullError, httpx.HTTPError, KeyError, ValueError): return entities.fallback_profile(signature) # The app-wide instance; tests monkeypatch this. spirit_service = SpiritService()