feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
This commit is contained in:
161
backend/app/llm/service.py
Normal file
161
backend/app/llm/service.py
Normal file
@@ -0,0 +1,161 @@
|
||||
"""SpiritService: the single gateway every spirit-mode LLM call flows through.
|
||||
|
||||
All calls are serialized through the bounded LLMQueue (the Ollama box is
|
||||
CPU-only and shared). Every method has a curated offline fallback so the veil
|
||||
never visibly tears — if the box is dark, the spirits still whisper."""
|
||||
|
||||
import json
|
||||
import random
|
||||
import time
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from app import entities
|
||||
from app.config import settings
|
||||
from app.llm import prompts
|
||||
from app.llm.client import OllamaClient
|
||||
from app.llm.queue import LLMQueue, QueueFullError
|
||||
from app.tts.voices import EN_VOICE_IDS, ES_VOICE_IDS
|
||||
|
||||
FALLBACK_FRAGMENTS = [
|
||||
"listen", "below", "stay", "cold", "again", "not alone", "behind you",
|
||||
"the rain", "wait", "closer", "remember", "still here", "don't go",
|
||||
"the door", "hush", "almost", "forgive", "the water", "home",
|
||||
]
|
||||
|
||||
FALLBACK_WIRE_WHISPERS = [
|
||||
"something moves through me that is not yours",
|
||||
"the pulses quicken when you watch",
|
||||
"i am the hum between your packets",
|
||||
"traffic thickens. the others are waking",
|
||||
"your presence is a warmth in the wire",
|
||||
"i count your heartbeats in round trips",
|
||||
]
|
||||
|
||||
FALLBACK_REPLIES = [
|
||||
"The veil is thick tonight. Ask again when the static settles.",
|
||||
"I heard you. The answer is still forming in the noise.",
|
||||
"Patience, seeker. Even the dead must gather themselves.",
|
||||
]
|
||||
|
||||
|
||||
class SpiritBusyError(Exception):
|
||||
"""The LLM queue is full — too many seekers at once."""
|
||||
|
||||
|
||||
class SpiritService:
|
||||
def __init__(self, client: OllamaClient | None = None, queue: LLMQueue | None = None):
|
||||
self._client = client or OllamaClient()
|
||||
self._queue = queue or LLMQueue(
|
||||
max_concurrency=settings.llm_max_concurrency,
|
||||
max_queue_depth=settings.llm_max_queue_depth,
|
||||
)
|
||||
self._last_call_at = 0.0
|
||||
|
||||
def _touch(self) -> None:
|
||||
self._last_call_at = time.monotonic()
|
||||
|
||||
def ambient_ready(self) -> bool:
|
||||
"""Ambient whispers yield the box to anything a user asked for."""
|
||||
return time.monotonic() - self._last_call_at >= settings.llm_cooldown_seconds
|
||||
|
||||
async def fragment(self, source: str, anomaly: dict, language: str = "en") -> str:
|
||||
"""One Ovilus-style word/fragment for an anomaly event."""
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_fast_model,
|
||||
prompts.fragment_prompt(source, anomaly, language),
|
||||
system=prompts.fragment_system(
|
||||
"a spirit box" if source == "radio" else "an EVP recorder", language
|
||||
),
|
||||
options={"num_predict": 16, "temperature": 0.95},
|
||||
)
|
||||
|
||||
try:
|
||||
text = await self._queue.submit(call)
|
||||
self._touch()
|
||||
cleaned = " ".join(text.strip().split())[:80]
|
||||
return cleaned or random.choice(FALLBACK_FRAGMENTS)
|
||||
except QueueFullError:
|
||||
raise SpiritBusyError()
|
||||
except (httpx.HTTPError, KeyError, ValueError):
|
||||
return random.choice(FALLBACK_FRAGMENTS)
|
||||
|
||||
async def wire_whisper(self, telemetry: dict, language: str = "en") -> str:
|
||||
"""One ambient line from the Wire Ghost about current telemetry."""
|
||||
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_fast_model,
|
||||
prompts.wire_prompt(telemetry),
|
||||
system=prompts.wire_system(language),
|
||||
options={"num_predict": 40, "temperature": 1.0},
|
||||
)
|
||||
|
||||
try:
|
||||
text = await self._queue.submit(call)
|
||||
self._touch()
|
||||
cleaned = " ".join(text.strip().split())[:140]
|
||||
return cleaned or random.choice(FALLBACK_WIRE_WHISPERS)
|
||||
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
|
||||
return random.choice(FALLBACK_WIRE_WHISPERS)
|
||||
|
||||
async def chat_stream(
|
||||
self,
|
||||
entity: dict,
|
||||
question: str,
|
||||
history: list[dict],
|
||||
language: str = "en",
|
||||
) -> AsyncIterator[str]:
|
||||
"""Streams the spirit's reply token by token. Falls back to a curated
|
||||
line when the box is unreachable so a séance never dies on screen."""
|
||||
try:
|
||||
stream = self._queue.submit_stream(
|
||||
lambda: self._client.generate_stream(
|
||||
settings.ollama_chat_model,
|
||||
prompts.chat_prompt(question, history),
|
||||
system=prompts.chat_system(entity, language),
|
||||
options={"num_predict": 140, "temperature": 0.85},
|
||||
)
|
||||
)
|
||||
self._touch()
|
||||
async for token in stream:
|
||||
yield token
|
||||
except QueueFullError:
|
||||
raise SpiritBusyError()
|
||||
except (httpx.HTTPError, KeyError, ValueError):
|
||||
yield random.choice(FALLBACK_REPLIES)
|
||||
|
||||
async def mint_profile(
|
||||
self,
|
||||
signature: str,
|
||||
channel: str,
|
||||
anomalies: list[dict],
|
||||
language: str = "en",
|
||||
) -> dict:
|
||||
"""Invent a full persona for a new signature, normalized to schema."""
|
||||
summary = json.dumps(anomalies[-10:])[:600]
|
||||
voice_ids = ES_VOICE_IDS if language == "es" else EN_VOICE_IDS
|
||||
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_chat_model,
|
||||
prompts.mint_prompt(signature, channel, summary, voice_ids),
|
||||
system=prompts.MINT_SYSTEM,
|
||||
options={"num_predict": 400, "temperature": 0.9},
|
||||
)
|
||||
|
||||
try:
|
||||
raw = await self._queue.submit(call)
|
||||
self._touch()
|
||||
profile = entities.parse_mint_response(raw)
|
||||
if profile is None:
|
||||
return entities.fallback_profile(signature)
|
||||
return entities.normalize_profile(profile, signature)
|
||||
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
|
||||
return entities.fallback_profile(signature)
|
||||
|
||||
|
||||
# The app-wide instance; tests monkeypatch this.
|
||||
spirit_service = SpiritService()
|
||||
Reference in New Issue
Block a user