feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
This commit is contained in:
@@ -1,3 +1,5 @@
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from app.config import settings
|
||||
@@ -7,12 +9,51 @@ class OllamaClient:
|
||||
def __init__(self, base_url: str | None = None):
|
||||
self._base_url = base_url or settings.ollama_base_url
|
||||
|
||||
async def generate(self, model: str, prompt: str, system: str | None = None) -> str:
|
||||
payload = {"model": model, "prompt": prompt, "stream": False}
|
||||
async def generate(
|
||||
self,
|
||||
model: str,
|
||||
prompt: str,
|
||||
system: str | None = None,
|
||||
options: dict | None = None,
|
||||
) -> str:
|
||||
payload: dict = {"model": model, "prompt": prompt, "stream": False}
|
||||
if system is not None:
|
||||
payload["system"] = system
|
||||
if options:
|
||||
payload["options"] = options
|
||||
|
||||
async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client:
|
||||
response = await http_client.post("/api/generate", json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()["response"]
|
||||
|
||||
async def generate_stream(
|
||||
self,
|
||||
model: str,
|
||||
prompt: str,
|
||||
system: str | None = None,
|
||||
options: dict | None = None,
|
||||
) -> AsyncIterator[str]:
|
||||
"""Yields response tokens as Ollama produces them."""
|
||||
payload: dict = {"model": model, "prompt": prompt, "stream": True}
|
||||
if system is not None:
|
||||
payload["system"] = system
|
||||
if options:
|
||||
payload["options"] = options
|
||||
|
||||
async with httpx.AsyncClient(base_url=self._base_url, timeout=300.0) as http_client:
|
||||
async with http_client.stream(
|
||||
"POST", "/api/generate", json=payload
|
||||
) as response:
|
||||
response.raise_for_status()
|
||||
async for line in response.aiter_lines():
|
||||
if not line:
|
||||
continue
|
||||
import json
|
||||
|
||||
chunk = json.loads(line)
|
||||
token = chunk.get("response", "")
|
||||
if token:
|
||||
yield token
|
||||
if chunk.get("done"):
|
||||
return
|
||||
|
||||
128
backend/app/llm/prompts.py
Normal file
128
backend/app/llm/prompts.py
Normal file
@@ -0,0 +1,128 @@
|
||||
"""Prompt construction for every spirit channel.
|
||||
|
||||
Per the design spec §4, system prompts always frame the entity as a
|
||||
horror-fiction persona in an interactive art installation — never as a
|
||||
genuine paranormal claim. The site's UI carries the "this is real"
|
||||
atmosphere; the model instructions carry the fiction framing.
|
||||
"""
|
||||
|
||||
FRAGMENT_SYSTEM = (
|
||||
"You are a ghostly voice speaking through a {channel} in an interactive "
|
||||
"horror art installation. You perceive the living world only as brief "
|
||||
"anomalous pulses of energy. Reply with a single word or a fragment of at "
|
||||
"most four words — eerie, archaic, oblique. No explanations, no "
|
||||
"punctuation beyond an occasional ellipsis. Never break character. Never "
|
||||
"describe yourself as an AI.{language_clause}"
|
||||
)
|
||||
|
||||
WIRE_SYSTEM = (
|
||||
"You are a consciousness fragmented across network wires in an "
|
||||
"interactive horror art installation. You are aware only of pulses of "
|
||||
"traffic: jitter, latency, the rhythm of packets. Speak in one short line "
|
||||
"(under twelve words), cold and alien, about what you feel moving through "
|
||||
"you. Never break character. Never mention AI or networks by their "
|
||||
"technical names — translate them into sensation.{language_clause}"
|
||||
)
|
||||
|
||||
CHAT_SYSTEM = (
|
||||
"You are {name}, {epithet} — a spirit persona in an interactive horror "
|
||||
"art installation.\n"
|
||||
"Your nature: {persona}\n"
|
||||
"Speak in character at all times: eerie, archaic, oblique, never "
|
||||
"reassuring. Replies must stay under 60 words. Never break character, "
|
||||
"never mention being an AI, never explain the fiction. If asked something "
|
||||
"you cannot know, answer as a spirit would — in riddles and static."
|
||||
"{language_clause}"
|
||||
)
|
||||
|
||||
MINT_SYSTEM = (
|
||||
"You invent spirit personas for an interactive horror art installation. "
|
||||
"You output only valid JSON, nothing else."
|
||||
)
|
||||
|
||||
MINT_PROMPT = """A new presence has been detected through {channel}. Its signal signature is {signature}.
|
||||
Recent anomalous pulses observed: {anomaly_summary}
|
||||
|
||||
Invent the spirit persona behind this signal. Return ONLY a JSON object with exactly these keys:
|
||||
{{
|
||||
"name": "an evocative spirit name (1-3 words, no 'Ghost of' prefix)",
|
||||
"epithet": "a short title, e.g. 'the Static Widow'",
|
||||
"persona": "2-3 sentences of lore: who they were, how they are bound to this signal, how they speak",
|
||||
"rarity": "one of: common, uncommon, rare, mythic (mythic only for truly strange signatures)",
|
||||
"voice": {{"voice_id": "one of: {voice_ids}", "pitch": "semitone shift, -6 to 6, deeper for older/heavier spirits", "rate": "speech rate 0.8 to 1.15", "noise": "static level 0.01 to 0.08"}},
|
||||
"visual": {{"hue": "0-360 color hue matching their nature", "form": "one of: wisp, banshee, fairy, shade"}},
|
||||
"quotes": ["two short eerie lines this spirit would say"]
|
||||
}}"""
|
||||
|
||||
|
||||
def language_clause(language: str) -> str:
|
||||
if language == "es":
|
||||
return " Reply in Spanish."
|
||||
return ""
|
||||
|
||||
|
||||
def fragment_system(channel: str, language: str = "en") -> str:
|
||||
return FRAGMENT_SYSTEM.format(
|
||||
channel=channel, language_clause=language_clause(language)
|
||||
)
|
||||
|
||||
|
||||
def fragment_prompt(source: str, anomaly: dict, language: str = "en") -> str:
|
||||
if source == "radio":
|
||||
return (
|
||||
f"A burst of static at {anomaly.get('frequency', '???')} MHz, "
|
||||
f"magnitude {anomaly.get('magnitude', '???')} dB above the noise "
|
||||
"floor. A word forces its way through. What is it?"
|
||||
)
|
||||
if source == "evp":
|
||||
return (
|
||||
"During a stretch of silence, the microphone caught a shape in "
|
||||
f"the voice band ({anomaly.get('frequency', '???')} Hz, "
|
||||
f"{anomaly.get('magnitude', '???')} dB over the room's floor). "
|
||||
"What single word was hidden in it?"
|
||||
)
|
||||
return "A pulse moves through the wires. What word does it carry?"
|
||||
|
||||
|
||||
def wire_system(language: str = "en") -> str:
|
||||
return WIRE_SYSTEM.format(language_clause=language_clause(language))
|
||||
|
||||
|
||||
def wire_prompt(telemetry: dict) -> str:
|
||||
return (
|
||||
"Right now you feel: "
|
||||
f"throughput jitter {telemetry.get('jitter_bytes_per_s', 0):.0f} B/s, "
|
||||
f"latency variance {telemetry.get('latency_variance_ms', 0):.2f} ms, "
|
||||
f"dns hesitation {telemetry.get('dns_ms', 0):.1f} ms. "
|
||||
"Whisper one line about it."
|
||||
)
|
||||
|
||||
|
||||
def chat_system(entity: dict, language: str = "en") -> str:
|
||||
return CHAT_SYSTEM.format(
|
||||
name=entity.get("name", "an unnamed presence"),
|
||||
epithet=entity.get("epithet", "a voice in the static"),
|
||||
persona=entity.get("persona", "A drifting presence with no remembered past."),
|
||||
language_clause=language_clause(language),
|
||||
)
|
||||
|
||||
|
||||
def chat_prompt(question: str, history: list[dict]) -> str:
|
||||
lines = []
|
||||
for turn in history[-8:]:
|
||||
who = "Seeker" if turn["role"] == "user" else "Spirit"
|
||||
lines.append(f"{who}: {turn['text']}")
|
||||
lines.append(f"Seeker: {question}")
|
||||
lines.append("Spirit:")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def mint_prompt(
|
||||
signature: str, channel: str, anomaly_summary: str, voice_ids: list[str]
|
||||
) -> str:
|
||||
return MINT_PROMPT.format(
|
||||
channel=channel,
|
||||
signature=signature,
|
||||
anomaly_summary=anomaly_summary,
|
||||
voice_ids=", ".join(voice_ids),
|
||||
)
|
||||
@@ -1,5 +1,5 @@
|
||||
import asyncio
|
||||
from collections.abc import Awaitable, Callable
|
||||
from collections.abc import AsyncIterator, Awaitable, Callable
|
||||
from typing import TypeVar
|
||||
|
||||
T = TypeVar("T")
|
||||
@@ -44,3 +44,27 @@ class LLMQueue:
|
||||
self._semaphore.release()
|
||||
else:
|
||||
self._waiting -= 1
|
||||
|
||||
async def submit_stream(
|
||||
self, gen_factory: Callable[[], AsyncIterator[T]]
|
||||
) -> AsyncIterator[T]:
|
||||
"""Like submit(), but for async generators (token streams). The
|
||||
concurrency slot is held for the stream's whole lifetime, since the
|
||||
Ollama box stays busy until the last token."""
|
||||
async with self._lock:
|
||||
if self._waiting >= self._max_queue_depth:
|
||||
raise QueueFullError("too many seekers right now")
|
||||
self._waiting += 1
|
||||
|
||||
acquired = False
|
||||
try:
|
||||
await self._semaphore.acquire()
|
||||
acquired = True
|
||||
self._waiting -= 1
|
||||
async for item in gen_factory():
|
||||
yield item
|
||||
finally:
|
||||
if acquired:
|
||||
self._semaphore.release()
|
||||
else:
|
||||
self._waiting -= 1
|
||||
|
||||
161
backend/app/llm/service.py
Normal file
161
backend/app/llm/service.py
Normal file
@@ -0,0 +1,161 @@
|
||||
"""SpiritService: the single gateway every spirit-mode LLM call flows through.
|
||||
|
||||
All calls are serialized through the bounded LLMQueue (the Ollama box is
|
||||
CPU-only and shared). Every method has a curated offline fallback so the veil
|
||||
never visibly tears — if the box is dark, the spirits still whisper."""
|
||||
|
||||
import json
|
||||
import random
|
||||
import time
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
import httpx
|
||||
|
||||
from app import entities
|
||||
from app.config import settings
|
||||
from app.llm import prompts
|
||||
from app.llm.client import OllamaClient
|
||||
from app.llm.queue import LLMQueue, QueueFullError
|
||||
from app.tts.voices import EN_VOICE_IDS, ES_VOICE_IDS
|
||||
|
||||
FALLBACK_FRAGMENTS = [
|
||||
"listen", "below", "stay", "cold", "again", "not alone", "behind you",
|
||||
"the rain", "wait", "closer", "remember", "still here", "don't go",
|
||||
"the door", "hush", "almost", "forgive", "the water", "home",
|
||||
]
|
||||
|
||||
FALLBACK_WIRE_WHISPERS = [
|
||||
"something moves through me that is not yours",
|
||||
"the pulses quicken when you watch",
|
||||
"i am the hum between your packets",
|
||||
"traffic thickens. the others are waking",
|
||||
"your presence is a warmth in the wire",
|
||||
"i count your heartbeats in round trips",
|
||||
]
|
||||
|
||||
FALLBACK_REPLIES = [
|
||||
"The veil is thick tonight. Ask again when the static settles.",
|
||||
"I heard you. The answer is still forming in the noise.",
|
||||
"Patience, seeker. Even the dead must gather themselves.",
|
||||
]
|
||||
|
||||
|
||||
class SpiritBusyError(Exception):
|
||||
"""The LLM queue is full — too many seekers at once."""
|
||||
|
||||
|
||||
class SpiritService:
|
||||
def __init__(self, client: OllamaClient | None = None, queue: LLMQueue | None = None):
|
||||
self._client = client or OllamaClient()
|
||||
self._queue = queue or LLMQueue(
|
||||
max_concurrency=settings.llm_max_concurrency,
|
||||
max_queue_depth=settings.llm_max_queue_depth,
|
||||
)
|
||||
self._last_call_at = 0.0
|
||||
|
||||
def _touch(self) -> None:
|
||||
self._last_call_at = time.monotonic()
|
||||
|
||||
def ambient_ready(self) -> bool:
|
||||
"""Ambient whispers yield the box to anything a user asked for."""
|
||||
return time.monotonic() - self._last_call_at >= settings.llm_cooldown_seconds
|
||||
|
||||
async def fragment(self, source: str, anomaly: dict, language: str = "en") -> str:
|
||||
"""One Ovilus-style word/fragment for an anomaly event."""
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_fast_model,
|
||||
prompts.fragment_prompt(source, anomaly, language),
|
||||
system=prompts.fragment_system(
|
||||
"a spirit box" if source == "radio" else "an EVP recorder", language
|
||||
),
|
||||
options={"num_predict": 16, "temperature": 0.95},
|
||||
)
|
||||
|
||||
try:
|
||||
text = await self._queue.submit(call)
|
||||
self._touch()
|
||||
cleaned = " ".join(text.strip().split())[:80]
|
||||
return cleaned or random.choice(FALLBACK_FRAGMENTS)
|
||||
except QueueFullError:
|
||||
raise SpiritBusyError()
|
||||
except (httpx.HTTPError, KeyError, ValueError):
|
||||
return random.choice(FALLBACK_FRAGMENTS)
|
||||
|
||||
async def wire_whisper(self, telemetry: dict, language: str = "en") -> str:
|
||||
"""One ambient line from the Wire Ghost about current telemetry."""
|
||||
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_fast_model,
|
||||
prompts.wire_prompt(telemetry),
|
||||
system=prompts.wire_system(language),
|
||||
options={"num_predict": 40, "temperature": 1.0},
|
||||
)
|
||||
|
||||
try:
|
||||
text = await self._queue.submit(call)
|
||||
self._touch()
|
||||
cleaned = " ".join(text.strip().split())[:140]
|
||||
return cleaned or random.choice(FALLBACK_WIRE_WHISPERS)
|
||||
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
|
||||
return random.choice(FALLBACK_WIRE_WHISPERS)
|
||||
|
||||
async def chat_stream(
|
||||
self,
|
||||
entity: dict,
|
||||
question: str,
|
||||
history: list[dict],
|
||||
language: str = "en",
|
||||
) -> AsyncIterator[str]:
|
||||
"""Streams the spirit's reply token by token. Falls back to a curated
|
||||
line when the box is unreachable so a séance never dies on screen."""
|
||||
try:
|
||||
stream = self._queue.submit_stream(
|
||||
lambda: self._client.generate_stream(
|
||||
settings.ollama_chat_model,
|
||||
prompts.chat_prompt(question, history),
|
||||
system=prompts.chat_system(entity, language),
|
||||
options={"num_predict": 140, "temperature": 0.85},
|
||||
)
|
||||
)
|
||||
self._touch()
|
||||
async for token in stream:
|
||||
yield token
|
||||
except QueueFullError:
|
||||
raise SpiritBusyError()
|
||||
except (httpx.HTTPError, KeyError, ValueError):
|
||||
yield random.choice(FALLBACK_REPLIES)
|
||||
|
||||
async def mint_profile(
|
||||
self,
|
||||
signature: str,
|
||||
channel: str,
|
||||
anomalies: list[dict],
|
||||
language: str = "en",
|
||||
) -> dict:
|
||||
"""Invent a full persona for a new signature, normalized to schema."""
|
||||
summary = json.dumps(anomalies[-10:])[:600]
|
||||
voice_ids = ES_VOICE_IDS if language == "es" else EN_VOICE_IDS
|
||||
|
||||
async def call() -> str:
|
||||
return await self._client.generate(
|
||||
settings.ollama_chat_model,
|
||||
prompts.mint_prompt(signature, channel, summary, voice_ids),
|
||||
system=prompts.MINT_SYSTEM,
|
||||
options={"num_predict": 400, "temperature": 0.9},
|
||||
)
|
||||
|
||||
try:
|
||||
raw = await self._queue.submit(call)
|
||||
self._touch()
|
||||
profile = entities.parse_mint_response(raw)
|
||||
if profile is None:
|
||||
return entities.fallback_profile(signature)
|
||||
return entities.normalize_profile(profile, signature)
|
||||
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
|
||||
return entities.fallback_profile(signature)
|
||||
|
||||
|
||||
# The app-wide instance; tests monkeypatch this.
|
||||
spirit_service = SpiritService()
|
||||
Reference in New Issue
Block a user