Files
qtalker---/backend/app/llm/service.py
Indiana 16c8bae00d fix: the test suite was destroying the production database
The worst bug of the session. tests/conftest.py built its engine from
settings.database_url — the live database — and an autouse fixture calls
drop_all() before EVERY test. So every backend run silently annihilated the
real install: accounts, discovered spirits, Ghost Logs, devices, all of it.
Found it because /sitemap.xml listed zero entities minutes after I had
watched live séances mint real ones.

Tests now use TEST_DATABASE_URL, or `<configured-db>_test` derived from it,
and refuse to start at all if that ever resolves back to the production URL
— this box both serves the app and holds the repo, so "don't run tests in
prod" is not a workable guard.

Proven: inserted a canary row into production, ran 50 tests, canary
survived. Before this it would have been dropped.

Also in this commit:

SEO (routes/seo.py, lib/pageMeta.ts)
- Live /sitemap.xml generated from real entity rows, and /robots.txt, both
  registered BEFORE the SPA catch-all or they'd be served index.html.
  Crawlers are disallowed from /seance specifically because the open door
  provisions a guest on arrival — a crawler would fill the users table with
  wanderers who never existed.
- Per-route <title>, description, canonical and JSON-LD. The Codex is the
  indexable asset here (every spirit is unique long-form prose) and all of
  it previously shared one static title, so entities competed with each
  other instead of ranking. Entities are marked up as fictional Persons so
  a rich result can never imply a record of a real dead human.
- public_base_url setting: absolute URLs for crawlers can't be derived from
  the request, since behind the tunnel the app only sees an internal host.

Camera channel, first half (lib/camera.ts, llm scry path)
- OllamaClient.generate() now accepts `images`; the configured chat model
  (minicpm-v4.5:8b) is vision-capable, so the entity can speak about what
  the seeker's camera actually shows. Verified against a synthetic room
  image: it named the pale column and the small red cube, then misread them
  as oak in a farmhouse parlor — real perception, in character.
- Frames are captured only on an explicit act, downscaled to 768px and
  JPEG-compressed, never stored, and the prompt forbids describing faces or
  guessing identity. CameraEye carries the same generation guard as the EVP
  listener so closing during the permission prompt can't leave the camera
  live after teardown.

338 backend tests pass; 375 frontend; i18n parity holds.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-07-30 01:44:01 +00:00

274 lines
10 KiB
Python

"""SpiritService: the single gateway every spirit-mode LLM call flows through.
All calls are serialized through the bounded LLMQueue (the Ollama box is
CPU-only and shared). Every method has a curated offline fallback so the veil
never visibly tears — if the box is dark, the spirits still whisper."""
import json
import random
import time
from collections.abc import AsyncIterator
import httpx
from app import entities
from app.config import settings
from app.entropy import veil_seed
from app.llm import prompts
from app.llm.client import OllamaClient
from app.llm.queue import LLMQueue, QueueFullError
from app.tts.voices import EN_VOICE_IDS, ES_VOICE_IDS
FALLBACK_FRAGMENTS = [
"listen", "below", "stay", "cold", "again", "not alone", "behind you",
"the rain", "wait", "closer", "remember", "still here", "don't go",
"the door", "hush", "almost", "forgive", "the water", "home",
]
FALLBACK_WIRE_WHISPERS = [
"something moves through me that is not yours",
"the pulses quicken when you watch",
"i am the hum between your packets",
"traffic thickens. the others are waking",
"your presence is a warmth in the wire",
"i count your heartbeats in round trips",
]
FALLBACK_REPLIES = [
"The veil is thick tonight. Ask again when the static settles.",
"I heard you. The answer is still forming in the noise.",
"Patience, seeker. Even the dead must gather themselves.",
]
class SpiritBusyError(Exception):
"""The LLM queue is full — too many seekers at once."""
class SpiritService:
def __init__(self, client: OllamaClient | None = None, queue: LLMQueue | None = None):
self._client = client or OllamaClient()
self._queue = queue or LLMQueue(
max_concurrency=settings.llm_max_concurrency,
max_queue_depth=settings.llm_max_queue_depth,
)
self._last_call_at = 0.0
def _touch(self) -> None:
self._last_call_at = time.monotonic()
def ambient_ready(self) -> bool:
"""Ambient whispers yield the box to anything a user asked for."""
return time.monotonic() - self._last_call_at >= settings.llm_cooldown_seconds
async def fragment(self, source: str, anomaly: dict, language: str = "en") -> str:
"""One Ovilus-style word/fragment for an anomaly event."""
async def call() -> str:
return await self._client.generate(
settings.ollama_fast_model,
prompts.fragment_prompt(source, anomaly, language),
system=prompts.fragment_system(
{
"radio": "a spirit box",
"evp": "an EVP recorder",
"emf": "an EMF field meter",
}.get(source, "the veil"),
language,
),
options={"num_predict": 16, "temperature": 0.95},
)
try:
text = await self._queue.submit(call)
self._touch()
cleaned = " ".join(text.strip().split())[:80]
return cleaned or random.choice(FALLBACK_FRAGMENTS)
except QueueFullError:
raise SpiritBusyError()
except (httpx.HTTPError, KeyError, ValueError):
return random.choice(FALLBACK_FRAGMENTS)
async def wire_whisper(self, telemetry: dict, language: str = "en") -> str:
"""One ambient line from the Wire Ghost about current telemetry."""
async def call() -> str:
return await self._client.generate(
settings.ollama_fast_model,
prompts.wire_prompt(telemetry),
system=prompts.wire_system(language),
options={"num_predict": 40, "temperature": 1.0},
)
try:
text = await self._queue.submit(call)
self._touch()
cleaned = " ".join(text.strip().split())[:140]
return cleaned or random.choice(FALLBACK_WIRE_WHISPERS)
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
return random.choice(FALLBACK_WIRE_WHISPERS)
async def chat_stream(
self,
entity: dict,
question: str,
history: list[dict],
language: str = "en",
) -> AsyncIterator[str]:
"""Streams the spirit's reply token by token. Falls back to a curated
line when the box is unreachable so a séance never dies on screen."""
try:
stream = self._queue.submit_stream(
lambda: self._client.generate_stream(
settings.ollama_chat_model,
prompts.chat_prompt(question, history),
system=prompts.chat_system(entity, language),
options={"num_predict": 140, "temperature": 0.85},
)
)
self._touch()
async for token in stream:
yield token
except QueueFullError:
raise SpiritBusyError()
except (httpx.HTTPError, KeyError, ValueError):
yield random.choice(FALLBACK_REPLIES)
async def manifest(
self,
entity: dict,
readings: dict,
language: str = "en",
entropy: object = None,
) -> str:
"""Unprompted speech — the entity says something nobody asked for.
This is deliberately NOT chat_stream with an empty question. Two
things make it generation *from* something rather than a reply
*to* something:
1. The prompt contains no seeker input at all. The model's only
stimulus is the measured state of the room (see
prompts.manifest_prompt), so there is nothing to answer.
2. The sampling seed is derived from physical entropy harvested in
that room — the microphone's noise floor, RF noise, magnetometer
jitter. Ollama's `seed` option fixes the token-sampling path, so
seeding it from real physical noise means the room genuinely
selects the words. Not a metaphor: change the noise, get
different speech, and no two rooms produce the same utterance.
High temperature and top_k on purpose — a tight, "correct" decode
produces a well-behaved assistant sentence, which is exactly the
failure mode here. This should sound like something surfacing, not
something composed.
"""
# 63-bit: Ollama takes a signed 64-bit seed, and staying under the
# sign bit avoids any wraparound surprises across versions.
seed = int.from_bytes(veil_seed(entropy, "manifest")[:8], "big") % (2**63)
async def call() -> str:
return await self._client.generate(
settings.ollama_chat_model,
prompts.manifest_prompt(readings),
system=prompts.manifest_system(entity, language),
options={
"num_predict": 40,
"temperature": 1.15,
"top_k": 100,
"top_p": 0.98,
"repeat_penalty": 1.05,
"seed": seed,
},
)
raw = await self._queue.submit(call)
self._touch()
return raw.strip().strip('"')[:200]
async def scry(
self,
entity: dict,
image_b64: str,
language: str = "en",
entropy: object = None,
) -> str:
"""The entity speaks about what the seeker's camera actually shows.
The configured chat model (minicpm-v4.5) is vision-capable, so this
is a genuine look at the real room rather than an invented
description — the same principle as every other channel here: real
measurement first, interpretation second.
Seeded from physical entropy like manifest(), so two identical rooms
still produce different speech.
"""
seed = int.from_bytes(veil_seed(entropy, "scry")[:8], "big") % (2**63)
async def call() -> str:
return await self._client.generate(
settings.ollama_chat_model,
prompts.SCRY_PROMPT,
system=prompts.scry_system(entity, language),
options={
"num_predict": 90,
"temperature": 0.95,
"top_p": 0.95,
"seed": seed,
},
images=[image_b64],
)
raw = await self._queue.submit(call)
self._touch()
return raw.strip().strip('"')[:400]
async def mint_profile(
self,
signature: str,
channel: str,
anomalies: list[dict],
language: str = "en",
entropy: object = None,
sky: dict | None = None,
) -> dict:
"""Invent a full persona for a new signature, normalized to schema.
`entropy` is the seeker's physical-noise contribution (see
app/entropy.py); it drives the hidden trait roll and any cosmetic
defaults the LLM left unfilled, so two spirits minted on the same
channel are genuinely different rather than identical."""
summary = json.dumps(anomalies[-10:])[:600]
voice_ids = ES_VOICE_IDS if language == "es" else EN_VOICE_IDS
illumination = sky.get("moon_illumination") if sky else None
sky_text = (
f"{sky['moon_name']}, {sky['moon_illumination']:.0%} lit"
if sky
else "unknown"
)
async def call() -> str:
return await self._client.generate(
settings.ollama_chat_model,
prompts.mint_prompt(signature, channel, summary, voice_ids, sky_text),
system=prompts.MINT_SYSTEM,
options={"num_predict": 400, "temperature": 0.9},
)
try:
raw = await self._queue.submit(call)
self._touch()
profile = entities.parse_mint_response(raw)
if profile is None:
return entities.fallback_profile(signature, entropy, illumination)
normalized = entities.normalize_profile(profile, signature, entropy)
normalized["traits"] = entities.moon_trait_bias(
normalized["traits"], illumination
)
return normalized
except (QueueFullError, httpx.HTTPError, KeyError, ValueError):
return entities.fallback_profile(signature, entropy, illumination)
# The app-wide instance; tests monkeypatch this.
spirit_service = SpiritService()