feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry

- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
  passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
  fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
  effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
  with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
This commit is contained in:
Indiana
2026-07-20 20:13:28 +00:00
parent 52afad1ad5
commit b9110f45de
28 changed files with 1997 additions and 4 deletions

View File

View File

@@ -0,0 +1,90 @@
"""Spirit-box effects chain: degrades clean Piper output into the classic
static-choked vocal texture. Pure numpy signal processing on mono 16-bit WAV."""
import wave
from io import BytesIO
import numpy as np
def _read_wav(wav_bytes: bytes) -> tuple[np.ndarray, wave._wave_params]:
with wave.open(BytesIO(wav_bytes)) as wav_in:
params = wav_in.getparams()
frames = wav_in.readframes(wav_in.getnframes())
return np.frombuffer(frames, dtype=np.int16).astype(np.float32), params
def _write_wav(samples: np.ndarray, params: wave._wave_params) -> bytes:
output = BytesIO()
with wave.open(output, "wb") as wav_out:
wav_out.setparams(params)
wav_out.writeframes(np.clip(samples, -32768, 32767).astype(np.int16).tobytes())
return output.getvalue()
def _resample(samples: np.ndarray, factor: float) -> np.ndarray:
"""Naive pitch/rate shift by linear-interpolation resampling."""
if factor <= 0 or len(samples) == 0:
return samples
new_length = max(1, int(len(samples) / factor))
old_x = np.arange(len(samples))
new_x = np.linspace(0, len(samples) - 1, new_length)
return np.interp(new_x, old_x, samples).astype(np.float32)
def apply_effects(
wav_bytes: bytes,
*,
noise_level: float = 0.02,
pitch_semitones: float = 0.0,
rate: float = 1.0,
bitcrush_bits: int = 0,
echo: float = 0.0,
) -> bytes:
"""Apply the full spirit-voice chain: rate → pitch → bitcrush → echo → static."""
samples, params = _read_wav(wav_bytes)
if len(samples) == 0:
return wav_bytes
# Rate change (tempo) and pitch shift. Resampling by rate*pitch_factor
# changes both duration and pitch; dividing by rate keeps duration change
# governed by `rate` alone while pitch shifts by the semitone amount.
pitch_factor = 2.0 ** (pitch_semitones / 12.0)
if rate != 1.0:
samples = _resample(samples, rate)
if pitch_factor != 1.0:
shifted = _resample(samples, pitch_factor)
# Re-fit to the original (post-rate) length so pitch shift doesn't
# also change duration.
samples = _resample(shifted, len(shifted) / max(len(samples), 1))
if bitcrush_bits and 0 < bitcrush_bits < 16:
levels = 2 ** (16 - bitcrush_bits)
samples = np.round(samples / levels) * levels
if echo > 0:
delay = int(params.framerate * 0.18)
if len(samples) > delay:
echoed = np.zeros_like(samples)
echoed[delay:] = samples[:-delay] * echo
samples = samples + echoed
samples *= 32767.0 / max(np.max(np.abs(samples)), 1.0)
if noise_level > 0:
noise = np.random.normal(0, noise_level * 32767, size=samples.shape)
# Fade static in/out at the clip edges so it breathes like a real
# spirit-box sweep instead of clicking.
fade = np.ones(len(samples), dtype=np.float32)
edge = min(len(samples) // 8, int(params.framerate * 0.05))
if edge > 0:
ramp = np.linspace(0.15, 1.0, edge)
fade[:edge] = ramp
fade[-edge:] = ramp[::-1]
samples = samples + noise * fade
return _write_wav(samples, params)
def apply_static_effect(wav_bytes: bytes, noise_level: float = 0.02) -> bytes:
"""Adds white noise to a mono 16-bit PCM WAV, simulating spirit-box static."""
return apply_effects(wav_bytes, noise_level=noise_level)

67
backend/app/tts/piper.py Normal file
View File

@@ -0,0 +1,67 @@
"""Async wrapper around the local Piper CLI (`python -m piper`).
Piper runs fully on CPU on this host; a small semaphore keeps concurrent
syntheses from stomping each other's onnxruntime thread pools."""
import asyncio
import sys
import tempfile
from pathlib import Path
from app.config import settings
from app.tts.effects import apply_effects
from app.tts.voices import Voice
_synth_semaphore = asyncio.Semaphore(2)
class PiperTTS:
"""Wraps the Piper CLI to synthesize speech locally, no cloud calls."""
def __init__(self, voice_model_path: str):
self._voice_model_path = voice_model_path
async def synthesize(self, text: str) -> bytes:
"""Synthesize text to raw WAV bytes via the piper CLI."""
async with _synth_semaphore:
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
out_path = tmp.name
try:
process = await asyncio.create_subprocess_exec(
sys.executable,
"-m",
"piper",
"--model",
self._voice_model_path,
"--output_file",
out_path,
stdin=asyncio.subprocess.PIPE,
stdout=asyncio.subprocess.DEVNULL,
stderr=asyncio.subprocess.DEVNULL,
)
await asyncio.wait_for(
process.communicate(text.encode("utf-8")), timeout=60.0
)
if process.returncode != 0:
raise RuntimeError(f"piper exited with {process.returncode}")
return Path(out_path).read_bytes()
finally:
Path(out_path).unlink(missing_ok=True)
async def synthesize_spirit_voice(
text: str, voice: Voice, voice_profile: dict | None = None
) -> bytes:
"""One call: Piper synth + the spirit's signature effects chain."""
profile = voice_profile or {}
model_path = Path(settings.piper_voices_dir) / voice.model_file
wav = await PiperTTS(str(model_path)).synthesize(text)
return await asyncio.to_thread(
apply_effects,
wav,
noise_level=float(profile.get("noise", 0.03)),
pitch_semitones=float(profile.get("pitch", 0.0)),
rate=float(profile.get("rate", 1.0)),
bitcrush_bits=int(profile.get("bitcrush", 0)),
echo=float(profile.get("echo", 0.2)),
)

40
backend/app/tts/voices.py Normal file
View File

@@ -0,0 +1,40 @@
"""Catalog of locally installed Piper voices.
Each entity's voice_profile pins one of these voice ids plus pitch/rate/noise
shaping, giving every spirit its own recognizable throat."""
from dataclasses import dataclass
@dataclass(frozen=True)
class Voice:
id: str
model_file: str
language: str
description: str
VOICES: dict[str, Voice] = {
"lessac": Voice("lessac", "en_US-lessac-low.onnx", "en", "a measured American woman"),
"amy": Voice("amy", "en_US-amy-low.onnx", "en", "a soft American woman"),
"ryan": Voice("ryan", "en_US-ryan-low.onnx", "en", "a deep American man"),
"alan": Voice("alan", "en_GB-alan-low.onnx", "en", "a low British man"),
"hfc_male": Voice("hfc_male", "en_US-hfc_male-medium.onnx", "en", "a worn male voice"),
"hfc_female": Voice(
"hfc_female", "en_US-hfc_female-medium.onnx", "en", "a worn female voice"
),
"davefx": Voice("davefx", "es_ES-davefx-medium.onnx", "es", "una voz masculina grave"),
"ald": Voice("ald", "es_MX-ald-medium.onnx", "es", "una voz masculina seca"),
}
EN_VOICE_IDS = [v.id for v in VOICES.values() if v.language == "en"]
ES_VOICE_IDS = [v.id for v in VOICES.values() if v.language == "es"]
def pick_voice(voice_id: str | None, language: str) -> Voice:
"""Resolve an entity's voice id, falling back to a language-appropriate default."""
if voice_id and voice_id in VOICES:
voice = VOICES[voice_id]
if voice.language == language:
return voice
return VOICES["davefx" if language == "es" else "lessac"]