feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
This commit is contained in:
0
backend/app/tts/__init__.py
Normal file
0
backend/app/tts/__init__.py
Normal file
90
backend/app/tts/effects.py
Normal file
90
backend/app/tts/effects.py
Normal file
@@ -0,0 +1,90 @@
|
||||
"""Spirit-box effects chain: degrades clean Piper output into the classic
|
||||
static-choked vocal texture. Pure numpy signal processing on mono 16-bit WAV."""
|
||||
|
||||
import wave
|
||||
from io import BytesIO
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
def _read_wav(wav_bytes: bytes) -> tuple[np.ndarray, wave._wave_params]:
|
||||
with wave.open(BytesIO(wav_bytes)) as wav_in:
|
||||
params = wav_in.getparams()
|
||||
frames = wav_in.readframes(wav_in.getnframes())
|
||||
return np.frombuffer(frames, dtype=np.int16).astype(np.float32), params
|
||||
|
||||
|
||||
def _write_wav(samples: np.ndarray, params: wave._wave_params) -> bytes:
|
||||
output = BytesIO()
|
||||
with wave.open(output, "wb") as wav_out:
|
||||
wav_out.setparams(params)
|
||||
wav_out.writeframes(np.clip(samples, -32768, 32767).astype(np.int16).tobytes())
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def _resample(samples: np.ndarray, factor: float) -> np.ndarray:
|
||||
"""Naive pitch/rate shift by linear-interpolation resampling."""
|
||||
if factor <= 0 or len(samples) == 0:
|
||||
return samples
|
||||
new_length = max(1, int(len(samples) / factor))
|
||||
old_x = np.arange(len(samples))
|
||||
new_x = np.linspace(0, len(samples) - 1, new_length)
|
||||
return np.interp(new_x, old_x, samples).astype(np.float32)
|
||||
|
||||
|
||||
def apply_effects(
|
||||
wav_bytes: bytes,
|
||||
*,
|
||||
noise_level: float = 0.02,
|
||||
pitch_semitones: float = 0.0,
|
||||
rate: float = 1.0,
|
||||
bitcrush_bits: int = 0,
|
||||
echo: float = 0.0,
|
||||
) -> bytes:
|
||||
"""Apply the full spirit-voice chain: rate → pitch → bitcrush → echo → static."""
|
||||
samples, params = _read_wav(wav_bytes)
|
||||
if len(samples) == 0:
|
||||
return wav_bytes
|
||||
|
||||
# Rate change (tempo) and pitch shift. Resampling by rate*pitch_factor
|
||||
# changes both duration and pitch; dividing by rate keeps duration change
|
||||
# governed by `rate` alone while pitch shifts by the semitone amount.
|
||||
pitch_factor = 2.0 ** (pitch_semitones / 12.0)
|
||||
if rate != 1.0:
|
||||
samples = _resample(samples, rate)
|
||||
if pitch_factor != 1.0:
|
||||
shifted = _resample(samples, pitch_factor)
|
||||
# Re-fit to the original (post-rate) length so pitch shift doesn't
|
||||
# also change duration.
|
||||
samples = _resample(shifted, len(shifted) / max(len(samples), 1))
|
||||
|
||||
if bitcrush_bits and 0 < bitcrush_bits < 16:
|
||||
levels = 2 ** (16 - bitcrush_bits)
|
||||
samples = np.round(samples / levels) * levels
|
||||
|
||||
if echo > 0:
|
||||
delay = int(params.framerate * 0.18)
|
||||
if len(samples) > delay:
|
||||
echoed = np.zeros_like(samples)
|
||||
echoed[delay:] = samples[:-delay] * echo
|
||||
samples = samples + echoed
|
||||
samples *= 32767.0 / max(np.max(np.abs(samples)), 1.0)
|
||||
|
||||
if noise_level > 0:
|
||||
noise = np.random.normal(0, noise_level * 32767, size=samples.shape)
|
||||
# Fade static in/out at the clip edges so it breathes like a real
|
||||
# spirit-box sweep instead of clicking.
|
||||
fade = np.ones(len(samples), dtype=np.float32)
|
||||
edge = min(len(samples) // 8, int(params.framerate * 0.05))
|
||||
if edge > 0:
|
||||
ramp = np.linspace(0.15, 1.0, edge)
|
||||
fade[:edge] = ramp
|
||||
fade[-edge:] = ramp[::-1]
|
||||
samples = samples + noise * fade
|
||||
|
||||
return _write_wav(samples, params)
|
||||
|
||||
|
||||
def apply_static_effect(wav_bytes: bytes, noise_level: float = 0.02) -> bytes:
|
||||
"""Adds white noise to a mono 16-bit PCM WAV, simulating spirit-box static."""
|
||||
return apply_effects(wav_bytes, noise_level=noise_level)
|
||||
67
backend/app/tts/piper.py
Normal file
67
backend/app/tts/piper.py
Normal file
@@ -0,0 +1,67 @@
|
||||
"""Async wrapper around the local Piper CLI (`python -m piper`).
|
||||
|
||||
Piper runs fully on CPU on this host; a small semaphore keeps concurrent
|
||||
syntheses from stomping each other's onnxruntime thread pools."""
|
||||
|
||||
import asyncio
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
from app.config import settings
|
||||
from app.tts.effects import apply_effects
|
||||
from app.tts.voices import Voice
|
||||
|
||||
_synth_semaphore = asyncio.Semaphore(2)
|
||||
|
||||
|
||||
class PiperTTS:
|
||||
"""Wraps the Piper CLI to synthesize speech locally, no cloud calls."""
|
||||
|
||||
def __init__(self, voice_model_path: str):
|
||||
self._voice_model_path = voice_model_path
|
||||
|
||||
async def synthesize(self, text: str) -> bytes:
|
||||
"""Synthesize text to raw WAV bytes via the piper CLI."""
|
||||
async with _synth_semaphore:
|
||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
||||
out_path = tmp.name
|
||||
try:
|
||||
process = await asyncio.create_subprocess_exec(
|
||||
sys.executable,
|
||||
"-m",
|
||||
"piper",
|
||||
"--model",
|
||||
self._voice_model_path,
|
||||
"--output_file",
|
||||
out_path,
|
||||
stdin=asyncio.subprocess.PIPE,
|
||||
stdout=asyncio.subprocess.DEVNULL,
|
||||
stderr=asyncio.subprocess.DEVNULL,
|
||||
)
|
||||
await asyncio.wait_for(
|
||||
process.communicate(text.encode("utf-8")), timeout=60.0
|
||||
)
|
||||
if process.returncode != 0:
|
||||
raise RuntimeError(f"piper exited with {process.returncode}")
|
||||
return Path(out_path).read_bytes()
|
||||
finally:
|
||||
Path(out_path).unlink(missing_ok=True)
|
||||
|
||||
|
||||
async def synthesize_spirit_voice(
|
||||
text: str, voice: Voice, voice_profile: dict | None = None
|
||||
) -> bytes:
|
||||
"""One call: Piper synth + the spirit's signature effects chain."""
|
||||
profile = voice_profile or {}
|
||||
model_path = Path(settings.piper_voices_dir) / voice.model_file
|
||||
wav = await PiperTTS(str(model_path)).synthesize(text)
|
||||
return await asyncio.to_thread(
|
||||
apply_effects,
|
||||
wav,
|
||||
noise_level=float(profile.get("noise", 0.03)),
|
||||
pitch_semitones=float(profile.get("pitch", 0.0)),
|
||||
rate=float(profile.get("rate", 1.0)),
|
||||
bitcrush_bits=int(profile.get("bitcrush", 0)),
|
||||
echo=float(profile.get("echo", 0.2)),
|
||||
)
|
||||
40
backend/app/tts/voices.py
Normal file
40
backend/app/tts/voices.py
Normal file
@@ -0,0 +1,40 @@
|
||||
"""Catalog of locally installed Piper voices.
|
||||
|
||||
Each entity's voice_profile pins one of these voice ids plus pitch/rate/noise
|
||||
shaping, giving every spirit its own recognizable throat."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Voice:
|
||||
id: str
|
||||
model_file: str
|
||||
language: str
|
||||
description: str
|
||||
|
||||
|
||||
VOICES: dict[str, Voice] = {
|
||||
"lessac": Voice("lessac", "en_US-lessac-low.onnx", "en", "a measured American woman"),
|
||||
"amy": Voice("amy", "en_US-amy-low.onnx", "en", "a soft American woman"),
|
||||
"ryan": Voice("ryan", "en_US-ryan-low.onnx", "en", "a deep American man"),
|
||||
"alan": Voice("alan", "en_GB-alan-low.onnx", "en", "a low British man"),
|
||||
"hfc_male": Voice("hfc_male", "en_US-hfc_male-medium.onnx", "en", "a worn male voice"),
|
||||
"hfc_female": Voice(
|
||||
"hfc_female", "en_US-hfc_female-medium.onnx", "en", "a worn female voice"
|
||||
),
|
||||
"davefx": Voice("davefx", "es_ES-davefx-medium.onnx", "es", "una voz masculina grave"),
|
||||
"ald": Voice("ald", "es_MX-ald-medium.onnx", "es", "una voz masculina seca"),
|
||||
}
|
||||
|
||||
EN_VOICE_IDS = [v.id for v in VOICES.values() if v.language == "en"]
|
||||
ES_VOICE_IDS = [v.id for v in VOICES.values() if v.language == "es"]
|
||||
|
||||
|
||||
def pick_voice(voice_id: str | None, language: str) -> Voice:
|
||||
"""Resolve an entity's voice id, falling back to a language-appropriate default."""
|
||||
if voice_id and voice_id in VOICES:
|
||||
voice = VOICES[voice_id]
|
||||
if voice.language == language:
|
||||
return voice
|
||||
return VOICES["davefx" if language == "es" else "lessac"]
|
||||
Reference in New Issue
Block a user