Files
qtalker---/backend/app/tts/effects.py
Indiana b9110f45de feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
  passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
  fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
  effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
  with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
2026-07-20 20:13:28 +00:00

91 lines
3.4 KiB
Python

"""Spirit-box effects chain: degrades clean Piper output into the classic
static-choked vocal texture. Pure numpy signal processing on mono 16-bit WAV."""
import wave
from io import BytesIO
import numpy as np
def _read_wav(wav_bytes: bytes) -> tuple[np.ndarray, wave._wave_params]:
with wave.open(BytesIO(wav_bytes)) as wav_in:
params = wav_in.getparams()
frames = wav_in.readframes(wav_in.getnframes())
return np.frombuffer(frames, dtype=np.int16).astype(np.float32), params
def _write_wav(samples: np.ndarray, params: wave._wave_params) -> bytes:
output = BytesIO()
with wave.open(output, "wb") as wav_out:
wav_out.setparams(params)
wav_out.writeframes(np.clip(samples, -32768, 32767).astype(np.int16).tobytes())
return output.getvalue()
def _resample(samples: np.ndarray, factor: float) -> np.ndarray:
"""Naive pitch/rate shift by linear-interpolation resampling."""
if factor <= 0 or len(samples) == 0:
return samples
new_length = max(1, int(len(samples) / factor))
old_x = np.arange(len(samples))
new_x = np.linspace(0, len(samples) - 1, new_length)
return np.interp(new_x, old_x, samples).astype(np.float32)
def apply_effects(
wav_bytes: bytes,
*,
noise_level: float = 0.02,
pitch_semitones: float = 0.0,
rate: float = 1.0,
bitcrush_bits: int = 0,
echo: float = 0.0,
) -> bytes:
"""Apply the full spirit-voice chain: rate → pitch → bitcrush → echo → static."""
samples, params = _read_wav(wav_bytes)
if len(samples) == 0:
return wav_bytes
# Rate change (tempo) and pitch shift. Resampling by rate*pitch_factor
# changes both duration and pitch; dividing by rate keeps duration change
# governed by `rate` alone while pitch shifts by the semitone amount.
pitch_factor = 2.0 ** (pitch_semitones / 12.0)
if rate != 1.0:
samples = _resample(samples, rate)
if pitch_factor != 1.0:
shifted = _resample(samples, pitch_factor)
# Re-fit to the original (post-rate) length so pitch shift doesn't
# also change duration.
samples = _resample(shifted, len(shifted) / max(len(samples), 1))
if bitcrush_bits and 0 < bitcrush_bits < 16:
levels = 2 ** (16 - bitcrush_bits)
samples = np.round(samples / levels) * levels
if echo > 0:
delay = int(params.framerate * 0.18)
if len(samples) > delay:
echoed = np.zeros_like(samples)
echoed[delay:] = samples[:-delay] * echo
samples = samples + echoed
samples *= 32767.0 / max(np.max(np.abs(samples)), 1.0)
if noise_level > 0:
noise = np.random.normal(0, noise_level * 32767, size=samples.shape)
# Fade static in/out at the clip edges so it breathes like a real
# spirit-box sweep instead of clicking.
fade = np.ones(len(samples), dtype=np.float32)
edge = min(len(samples) // 8, int(params.framerate * 0.05))
if edge > 0:
ramp = np.linspace(0.15, 1.0, edge)
fade[:edge] = ramp
fade[-edge:] = ramp[::-1]
samples = samples + noise * fade
return _write_wav(samples, params)
def apply_static_effect(wav_bytes: bytes, noise_level: float = 0.02) -> bytes:
"""Adds white noise to a mono 16-bit PCM WAV, simulating spirit-box static."""
return apply_effects(wav_bytes, noise_level=noise_level)