"""Async wrapper around the local Piper CLI (`python -m piper`). Piper runs fully on CPU on this host; a small semaphore keeps concurrent syntheses from stomping each other's onnxruntime thread pools.""" import asyncio import sys import tempfile from pathlib import Path from app.config import settings from app.tts.effects import apply_effects from app.tts.voices import Voice _synth_semaphore = asyncio.Semaphore(2) class PiperTTS: """Wraps the Piper CLI to synthesize speech locally, no cloud calls.""" def __init__(self, voice_model_path: str): self._voice_model_path = voice_model_path async def synthesize(self, text: str) -> bytes: """Synthesize text to raw WAV bytes via the piper CLI.""" async with _synth_semaphore: with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp: out_path = tmp.name try: process = await asyncio.create_subprocess_exec( sys.executable, "-m", "piper", "--model", self._voice_model_path, "--output_file", out_path, stdin=asyncio.subprocess.PIPE, stdout=asyncio.subprocess.DEVNULL, stderr=asyncio.subprocess.DEVNULL, ) await asyncio.wait_for( process.communicate(text.encode("utf-8")), timeout=60.0 ) if process.returncode != 0: raise RuntimeError(f"piper exited with {process.returncode}") return Path(out_path).read_bytes() finally: Path(out_path).unlink(missing_ok=True) async def synthesize_spirit_voice( text: str, voice: Voice, voice_profile: dict | None = None, instability: float = 0.0, ) -> bytes: """One call: Piper synth + the spirit's signature effects chain. `instability` (0.0-1.0, i.e. `1 - stability`) scales up the noise and bitcrush fed into the effects chain when the spirit is fighting to hold the channel — a shaky possession sounds shakier. It never touches `apply_effects` itself, only the values handed to it here. """ profile = voice_profile or {} instability = max(0.0, min(1.0, instability)) model_path = Path(settings.piper_voices_dir) / voice.model_file base_noise = float(profile.get("noise", 0.03)) # Noise ramps linearly across the whole range so degradation is audible # from the first sign of instability, but doubling the base level at # instability=1.0 (rather than, say, 5x) keeps the words intelligible # even at a full-static possession — this is texture, not white noise. noise_level = base_noise + instability * base_noise base_bitcrush = int(profile.get("bitcrush", 0)) # Bitcrush only kicks in once instability crosses ~0.5 -- a slightly # unstable channel just sounds noisier, not crunchy/robotic. Above that # threshold it ramps up to +6 bits of crush on top of whatever the # voice's own profile already applies, capped at 15 (apply_effects only # crushes when 0 < bits < 16). if instability > 0.5: bitcrush_bonus = round((instability - 0.5) / 0.5 * 6) else: bitcrush_bonus = 0 bitcrush_bits = min(15, base_bitcrush + bitcrush_bonus) wav = await PiperTTS(str(model_path)).synthesize(text) return await asyncio.to_thread( apply_effects, wav, noise_level=noise_level, pitch_semitones=float(profile.get("pitch", 0.0)), rate=float(profile.get("rate", 1.0)), bitcrush_bits=bitcrush_bits, echo=float(profile.get("echo", 0.2)), )