Files
astraea/tts.py

34 lines
1.5 KiB
Python

# -*- coding: utf-8 -*-
"""Astraea TTS — realistic neural voices via edge-tts (Microsoft neural, sounds human).
Structured so a local GPU TTS (Kokoro/XTTS on the Windows box) can be swapped in later.
"""
import os
import subprocess
import sys
import tempfile
# Resolve edge-tts relative to the venv that runs this app (robust vs hardcoded paths).
EDGE_TTS_BIN = os.path.join(os.path.dirname(sys.executable), "edge-tts")
VOICES = {
"aria": {"name": "en-US-AriaNeural", "label": "Aria — confident (F)"},
"jenny": {"name": "en-US-JennyNeural", "label": "Jenny — warm (F)"},
"ana": {"name": "en-US-AnaNeural", "label": "Ana — calm (F)"},
"guy": {"name": "en-US-GuyNeural", "label": "Guy — professional (M)"},
"christopher": {"name": "en-US-ChristopherNeural", "label": "Christopher — deep (M)"},
"eric": {"name": "en-US-EricNeural", "label": "Eric — measured (M)"},
}
def synthesize(text, voice="aria", rate="-5%"):
v = VOICES.get(voice, VOICES["aria"])["name"]
text = text.replace("\n", " ").replace(" ", " ")[:4000]
out = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False).name
subprocess.run([EDGE_TTS_BIN, "--voice", v, f"--rate={rate}", "--text", text,
"--write-media", out], capture_output=True, timeout=60, check=True)
return out
def list_voices():
return [{"id": k, "label": v["label"]} for k, v in VOICES.items()]