#!/usr/bin/env python3 """End-to-end voice test: synthesize speech, push it through the /ws browser channel, assert we get back a transcript + reply audio (PCM16).""" import asyncio import json import sys sys.path.insert(0, "/opt/collector") from app import TTS, CFG import websockets tts = TTS(CFG["tts"]) tts.load() pcm = tts.synthesize_16k("Hello Collector, tell me who you are in one sentence.") FRAME = 640 # 20ms @ 16kHz async def main(): async with websockets.connect("ws://127.0.0.1:8766/ws") as ws: await asyncio.sleep(6) # let server warm (STT+TTS+LLM) # speech in 20ms frames for i in range(0, len(pcm), FRAME): await ws.send(pcm[i:i + FRAME]) await asyncio.sleep(0.02) # 1s silence -> trigger VAD turn end silence = b"\x00" * FRAME for _ in range(50): await ws.send(silence) await asyncio.sleep(0.02) got_text = got_audio = False deadline = asyncio.get_event_loop().time() + 60 while asyncio.get_event_loop().time() < deadline: try: msg = await asyncio.wait_for(ws.recv(), timeout=40) except Exception: break if isinstance(msg, str): print("TRANSCRIPT:", msg) got_text = True else: print("AUDIO bytes:", len(msg)) got_audio = True if got_text and got_audio: break print("VOICE_TEST", "PASS" if (got_text and got_audio) else "FAIL") asyncio.run(main())