#!/usr/bin/env python3 """GPU STT service for The Collector — faster-whisper small.en on CUDA (4080S). POST /transcribe (raw PCM16 mono 16 kHz body) -> {"text": "..."} GET /health """ import numpy as np from fastapi import FastAPI, Request from faster_whisper import WhisperModel import uvicorn MODEL = "small.en" app = FastAPI() model = None @app.on_event("startup") async def load(): global model model = WhisperModel(MODEL, device="cuda", compute_type="float16") print(f"STT ready: {MODEL} on cuda") @app.get("/health") async def health(): return {"ok": model is not None, "model": MODEL, "device": "cuda"} @app.post("/transcribe") async def transcribe(request: Request): body = await request.body() audio = np.frombuffer(body, dtype=np.int16).astype(np.float32) / 32768.0 segs, _ = model.transcribe( audio, beam_size=1, language="en", vad_filter=True, condition_on_previous_text=False, no_speech_threshold=0.6) text = " ".join(s.text.strip() for s in segs).strip() return {"text": text} if __name__ == "__main__": uvicorn.run(app, host="0.0.0.0", port=8770)