Stream realtime TTS audio as it is synthesized

This commit is contained in:
Mikei386
2026-09-16 14:55:12 +02:00
parent 97c8072b3d
commit 3f684b7325
3 changed files with 60 additions and 31 deletions
+10 -3
View File
@@ -116,12 +116,19 @@ class VoiceSmokeTest(unittest.IsolatedAsyncioTestCase):
async def tts(request):
body = await request.json()
self.assertEqual(body["input"], "Hallo zurück")
return web.Response(body=make_wav((6000).to_bytes(2, "little", signed=True) * SAMPLE_RATE),
content_type="audio/wav")
self.assertEqual(body["chunk_size"], 4)
stream = web.StreamResponse(headers={"Content-Type": "application/octet-stream"})
await stream.prepare(request)
pcm = (6000).to_bytes(2, "little", signed=True) * SAMPLE_RATE
await stream.write(pcm[:SAMPLE_RATE])
await asyncio.wait_for(audio_received.wait(), timeout=3)
await stream.write(pcm[SAMPLE_RATE:])
await stream.write_eof()
return stream
fake = web.Application()
fake.router.add_post("/audio/transcriptions", stt)
fake.router.add_post("/audio/speech", tts)
fake.router.add_post("/audio/speech/pcm-stream", tts)
fake_runner, fake_url = await serve(fake)
os.environ.update({
"ATHENA_TALK_REALTIME_SECRET": "test-secret-" * 4,