From 435c59da418eab10642b1eb1b5423ecccb93bb7d Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Thu, 3 Sep 2026 23:36:51 +0200 Subject: [PATCH] Stream Athena Talk replies incrementally --- integrations/openclaw-athena-talk/README.md | 3 + .../openclaw-athena-talk/dist/index.js | 53 +++++++++++++--- integrations/openclaw-athena-talk/index.ts | 61 +++++++++++++------ .../openclaw-athena-talk/openclaw.plugin.json | 2 +- .../openclaw-athena-talk/package.json | 2 +- 5 files changed, 90 insertions(+), 31 deletions(-) diff --git a/integrations/openclaw-athena-talk/README.md b/integrations/openclaw-athena-talk/README.md index 79662d5..1c9c8eb 100644 --- a/integrations/openclaw-athena-talk/README.md +++ b/integrations/openclaw-athena-talk/README.md @@ -9,6 +9,9 @@ Athena speech stack: tools, 4. Athena XTTS/Piper returns PCM audio to the Talk client. +Long replies are synthesized incrementally. The first short phrase starts +playing as soon as it is ready while the next phrase is generated in parallel. + The provider intentionally uses half-duplex audio: microphone input is paused while a response is being transcribed, generated, synthesized, or played. This prevents speaker feedback from aborting TTS. Spoken interruption (barge-in) is diff --git a/integrations/openclaw-athena-talk/dist/index.js b/integrations/openclaw-athena-talk/dist/index.js index 5c527a0..a8cfa28 100644 --- a/integrations/openclaw-athena-talk/dist/index.js +++ b/integrations/openclaw-athena-talk/dist/index.js @@ -48,6 +48,28 @@ function pcmRms(pcm) { } return Math.sqrt(sum / count); } +function splitForIncrementalSpeech(text, limit = 80) { + const words = text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean); + const chunks = []; + let current = ""; + for (const word of words) { + const candidate = current ? `${current} ${word}` : word; + if (current && candidate.length > limit) { + chunks.push(current); + current = word; + } + else { + current = candidate; + } + if (current.length >= 35 && /[.!?](?:["')\]_*]+)?$/.test(word)) { + chunks.push(current); + current = ""; + } + } + if (current) + chunks.push(current); + return chunks; +} function readWavPcm24k(wav) { if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") { throw new Error("Athena TTS did not return PCM WAV audio"); @@ -226,7 +248,8 @@ class AthenaTalkBridge { const generation = ++this.generation; const responseId = `athena-${randomUUID()}`; try { - const result = await this.req.runAgentConsult({ prompt, signal: controller.signal }); + const voicePrompt = `${prompt}\n\nAntwortregeln für diese Sprachantwort: Antworte ausschließlich auf Deutsch, kurz und direkt. Keine Analyse, keine Meta-Kommentare, kein Markdown und keine Wiederholung der Anfrage. Gib nur den Text aus, der gesprochen werden soll.`; + const result = await this.req.runAgentConsult({ prompt: voicePrompt, signal: controller.signal }); if (generation !== this.generation || !this.isConnected()) return; const text = String(result?.text || "").trim(); @@ -234,20 +257,30 @@ class AthenaTalkBridge { return; this.req.onEvent?.({ direction: "server", type: "response.created", responseId }); this.req.onTranscript?.("assistant", text, true); - const pcm = await this.synthesize(text, controller.signal); - if (generation !== this.generation || !this.isConnected()) + const speechChunks = splitForIncrementalSpeech(text); + if (speechChunks.length === 0) return; const frameBytes = 24000 * 2 / 50; - const playbackStartedAt = Date.now(); - for (let offset = 0; offset < pcm.length; offset += frameBytes) { - if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) { + let playbackEndsAt = 0; + let nextAudio = this.synthesize(speechChunks[0], controller.signal); + for (let index = 0; index < speechChunks.length; index += 1) { + const pcm = await nextAudio; + if (generation !== this.generation || !this.isConnected()) return; + if (index + 1 < speechChunks.length) { + nextAudio = this.synthesize(speechChunks[index + 1], controller.signal); + } + const audioDurationMs = pcm.length / (24000 * 2) * 1000; + playbackEndsAt = Math.max(playbackEndsAt, Date.now()) + audioDurationMs; + for (let offset = 0; offset < pcm.length; offset += frameBytes) { + if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) { + return; + } + this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); + await new Promise((resolve) => setTimeout(resolve, 18)); } - this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); - await new Promise((resolve) => setTimeout(resolve, 18)); } - const audioDurationMs = pcm.length / (24000 * 2) * 1000; - const remainingPlaybackMs = Math.max(0, audioDurationMs - (Date.now() - playbackStartedAt)); + const remainingPlaybackMs = Math.max(0, playbackEndsAt - Date.now()); await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300)); this.req.onResponseDone?.({ status: "completed", responseId }); this.req.onEvent?.({ direction: "server", type: "response.done", responseId }); diff --git a/integrations/openclaw-athena-talk/index.ts b/integrations/openclaw-athena-talk/index.ts index 986e98a..0776a1c 100644 --- a/integrations/openclaw-athena-talk/index.ts +++ b/integrations/openclaw-athena-talk/index.ts @@ -64,6 +64,27 @@ function pcmRms(pcm: Buffer): number { return Math.sqrt(sum / count); } +function splitForIncrementalSpeech(text: string, limit = 80): string[] { + const words = text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean); + const chunks: string[] = []; + let current = ""; + for (const word of words) { + const candidate = current ? `${current} ${word}` : word; + if (current && candidate.length > limit) { + chunks.push(current); + current = word; + } else { + current = candidate; + } + if (current.length >= 35 && /[.!?](?:["')\]_*]+)?$/.test(word)) { + chunks.push(current); + current = ""; + } + } + if (current) chunks.push(current); + return chunks; +} + function readWavPcm24k(wav: Buffer): Buffer { if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") { throw new Error("Athena TTS did not return PCM WAV audio"); @@ -238,33 +259,35 @@ class AthenaTalkBridge { const generation = ++this.generation; const responseId = `athena-${randomUUID()}`; try { - const result = await this.req.runAgentConsult({ prompt, signal: controller.signal }); + const voicePrompt = `${prompt}\n\nAntwortregeln für diese Sprachantwort: Antworte ausschließlich auf Deutsch, kurz und direkt. Keine Analyse, keine Meta-Kommentare, kein Markdown und keine Wiederholung der Anfrage. Gib nur den Text aus, der gesprochen werden soll.`; + const result = await this.req.runAgentConsult({ prompt: voicePrompt, signal: controller.signal }); if (generation !== this.generation || !this.isConnected()) return; const text = String(result?.text || "").trim(); if (!text) return; this.req.onEvent?.({ direction: "server", type: "response.created", responseId }); this.req.onTranscript?.("assistant", text, true); - const pcm = await this.synthesize(text, controller.signal); - if (generation !== this.generation || !this.isConnected()) return; - // Feed OpenClaw in its native 20 ms frame size. Sending larger bursts - // made the relay split several frames at once; slow-client protection - // could then drop individual frames and produce audible holes. + const speechChunks = splitForIncrementalSpeech(text); + if (speechChunks.length === 0) return; const frameBytes = 24000 * 2 / 50; - const playbackStartedAt = Date.now(); - for (let offset = 0; offset < pcm.length; offset += frameBytes) { - if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) { - return; + let playbackEndsAt = 0; + let nextAudio = this.synthesize(speechChunks[0], controller.signal); + for (let index = 0; index < speechChunks.length; index += 1) { + const pcm = await nextAudio; + if (generation !== this.generation || !this.isConnected()) return; + if (index + 1 < speechChunks.length) { + nextAudio = this.synthesize(speechChunks[index + 1], controller.signal); + } + const audioDurationMs = pcm.length / (24000 * 2) * 1000; + playbackEndsAt = Math.max(playbackEndsAt, Date.now()) + audioDurationMs; + for (let offset = 0; offset < pcm.length; offset += frameBytes) { + if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) { + return; + } + this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); + await new Promise((resolve) => setTimeout(resolve, 18)); } - this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); - // Run just ahead of realtime so the client builds a small jitter - // buffer without flooding the WebSocket queue. - await new Promise((resolve) => setTimeout(resolve, 18)); } - // Transmission is intentionally a little faster than playback. Keep the - // microphone gated until the client has consumed that buffered tail; - // otherwise the last spoken words are transcribed again as user input. - const audioDurationMs = pcm.length / (24000 * 2) * 1000; - const remainingPlaybackMs = Math.max(0, audioDurationMs - (Date.now() - playbackStartedAt)); + const remainingPlaybackMs = Math.max(0, playbackEndsAt - Date.now()); await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300)); this.req.onResponseDone?.({ status: "completed", responseId }); this.req.onEvent?.({ direction: "server", type: "response.done", responseId }); diff --git a/integrations/openclaw-athena-talk/openclaw.plugin.json b/integrations/openclaw-athena-talk/openclaw.plugin.json index eb19c70..3fc8652 100644 --- a/integrations/openclaw-athena-talk/openclaw.plugin.json +++ b/integrations/openclaw-athena-talk/openclaw.plugin.json @@ -2,7 +2,7 @@ "id": "athena-talk", "name": "Athena Local Talk", "description": "Local German voice conversations through Athena without a public speech provider.", - "version": "0.1.3", + "version": "0.2.0", "enabledByDefault": true, "activation": { "onStartup": true, diff --git a/integrations/openclaw-athena-talk/package.json b/integrations/openclaw-athena-talk/package.json index b51d729..4f5bf3c 100644 --- a/integrations/openclaw-athena-talk/package.json +++ b/integrations/openclaw-athena-talk/package.json @@ -1,6 +1,6 @@ { "name": "openclaw-plugin-athena-talk", - "version": "0.1.3", + "version": "0.2.0", "description": "Private realtime Talk bridge for Athena STT, OpenClaw agent consult, and Athena TTS.", "type": "module", "private": true,