Stream Athena Talk replies incrementally
This commit is contained in:
@@ -9,6 +9,9 @@ Athena speech stack:
|
|||||||
tools,
|
tools,
|
||||||
4. Athena XTTS/Piper returns PCM audio to the Talk client.
|
4. Athena XTTS/Piper returns PCM audio to the Talk client.
|
||||||
|
|
||||||
|
Long replies are synthesized incrementally. The first short phrase starts
|
||||||
|
playing as soon as it is ready while the next phrase is generated in parallel.
|
||||||
|
|
||||||
The provider intentionally uses half-duplex audio: microphone input is paused
|
The provider intentionally uses half-duplex audio: microphone input is paused
|
||||||
while a response is being transcribed, generated, synthesized, or played. This
|
while a response is being transcribed, generated, synthesized, or played. This
|
||||||
prevents speaker feedback from aborting TTS. Spoken interruption (barge-in) is
|
prevents speaker feedback from aborting TTS. Spoken interruption (barge-in) is
|
||||||
|
|||||||
+43
-10
@@ -48,6 +48,28 @@ function pcmRms(pcm) {
|
|||||||
}
|
}
|
||||||
return Math.sqrt(sum / count);
|
return Math.sqrt(sum / count);
|
||||||
}
|
}
|
||||||
|
function splitForIncrementalSpeech(text, limit = 80) {
|
||||||
|
const words = text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean);
|
||||||
|
const chunks = [];
|
||||||
|
let current = "";
|
||||||
|
for (const word of words) {
|
||||||
|
const candidate = current ? `${current} ${word}` : word;
|
||||||
|
if (current && candidate.length > limit) {
|
||||||
|
chunks.push(current);
|
||||||
|
current = word;
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
current = candidate;
|
||||||
|
}
|
||||||
|
if (current.length >= 35 && /[.!?](?:["')\]_*]+)?$/.test(word)) {
|
||||||
|
chunks.push(current);
|
||||||
|
current = "";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (current)
|
||||||
|
chunks.push(current);
|
||||||
|
return chunks;
|
||||||
|
}
|
||||||
function readWavPcm24k(wav) {
|
function readWavPcm24k(wav) {
|
||||||
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
||||||
throw new Error("Athena TTS did not return PCM WAV audio");
|
throw new Error("Athena TTS did not return PCM WAV audio");
|
||||||
@@ -226,7 +248,8 @@ class AthenaTalkBridge {
|
|||||||
const generation = ++this.generation;
|
const generation = ++this.generation;
|
||||||
const responseId = `athena-${randomUUID()}`;
|
const responseId = `athena-${randomUUID()}`;
|
||||||
try {
|
try {
|
||||||
const result = await this.req.runAgentConsult({ prompt, signal: controller.signal });
|
const voicePrompt = `${prompt}\n\nAntwortregeln für diese Sprachantwort: Antworte ausschließlich auf Deutsch, kurz und direkt. Keine Analyse, keine Meta-Kommentare, kein Markdown und keine Wiederholung der Anfrage. Gib nur den Text aus, der gesprochen werden soll.`;
|
||||||
|
const result = await this.req.runAgentConsult({ prompt: voicePrompt, signal: controller.signal });
|
||||||
if (generation !== this.generation || !this.isConnected())
|
if (generation !== this.generation || !this.isConnected())
|
||||||
return;
|
return;
|
||||||
const text = String(result?.text || "").trim();
|
const text = String(result?.text || "").trim();
|
||||||
@@ -234,20 +257,30 @@ class AthenaTalkBridge {
|
|||||||
return;
|
return;
|
||||||
this.req.onEvent?.({ direction: "server", type: "response.created", responseId });
|
this.req.onEvent?.({ direction: "server", type: "response.created", responseId });
|
||||||
this.req.onTranscript?.("assistant", text, true);
|
this.req.onTranscript?.("assistant", text, true);
|
||||||
const pcm = await this.synthesize(text, controller.signal);
|
const speechChunks = splitForIncrementalSpeech(text);
|
||||||
if (generation !== this.generation || !this.isConnected())
|
if (speechChunks.length === 0)
|
||||||
return;
|
return;
|
||||||
const frameBytes = 24000 * 2 / 50;
|
const frameBytes = 24000 * 2 / 50;
|
||||||
const playbackStartedAt = Date.now();
|
let playbackEndsAt = 0;
|
||||||
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
let nextAudio = this.synthesize(speechChunks[0], controller.signal);
|
||||||
if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) {
|
for (let index = 0; index < speechChunks.length; index += 1) {
|
||||||
|
const pcm = await nextAudio;
|
||||||
|
if (generation !== this.generation || !this.isConnected())
|
||||||
return;
|
return;
|
||||||
|
if (index + 1 < speechChunks.length) {
|
||||||
|
nextAudio = this.synthesize(speechChunks[index + 1], controller.signal);
|
||||||
|
}
|
||||||
|
const audioDurationMs = pcm.length / (24000 * 2) * 1000;
|
||||||
|
playbackEndsAt = Math.max(playbackEndsAt, Date.now()) + audioDurationMs;
|
||||||
|
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
||||||
|
if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
||||||
|
await new Promise((resolve) => setTimeout(resolve, 18));
|
||||||
}
|
}
|
||||||
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
|
||||||
await new Promise((resolve) => setTimeout(resolve, 18));
|
|
||||||
}
|
}
|
||||||
const audioDurationMs = pcm.length / (24000 * 2) * 1000;
|
const remainingPlaybackMs = Math.max(0, playbackEndsAt - Date.now());
|
||||||
const remainingPlaybackMs = Math.max(0, audioDurationMs - (Date.now() - playbackStartedAt));
|
|
||||||
await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300));
|
await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300));
|
||||||
this.req.onResponseDone?.({ status: "completed", responseId });
|
this.req.onResponseDone?.({ status: "completed", responseId });
|
||||||
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
||||||
|
|||||||
@@ -64,6 +64,27 @@ function pcmRms(pcm: Buffer): number {
|
|||||||
return Math.sqrt(sum / count);
|
return Math.sqrt(sum / count);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function splitForIncrementalSpeech(text: string, limit = 80): string[] {
|
||||||
|
const words = text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean);
|
||||||
|
const chunks: string[] = [];
|
||||||
|
let current = "";
|
||||||
|
for (const word of words) {
|
||||||
|
const candidate = current ? `${current} ${word}` : word;
|
||||||
|
if (current && candidate.length > limit) {
|
||||||
|
chunks.push(current);
|
||||||
|
current = word;
|
||||||
|
} else {
|
||||||
|
current = candidate;
|
||||||
|
}
|
||||||
|
if (current.length >= 35 && /[.!?](?:["')\]_*]+)?$/.test(word)) {
|
||||||
|
chunks.push(current);
|
||||||
|
current = "";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (current) chunks.push(current);
|
||||||
|
return chunks;
|
||||||
|
}
|
||||||
|
|
||||||
function readWavPcm24k(wav: Buffer): Buffer {
|
function readWavPcm24k(wav: Buffer): Buffer {
|
||||||
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
||||||
throw new Error("Athena TTS did not return PCM WAV audio");
|
throw new Error("Athena TTS did not return PCM WAV audio");
|
||||||
@@ -238,33 +259,35 @@ class AthenaTalkBridge {
|
|||||||
const generation = ++this.generation;
|
const generation = ++this.generation;
|
||||||
const responseId = `athena-${randomUUID()}`;
|
const responseId = `athena-${randomUUID()}`;
|
||||||
try {
|
try {
|
||||||
const result = await this.req.runAgentConsult({ prompt, signal: controller.signal });
|
const voicePrompt = `${prompt}\n\nAntwortregeln für diese Sprachantwort: Antworte ausschließlich auf Deutsch, kurz und direkt. Keine Analyse, keine Meta-Kommentare, kein Markdown und keine Wiederholung der Anfrage. Gib nur den Text aus, der gesprochen werden soll.`;
|
||||||
|
const result = await this.req.runAgentConsult({ prompt: voicePrompt, signal: controller.signal });
|
||||||
if (generation !== this.generation || !this.isConnected()) return;
|
if (generation !== this.generation || !this.isConnected()) return;
|
||||||
const text = String(result?.text || "").trim();
|
const text = String(result?.text || "").trim();
|
||||||
if (!text) return;
|
if (!text) return;
|
||||||
this.req.onEvent?.({ direction: "server", type: "response.created", responseId });
|
this.req.onEvent?.({ direction: "server", type: "response.created", responseId });
|
||||||
this.req.onTranscript?.("assistant", text, true);
|
this.req.onTranscript?.("assistant", text, true);
|
||||||
const pcm = await this.synthesize(text, controller.signal);
|
const speechChunks = splitForIncrementalSpeech(text);
|
||||||
if (generation !== this.generation || !this.isConnected()) return;
|
if (speechChunks.length === 0) return;
|
||||||
// Feed OpenClaw in its native 20 ms frame size. Sending larger bursts
|
|
||||||
// made the relay split several frames at once; slow-client protection
|
|
||||||
// could then drop individual frames and produce audible holes.
|
|
||||||
const frameBytes = 24000 * 2 / 50;
|
const frameBytes = 24000 * 2 / 50;
|
||||||
const playbackStartedAt = Date.now();
|
let playbackEndsAt = 0;
|
||||||
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
let nextAudio = this.synthesize(speechChunks[0], controller.signal);
|
||||||
if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) {
|
for (let index = 0; index < speechChunks.length; index += 1) {
|
||||||
return;
|
const pcm = await nextAudio;
|
||||||
|
if (generation !== this.generation || !this.isConnected()) return;
|
||||||
|
if (index + 1 < speechChunks.length) {
|
||||||
|
nextAudio = this.synthesize(speechChunks[index + 1], controller.signal);
|
||||||
|
}
|
||||||
|
const audioDurationMs = pcm.length / (24000 * 2) * 1000;
|
||||||
|
playbackEndsAt = Math.max(playbackEndsAt, Date.now()) + audioDurationMs;
|
||||||
|
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
||||||
|
if (controller.signal.aborted || generation !== this.generation || !this.isConnected()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
||||||
|
await new Promise((resolve) => setTimeout(resolve, 18));
|
||||||
}
|
}
|
||||||
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
|
||||||
// Run just ahead of realtime so the client builds a small jitter
|
|
||||||
// buffer without flooding the WebSocket queue.
|
|
||||||
await new Promise((resolve) => setTimeout(resolve, 18));
|
|
||||||
}
|
}
|
||||||
// Transmission is intentionally a little faster than playback. Keep the
|
const remainingPlaybackMs = Math.max(0, playbackEndsAt - Date.now());
|
||||||
// microphone gated until the client has consumed that buffered tail;
|
|
||||||
// otherwise the last spoken words are transcribed again as user input.
|
|
||||||
const audioDurationMs = pcm.length / (24000 * 2) * 1000;
|
|
||||||
const remainingPlaybackMs = Math.max(0, audioDurationMs - (Date.now() - playbackStartedAt));
|
|
||||||
await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300));
|
await new Promise((resolve) => setTimeout(resolve, remainingPlaybackMs + 300));
|
||||||
this.req.onResponseDone?.({ status: "completed", responseId });
|
this.req.onResponseDone?.({ status: "completed", responseId });
|
||||||
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
"id": "athena-talk",
|
"id": "athena-talk",
|
||||||
"name": "Athena Local Talk",
|
"name": "Athena Local Talk",
|
||||||
"description": "Local German voice conversations through Athena without a public speech provider.",
|
"description": "Local German voice conversations through Athena without a public speech provider.",
|
||||||
"version": "0.1.3",
|
"version": "0.2.0",
|
||||||
"enabledByDefault": true,
|
"enabledByDefault": true,
|
||||||
"activation": {
|
"activation": {
|
||||||
"onStartup": true,
|
"onStartup": true,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "openclaw-plugin-athena-talk",
|
"name": "openclaw-plugin-athena-talk",
|
||||||
"version": "0.1.3",
|
"version": "0.2.0",
|
||||||
"description": "Private realtime Talk bridge for Athena STT, OpenClaw agent consult, and Athena TTS.",
|
"description": "Private realtime Talk bridge for Athena STT, OpenClaw agent consult, and Athena TTS.",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"private": true,
|
"private": true,
|
||||||
|
|||||||
Reference in New Issue
Block a user