Stream long OpenClaw dictation in segments

This commit is contained in:
Mikei386
2026-09-21 16:06:15 +02:00
parent e577c55489
commit 46e5bbdf7f
10 changed files with 313 additions and 72 deletions
+21 -11
View File
@@ -60,19 +60,29 @@ openclaw plugins install . --force --accept-capabilities
openclaw plugins inspect athena-talk --runtime --json
```
Version 1.2.1 also registers **Athena Whisper (Diktieren)** as a separate
Version 1.3.0 also registers **Athena Whisper (Diktieren)** as a separate
realtime transcription provider through OpenClaw's official plugin API. In the
browser composer, hold the microphone for dictation, then release it to send
the 8 kHz G.711 audio through the Gateway. The plugin converts it to PCM WAV
and calls the same Athena `/audio/transcriptions` endpoint used by Talk. The
transcribed text is returned to the composer; this path does not invoke the
agent or TTS. The transcription provider reuses `talk.realtime.providers.athena-talk`
and the configured model provider for its URL/key. If that model provider has
no key, it reuses `tts.providers.openai.apiKey` only when the TTS and STT URLs
have the same origin. No second credential is needed. In `talk.catalog`, it
appears under `transcription.providers`. OpenClaw
currently gives a transcription provider five seconds to return its final text
after recording stops; the plugin caps its Whisper request at 4.5 seconds.
the 8 kHz G.711 audio through the Gateway. Short recordings are converted to
PCM WAV and sent to Athena's existing `/audio/transcriptions` endpoint in one
request. Longer recordings are split while the user is still speaking into
six-second windows with 0.5 seconds of overlap. The plugin sends these windows
sequentially to the persistent Whisper service, carries a short text prompt
into the next request, removes duplicated overlap words, and caches finished
segments until recording stops. Only the short final tail then remains inside
OpenClaw's fixed five-second final-drain window. Each Whisper request is capped
at 4.5 seconds.
This is incremental pre-transcription over OpenClaw's official transcription
provider API. Whisper.cpp still receives complete short WAV segments; it is
not a native token-streaming STT protocol. No OpenClaw core file was patched
and no additional speech container was introduced. The transcribed text is
returned to the composer; this path does not invoke the agent or TTS. The
provider reuses `talk.realtime.providers.athena-talk` and the configured model
provider for its URL/key. If that model provider has no key, it reuses
`tts.providers.openai.apiKey` only when the TTS and STT URLs have the same
origin. No second credential is needed. In `talk.catalog`, it appears under
`transcription.providers`.
### Voice-note file attachments
+95 -28
View File
@@ -4,6 +4,10 @@ const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 };
const BROWSER_OFFER_PATH = "/plugins/athena-talk/realtime/calls";
const BROWSER_KEY_PATH = "/plugins/athena-talk/realtime/public-key";
const MAX_OFFER_BYTES = 64 * 1024;
const DICTATION_SAMPLE_RATE_HZ = 8000;
const DICTATION_SEGMENT_BYTES = DICTATION_SAMPLE_RATE_HZ * 6;
const DICTATION_OVERLAP_BYTES = DICTATION_SAMPLE_RATE_HZ / 2;
const DICTATION_REQUEST_TIMEOUT_MS = 4500;
const browserKeys = generateKeyPairSync("ed25519");
const publicKeyPem = browserKeys.publicKey.export({ format: "pem", type: "spki" }).toString();
function base64url(value) {
@@ -149,13 +153,38 @@ function resolveTranscriptionConfig(cfg, rawConfig) {
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
function comparableWord(value) {
return value.toLocaleLowerCase("de-DE").replace(/[^\p{L}\p{N}]+/gu, "");
}
function removeTranscriptOverlap(previous, current) {
const priorWords = previous.trim().split(/\s+/).filter(Boolean);
const currentWords = current.trim().split(/\s+/).filter(Boolean);
const maximum = Math.min(12, priorWords.length, currentWords.length);
for (let count = maximum; count >= 1; count -= 1) {
const left = priorWords.slice(-count).map(comparableWord);
const right = currentWords.slice(0, count).map(comparableWord);
if (!left.every((word, index) => word && word === right[index]))
continue;
// A single short word is too ambiguous to remove safely. Longer words are
// sufficient because the audio overlap is only half a second.
if (count === 1 && left[0].length < 5)
continue;
return currentWords.slice(count).join(" ");
}
return currentWords.join(" ");
}
class AthenaTranscriptionSession {
req;
config;
connected = false;
closed = false;
audio = [];
bytes = 0;
bufferedBytes = 0;
totalBytes = 0;
processing = Promise.resolve();
completedTranscripts = [];
emittedTranscripts = 0;
processingError = null;
constructor(req, config) {
this.req = req;
this.config = config;
@@ -165,50 +194,88 @@ class AthenaTranscriptionSession {
sendAudio(audio) {
if (!this.isConnected() || audio.length === 0)
return;
if (this.bytes === 0)
if (this.totalBytes === 0)
this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
if (this.totalBytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
this.bufferedBytes += audio.length;
this.totalBytes += audio.length;
while (this.bufferedBytes >= DICTATION_SEGMENT_BYTES) {
this.queueFullSegment();
}
}
close() {
if (this.closed)
return;
this.closed = true;
this.connected = false;
if (!this.bytes)
if (!this.totalBytes)
return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
this.emitCompletedTranscripts();
const tail = Buffer.concat(this.audio);
this.audio = [];
this.bufferedBytes = 0;
if (tail.length > DICTATION_OVERLAP_BYTES || this.completedTranscripts.length === 0) {
this.queueTranscription(tail);
}
void this.processing.finally(() => {
this.emitCompletedTranscripts();
if (this.processingError && this.completedTranscripts.length === 0) {
this.req.onError?.(this.processingError);
}
});
}
async transcribe(audio) {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok)
throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text)
this.req.onTranscript?.(text);
}
catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
queueFullSegment() {
const buffered = Buffer.concat(this.audio);
const segment = Buffer.from(buffered.subarray(0, DICTATION_SEGMENT_BYTES));
const retained = Buffer.from(buffered.subarray(DICTATION_SEGMENT_BYTES - DICTATION_OVERLAP_BYTES));
this.audio = retained.length ? [retained] : [];
this.bufferedBytes = retained.length;
this.queueTranscription(segment);
}
queueTranscription(audio) {
if (!audio.length)
return;
this.processing = this.processing.then(async () => {
const previous = this.completedTranscripts.join(" ");
const text = await this.transcribe(audio, previous.slice(-240));
const novel = removeTranscriptOverlap(previous, text);
if (novel)
this.completedTranscripts.push(novel);
if (this.closed)
this.emitCompletedTranscripts();
}).catch((error) => {
this.processingError = error instanceof Error ? error : new Error(String(error));
});
}
emitCompletedTranscripts() {
while (this.emittedTranscripts < this.completedTranscripts.length) {
this.req.onTranscript?.(this.completedTranscripts[this.emittedTranscripts]);
this.emittedTranscripts += 1;
}
}
async transcribe(audio, prompt) {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
if (prompt)
form.append("prompt", prompt);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(DICTATION_REQUEST_TIMEOUT_MS),
});
if (!response.ok)
throw new Error(`Athena STT failed (HTTP ${response.status})`);
return String(record(await response.json()).text || "").trim();
}
}
function pcmRms(pcm) {
if (pcm.length < 2)
+94 -26
View File
@@ -20,6 +20,10 @@ type ProviderConfig = {
const BROWSER_OFFER_PATH = "/plugins/athena-talk/realtime/calls";
const BROWSER_KEY_PATH = "/plugins/athena-talk/realtime/public-key";
const MAX_OFFER_BYTES = 64 * 1024;
const DICTATION_SAMPLE_RATE_HZ = 8000;
const DICTATION_SEGMENT_BYTES = DICTATION_SAMPLE_RATE_HZ * 6;
const DICTATION_OVERLAP_BYTES = DICTATION_SAMPLE_RATE_HZ / 2;
const DICTATION_REQUEST_TIMEOUT_MS = 4500;
const browserKeys = generateKeyPairSync("ed25519");
const publicKeyPem = browserKeys.publicKey.export({ format: "pem", type: "spki" }).toString();
@@ -171,11 +175,36 @@ function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required<
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
function comparableWord(value: string): string {
return value.toLocaleLowerCase("de-DE").replace(/[^\p{L}\p{N}]+/gu, "");
}
function removeTranscriptOverlap(previous: string, current: string): string {
const priorWords = previous.trim().split(/\s+/).filter(Boolean);
const currentWords = current.trim().split(/\s+/).filter(Boolean);
const maximum = Math.min(12, priorWords.length, currentWords.length);
for (let count = maximum; count >= 1; count -= 1) {
const left = priorWords.slice(-count).map(comparableWord);
const right = currentWords.slice(0, count).map(comparableWord);
if (!left.every((word, index) => word && word === right[index])) continue;
// A single short word is too ambiguous to remove safely. Longer words are
// sufficient because the audio overlap is only half a second.
if (count === 1 && left[0].length < 5) continue;
return currentWords.slice(count).join(" ");
}
return currentWords.join(" ");
}
class AthenaTranscriptionSession {
private connected = false;
private closed = false;
private readonly audio: Buffer[] = [];
private bytes = 0;
private audio: Buffer[] = [];
private bufferedBytes = 0;
private totalBytes = 0;
private processing: Promise<void> = Promise.resolve();
private readonly completedTranscripts: string[] = [];
private emittedTranscripts = 0;
private processingError: Error | null = null;
constructor(
private readonly req: {
@@ -193,46 +222,85 @@ class AthenaTranscriptionSession {
sendAudio(audio: Buffer): void {
if (!this.isConnected() || audio.length === 0) return;
if (this.bytes === 0) this.req.onSpeechStart?.();
if (this.totalBytes === 0) this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
if (this.totalBytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
this.bufferedBytes += audio.length;
this.totalBytes += audio.length;
while (this.bufferedBytes >= DICTATION_SEGMENT_BYTES) {
this.queueFullSegment();
}
}
close(): void {
if (this.closed) return;
this.closed = true;
this.connected = false;
if (!this.bytes) return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
if (!this.totalBytes) return;
this.emitCompletedTranscripts();
const tail = Buffer.concat(this.audio);
this.audio = [];
this.bufferedBytes = 0;
if (tail.length > DICTATION_OVERLAP_BYTES || this.completedTranscripts.length === 0) {
this.queueTranscription(tail);
}
void this.processing.finally(() => {
this.emitCompletedTranscripts();
if (this.processingError && this.completedTranscripts.length === 0) {
this.req.onError?.(this.processingError);
}
});
}
private async transcribe(audio: Buffer): Promise<void> {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text) this.req.onTranscript?.(text);
} catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
private queueFullSegment(): void {
const buffered = Buffer.concat(this.audio);
const segment = Buffer.from(buffered.subarray(0, DICTATION_SEGMENT_BYTES));
const retained = Buffer.from(buffered.subarray(DICTATION_SEGMENT_BYTES - DICTATION_OVERLAP_BYTES));
this.audio = retained.length ? [retained] : [];
this.bufferedBytes = retained.length;
this.queueTranscription(segment);
}
private queueTranscription(audio: Buffer): void {
if (!audio.length) return;
this.processing = this.processing.then(async () => {
const previous = this.completedTranscripts.join(" ");
const text = await this.transcribe(audio, previous.slice(-240));
const novel = removeTranscriptOverlap(previous, text);
if (novel) this.completedTranscripts.push(novel);
if (this.closed) this.emitCompletedTranscripts();
}).catch((error: unknown) => {
this.processingError = error instanceof Error ? error : new Error(String(error));
});
}
private emitCompletedTranscripts(): void {
while (this.emittedTranscripts < this.completedTranscripts.length) {
this.req.onTranscript?.(this.completedTranscripts[this.emittedTranscripts]);
this.emittedTranscripts += 1;
}
}
private async transcribe(audio: Buffer, prompt: string): Promise<string> {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
if (prompt) form.append("prompt", prompt);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(DICTATION_REQUEST_TIMEOUT_MS),
});
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
return String(record(await response.json()).text || "").trim();
}
}
function pcmRms(pcm: Buffer): number {
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.2.1",
"version": "1.3.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.2.1",
"version": "1.3.0",
"devDependencies": {
"@types/node": "^24.0.0",
"openclaw": "2026.9.4",
@@ -1,6 +1,6 @@
{
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.2.1",
"version": "1.3.0",
"private": true,
"description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS",
"type": "module",
@@ -3,6 +3,14 @@ import { createServer } from "node:http";
import { test } from "node:test";
import plugin from "./dist/index.js";
async function waitFor(predicate, timeoutMs = 2000) {
const deadline = Date.now() + timeoutMs;
while (!predicate()) {
if (Date.now() >= deadline) throw new Error("timed out waiting for condition");
await new Promise((resolve) => setTimeout(resolve, 10));
}
}
test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => {
let transcription;
plugin.register({
@@ -79,3 +87,73 @@ test("dictation reuses only a TTS key for the same Athena origin", () => {
cfg: withTts("http://other:8081/v1"), rawConfig: {},
}).apiKey, "");
});
test("long dictation is transcribed incrementally before the recording closes", async () => {
let transcription;
plugin.register({
registerRealtimeTranscriptionProvider: (value) => { transcription = value; },
registerRealtimeVoiceProvider: () => {},
registerHttpRoute: () => {},
});
const answers = [
"Dies ist ein langer Abschnitt",
"langer Abschnitt mit einer Fortsetzung",
"einer Fortsetzung und einem Ende.",
];
let uploads = 0;
const server = createServer(async (req, res) => {
const index = uploads++;
const form = await new Request("http://localhost", {
method: "POST",
headers: { "Content-Type": req.headers["content-type"] },
body: req,
duplex: "half",
}).formData();
const wav = Buffer.from(await form.get("file").arrayBuffer());
assert.equal(wav.readUInt32LE(24), 8000);
if (index > 0) assert.ok(String(form.get("prompt") || "").length > 0);
res.writeHead(200, { "Content-Type": "application/json" })
.end(JSON.stringify({ text: answers[index] }));
});
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));
const cfg = { talk: { realtime: { providers: { "athena-talk": {
baseUrl: `http://127.0.0.1:${server.address().port}/v1`, language: "de",
} } } } };
const providerConfig = transcription.resolveConfig({ cfg, rawConfig: {} });
const transcripts = [];
const errors = [];
try {
const session = transcription.createSession({
cfg,
providerConfig,
onTranscript: (text) => transcripts.push(text),
onError: (error) => errors.push(error),
});
await session.connect();
// Six seconds start the first request while dictation is still active.
session.sendAudio(Buffer.alloc(48_000, 0xff));
await waitFor(() => uploads === 1);
assert.deepEqual(transcripts, []);
// Another 5.5 seconds form the next overlapping segment. The remaining
// 2 seconds are finalized only when the user stops dictation.
session.sendAudio(Buffer.alloc(44_000, 0xff));
await waitFor(() => uploads === 2);
session.sendAudio(Buffer.alloc(16_000, 0xff));
session.close();
await waitFor(() => transcripts.length === 3);
assert.deepEqual(transcripts, [
"Dies ist ein langer Abschnitt",
"mit einer Fortsetzung",
"und einem Ende.",
]);
assert.equal(uploads, 3);
assert.deepEqual(errors, []);
} finally {
server.closeAllConnections();
await new Promise((resolve) => server.close(resolve));
}
});