Add Athena Whisper dictation provider to OpenClaw Talk plugin

This commit is contained in:
Mikei386
2026-09-16 15:20:38 +02:00
parent 1c93b9c7c1
commit ba68d4e8fb
8 changed files with 257 additions and 3 deletions
@@ -60,6 +60,18 @@ openclaw plugins install . --force --accept-capabilities
openclaw plugins inspect athena-talk --runtime --json openclaw plugins inspect athena-talk --runtime --json
``` ```
Version 1.2.0 also registers **Athena Whisper (Diktieren)** as a separate
realtime transcription provider through OpenClaw's official plugin API. In the
browser composer, hold the microphone for dictation, then release it to send
the 8 kHz G.711 audio through the Gateway. The plugin converts it to PCM WAV
and calls the same Athena `/audio/transcriptions` endpoint used by Talk. The
transcribed text is returned to the composer; this path does not invoke the
agent or TTS. The transcription provider reuses `talk.realtime.providers.athena-talk`
and the configured model provider for its URL/key, so no second credential is
needed. In `talk.catalog`, it appears under `transcription.providers`. OpenClaw
currently gives a transcription provider five seconds to return its final text
after recording stops; the plugin caps its Whisper request at 4.5 seconds.
Restart the gateway once if the installation does not trigger an automatic Restart the gateway once if the installation does not trigger an automatic
reload. The managed plugin copy is stored in OpenClaw's persistent data reload. The managed plugin copy is stored in OpenClaw's persistent data
directory, so normal image updates do not remove it. Hermes remains unchanged; directory, so normal image updates do not remove it. Hermes remains unchanged;
+87
View File
@@ -124,6 +124,83 @@ function wavFromPcm16(pcm, sampleRate = 24000) {
header.writeUInt32LE(pcm.length, 40); header.writeUInt32LE(pcm.length, 40);
return Buffer.concat([header, pcm]); return Buffer.concat([header, pcm]);
} }
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio) {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
const sample = ~audio[i] & 0xff;
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
}
return wavFromPcm16(pcm, 8000);
}
function resolveTranscriptionConfig(cfg, rawConfig) {
const talkConfig = record(record(record(cfg).talk).realtime);
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
class AthenaTranscriptionSession {
req;
config;
connected = false;
closed = false;
audio = [];
bytes = 0;
constructor(req, config) {
this.req = req;
this.config = config;
}
async connect() { this.connected = true; }
isConnected() { return this.connected && !this.closed; }
sendAudio(audio) {
if (!this.isConnected() || audio.length === 0)
return;
if (this.bytes === 0)
this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
}
close() {
if (this.closed)
return;
this.closed = true;
this.connected = false;
if (!this.bytes)
return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
}
async transcribe(audio) {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok)
throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text)
this.req.onTranscript?.(text);
}
catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
}
}
}
function pcmRms(pcm) { function pcmRms(pcm) {
if (pcm.length < 2) if (pcm.length < 2)
return 0; return 0;
@@ -412,6 +489,16 @@ export default definePluginEntry({
name: "Athena Local Talk", name: "Athena Local Talk",
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
register(api) { register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
});
api.registerHttpRoute({ api.registerHttpRoute({
path: BROWSER_KEY_PATH, path: BROWSER_KEY_PATH,
auth: "plugin", auth: "plugin",
@@ -145,6 +145,88 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer {
return Buffer.concat([header, pcm]); return Buffer.concat([header, pcm]);
} }
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio: Buffer): Buffer {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
const sample = ~audio[i] & 0xff;
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
}
return wavFromPcm16(pcm, 8000);
}
function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required<ProviderConfig> {
const talkConfig = record(record(record(cfg).talk).realtime);
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
class AthenaTranscriptionSession {
private connected = false;
private closed = false;
private readonly audio: Buffer[] = [];
private bytes = 0;
constructor(
private readonly req: {
cfg?: unknown;
providerConfig: Record<string, unknown>;
onSpeechStart?: () => void;
onTranscript?: (text: string) => void;
onError?: (error: Error) => void;
},
private readonly config: Required<ProviderConfig>,
) {}
async connect(): Promise<void> { this.connected = true; }
isConnected(): boolean { return this.connected && !this.closed; }
sendAudio(audio: Buffer): void {
if (!this.isConnected() || audio.length === 0) return;
if (this.bytes === 0) this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
}
close(): void {
if (this.closed) return;
this.closed = true;
this.connected = false;
if (!this.bytes) return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
}
private async transcribe(audio: Buffer): Promise<void> {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text) this.req.onTranscript?.(text);
} catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
}
}
}
function pcmRms(pcm: Buffer): number { function pcmRms(pcm: Buffer): number {
if (pcm.length < 2) return 0; if (pcm.length < 2) return 0;
let sum = 0; let sum = 0;
@@ -418,6 +500,16 @@ export default definePluginEntry({
name: "Athena Local Talk", name: "Athena Local Talk",
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
register(api) { register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
});
api.registerHttpRoute({ api.registerHttpRoute({
path: BROWSER_KEY_PATH, path: BROWSER_KEY_PATH,
auth: "plugin", auth: "plugin",
@@ -6,6 +6,9 @@
"onStartup": true "onStartup": true
}, },
"contracts": { "contracts": {
"realtimeTranscriptionProviders": [
"athena-talk"
],
"realtimeVoiceProviders": [ "realtimeVoiceProviders": [
"athena-talk" "athena-talk"
] ]
+2 -2
View File
@@ -1,12 +1,12 @@
{ {
"name": "@casaderoll/openclaw-athena-talk", "name": "@casaderoll/openclaw-athena-talk",
"version": "1.1.0", "version": "1.2.0",
"lockfileVersion": 3, "lockfileVersion": 3,
"requires": true, "requires": true,
"packages": { "packages": {
"": { "": {
"name": "@casaderoll/openclaw-athena-talk", "name": "@casaderoll/openclaw-athena-talk",
"version": "1.1.0", "version": "1.2.0",
"devDependencies": { "devDependencies": {
"@types/node": "^24.0.0", "@types/node": "^24.0.0",
"openclaw": "2026.9.4", "openclaw": "2026.9.4",
@@ -1,6 +1,6 @@
{ {
"name": "@casaderoll/openclaw-athena-talk", "name": "@casaderoll/openclaw-athena-talk",
"version": "1.1.0", "version": "1.2.0",
"private": true, "private": true,
"description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS", "description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS",
"type": "module", "type": "module",
@@ -21,6 +21,7 @@ test("browser session signs a short-lived token and proxies the SDP offer", asyn
let provider; let provider;
const routes = []; const routes = [];
plugin.register({ plugin.register({
registerRealtimeTranscriptionProvider: () => {},
registerRealtimeVoiceProvider: (value) => { provider = value; }, registerRealtimeVoiceProvider: (value) => { provider = value; },
registerHttpRoute: (value) => { routes.push(value); }, registerHttpRoute: (value) => { routes.push(value); },
runtime: { config: { current: () => ({ talk: { realtime: { providers: { runtime: { config: { current: () => ({ talk: { realtime: { providers: {
@@ -0,0 +1,59 @@
import assert from "node:assert/strict";
import { createServer } from "node:http";
import { test } from "node:test";
import plugin from "./dist/index.js";
test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => {
let transcription;
plugin.register({
registerRealtimeTranscriptionProvider: (value) => { transcription = value; },
registerRealtimeVoiceProvider: () => {},
registerHttpRoute: () => {},
});
assert.equal(transcription.id, "athena-talk");
let uploads = 0;
const server = createServer(async (req, res) => {
uploads += 1;
assert.equal(req.url, "/v1/audio/transcriptions");
assert.equal(req.headers.authorization, "Bearer test-key");
const form = await new Request("http://localhost", {
method: "POST",
headers: { "Content-Type": req.headers["content-type"] },
body: req,
duplex: "half",
}).formData();
const wav = Buffer.from(await form.get("file").arrayBuffer());
assert.equal(wav.toString("ascii", 0, 4), "RIFF");
assert.equal(wav.readUInt32LE(24), 8000);
assert.equal(wav.readUInt16LE(34), 16);
assert.equal(wav.length, 48);
assert.equal(form.get("language"), "de");
assert.equal(form.get("model"), "whisper-1");
res.writeHead(200, { "Content-Type": "application/json" }).end(JSON.stringify({ text: "Hallo Athena" }));
});
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));
const cfg = { talk: { realtime: { providers: { "athena-talk": {
baseUrl: `http://127.0.0.1:${server.address().port}/v1`, apiKey: "test-key", language: "de",
} } } } };
const providerConfig = transcription.resolveConfig({ cfg, rawConfig: {} });
assert.equal(transcription.isConfigured({ cfg, providerConfig }), true);
try {
const result = new Promise((resolve, reject) => {
const session = transcription.createSession({
cfg, providerConfig,
onTranscript: resolve,
onError: reject,
});
session.connect().then(() => {
session.sendAudio(Buffer.from([0xff, 0xff]));
session.close();
}, reject);
});
assert.equal(await result, "Hallo Athena");
assert.equal(uploads, 1);
} finally {
server.closeAllConnections();
await new Promise((resolve) => server.close(resolve));
}
});