Add Athena Whisper dictation provider to OpenClaw Talk plugin
This commit is contained in:
@@ -60,6 +60,18 @@ openclaw plugins install . --force --accept-capabilities
|
||||
openclaw plugins inspect athena-talk --runtime --json
|
||||
```
|
||||
|
||||
Version 1.2.0 also registers **Athena Whisper (Diktieren)** as a separate
|
||||
realtime transcription provider through OpenClaw's official plugin API. In the
|
||||
browser composer, hold the microphone for dictation, then release it to send
|
||||
the 8 kHz G.711 audio through the Gateway. The plugin converts it to PCM WAV
|
||||
and calls the same Athena `/audio/transcriptions` endpoint used by Talk. The
|
||||
transcribed text is returned to the composer; this path does not invoke the
|
||||
agent or TTS. The transcription provider reuses `talk.realtime.providers.athena-talk`
|
||||
and the configured model provider for its URL/key, so no second credential is
|
||||
needed. In `talk.catalog`, it appears under `transcription.providers`. OpenClaw
|
||||
currently gives a transcription provider five seconds to return its final text
|
||||
after recording stops; the plugin caps its Whisper request at 4.5 seconds.
|
||||
|
||||
Restart the gateway once if the installation does not trigger an automatic
|
||||
reload. The managed plugin copy is stored in OpenClaw's persistent data
|
||||
directory, so normal image updates do not remove it. Hermes remains unchanged;
|
||||
|
||||
+87
@@ -124,6 +124,83 @@ function wavFromPcm16(pcm, sampleRate = 24000) {
|
||||
header.writeUInt32LE(pcm.length, 40);
|
||||
return Buffer.concat([header, pcm]);
|
||||
}
|
||||
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
|
||||
// existing Whisper endpoint accepts PCM WAV uploads.
|
||||
function wavFromMulaw8k(audio) {
|
||||
const pcm = Buffer.allocUnsafe(audio.length * 2);
|
||||
for (let i = 0; i < audio.length; i += 1) {
|
||||
const sample = ~audio[i] & 0xff;
|
||||
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
|
||||
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
|
||||
}
|
||||
return wavFromPcm16(pcm, 8000);
|
||||
}
|
||||
function resolveTranscriptionConfig(cfg, rawConfig) {
|
||||
const talkConfig = record(record(record(cfg).talk).realtime);
|
||||
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
|
||||
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
|
||||
}
|
||||
class AthenaTranscriptionSession {
|
||||
req;
|
||||
config;
|
||||
connected = false;
|
||||
closed = false;
|
||||
audio = [];
|
||||
bytes = 0;
|
||||
constructor(req, config) {
|
||||
this.req = req;
|
||||
this.config = config;
|
||||
}
|
||||
async connect() { this.connected = true; }
|
||||
isConnected() { return this.connected && !this.closed; }
|
||||
sendAudio(audio) {
|
||||
if (!this.isConnected() || audio.length === 0)
|
||||
return;
|
||||
if (this.bytes === 0)
|
||||
this.req.onSpeechStart?.();
|
||||
const maxBytes = this.config.maxSpeechSeconds * 8000;
|
||||
if (this.bytes + audio.length > maxBytes) {
|
||||
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
|
||||
this.close();
|
||||
return;
|
||||
}
|
||||
this.audio.push(Buffer.from(audio));
|
||||
this.bytes += audio.length;
|
||||
}
|
||||
close() {
|
||||
if (this.closed)
|
||||
return;
|
||||
this.closed = true;
|
||||
this.connected = false;
|
||||
if (!this.bytes)
|
||||
return;
|
||||
const audio = Buffer.concat(this.audio);
|
||||
this.audio.length = 0;
|
||||
void this.transcribe(audio);
|
||||
}
|
||||
async transcribe(audio) {
|
||||
try {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(4500),
|
||||
});
|
||||
if (!response.ok)
|
||||
throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
const text = String(record(await response.json()).text || "").trim();
|
||||
if (text)
|
||||
this.req.onTranscript?.(text);
|
||||
}
|
||||
catch (error) {
|
||||
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
}
|
||||
}
|
||||
}
|
||||
function pcmRms(pcm) {
|
||||
if (pcm.length < 2)
|
||||
return 0;
|
||||
@@ -412,6 +489,16 @@ export default definePluginEntry({
|
||||
name: "Athena Local Talk",
|
||||
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
|
||||
register(api) {
|
||||
api.registerRealtimeTranscriptionProvider({
|
||||
id: "athena-talk",
|
||||
label: "Athena Whisper (Diktieren)",
|
||||
defaultModel: "whisper-1",
|
||||
models: ["whisper-1"],
|
||||
autoSelectOrder: 1,
|
||||
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
|
||||
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
|
||||
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
|
||||
});
|
||||
api.registerHttpRoute({
|
||||
path: BROWSER_KEY_PATH,
|
||||
auth: "plugin",
|
||||
|
||||
@@ -145,6 +145,88 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer {
|
||||
return Buffer.concat([header, pcm]);
|
||||
}
|
||||
|
||||
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
|
||||
// existing Whisper endpoint accepts PCM WAV uploads.
|
||||
function wavFromMulaw8k(audio: Buffer): Buffer {
|
||||
const pcm = Buffer.allocUnsafe(audio.length * 2);
|
||||
for (let i = 0; i < audio.length; i += 1) {
|
||||
const sample = ~audio[i] & 0xff;
|
||||
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
|
||||
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
|
||||
}
|
||||
return wavFromPcm16(pcm, 8000);
|
||||
}
|
||||
|
||||
function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required<ProviderConfig> {
|
||||
const talkConfig = record(record(record(cfg).talk).realtime);
|
||||
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
|
||||
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
|
||||
}
|
||||
|
||||
class AthenaTranscriptionSession {
|
||||
private connected = false;
|
||||
private closed = false;
|
||||
private readonly audio: Buffer[] = [];
|
||||
private bytes = 0;
|
||||
|
||||
constructor(
|
||||
private readonly req: {
|
||||
cfg?: unknown;
|
||||
providerConfig: Record<string, unknown>;
|
||||
onSpeechStart?: () => void;
|
||||
onTranscript?: (text: string) => void;
|
||||
onError?: (error: Error) => void;
|
||||
},
|
||||
private readonly config: Required<ProviderConfig>,
|
||||
) {}
|
||||
|
||||
async connect(): Promise<void> { this.connected = true; }
|
||||
isConnected(): boolean { return this.connected && !this.closed; }
|
||||
|
||||
sendAudio(audio: Buffer): void {
|
||||
if (!this.isConnected() || audio.length === 0) return;
|
||||
if (this.bytes === 0) this.req.onSpeechStart?.();
|
||||
const maxBytes = this.config.maxSpeechSeconds * 8000;
|
||||
if (this.bytes + audio.length > maxBytes) {
|
||||
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
|
||||
this.close();
|
||||
return;
|
||||
}
|
||||
this.audio.push(Buffer.from(audio));
|
||||
this.bytes += audio.length;
|
||||
}
|
||||
|
||||
close(): void {
|
||||
if (this.closed) return;
|
||||
this.closed = true;
|
||||
this.connected = false;
|
||||
if (!this.bytes) return;
|
||||
const audio = Buffer.concat(this.audio);
|
||||
this.audio.length = 0;
|
||||
void this.transcribe(audio);
|
||||
}
|
||||
|
||||
private async transcribe(audio: Buffer): Promise<void> {
|
||||
try {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(4500),
|
||||
});
|
||||
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
const text = String(record(await response.json()).text || "").trim();
|
||||
if (text) this.req.onTranscript?.(text);
|
||||
} catch (error) {
|
||||
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function pcmRms(pcm: Buffer): number {
|
||||
if (pcm.length < 2) return 0;
|
||||
let sum = 0;
|
||||
@@ -418,6 +500,16 @@ export default definePluginEntry({
|
||||
name: "Athena Local Talk",
|
||||
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
|
||||
register(api) {
|
||||
api.registerRealtimeTranscriptionProvider({
|
||||
id: "athena-talk",
|
||||
label: "Athena Whisper (Diktieren)",
|
||||
defaultModel: "whisper-1",
|
||||
models: ["whisper-1"],
|
||||
autoSelectOrder: 1,
|
||||
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
|
||||
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
|
||||
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
|
||||
});
|
||||
api.registerHttpRoute({
|
||||
path: BROWSER_KEY_PATH,
|
||||
auth: "plugin",
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
"onStartup": true
|
||||
},
|
||||
"contracts": {
|
||||
"realtimeTranscriptionProviders": [
|
||||
"athena-talk"
|
||||
],
|
||||
"realtimeVoiceProviders": [
|
||||
"athena-talk"
|
||||
]
|
||||
|
||||
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "@casaderoll/openclaw-athena-talk",
|
||||
"version": "1.1.0",
|
||||
"version": "1.2.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@casaderoll/openclaw-athena-talk",
|
||||
"version": "1.1.0",
|
||||
"version": "1.2.0",
|
||||
"devDependencies": {
|
||||
"@types/node": "^24.0.0",
|
||||
"openclaw": "2026.9.4",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@casaderoll/openclaw-athena-talk",
|
||||
"version": "1.1.0",
|
||||
"version": "1.2.0",
|
||||
"private": true,
|
||||
"description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS",
|
||||
"type": "module",
|
||||
|
||||
@@ -21,6 +21,7 @@ test("browser session signs a short-lived token and proxies the SDP offer", asyn
|
||||
let provider;
|
||||
const routes = [];
|
||||
plugin.register({
|
||||
registerRealtimeTranscriptionProvider: () => {},
|
||||
registerRealtimeVoiceProvider: (value) => { provider = value; },
|
||||
registerHttpRoute: (value) => { routes.push(value); },
|
||||
runtime: { config: { current: () => ({ talk: { realtime: { providers: {
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { createServer } from "node:http";
|
||||
import { test } from "node:test";
|
||||
import plugin from "./dist/index.js";
|
||||
|
||||
test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => {
|
||||
let transcription;
|
||||
plugin.register({
|
||||
registerRealtimeTranscriptionProvider: (value) => { transcription = value; },
|
||||
registerRealtimeVoiceProvider: () => {},
|
||||
registerHttpRoute: () => {},
|
||||
});
|
||||
assert.equal(transcription.id, "athena-talk");
|
||||
|
||||
let uploads = 0;
|
||||
const server = createServer(async (req, res) => {
|
||||
uploads += 1;
|
||||
assert.equal(req.url, "/v1/audio/transcriptions");
|
||||
assert.equal(req.headers.authorization, "Bearer test-key");
|
||||
const form = await new Request("http://localhost", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": req.headers["content-type"] },
|
||||
body: req,
|
||||
duplex: "half",
|
||||
}).formData();
|
||||
const wav = Buffer.from(await form.get("file").arrayBuffer());
|
||||
assert.equal(wav.toString("ascii", 0, 4), "RIFF");
|
||||
assert.equal(wav.readUInt32LE(24), 8000);
|
||||
assert.equal(wav.readUInt16LE(34), 16);
|
||||
assert.equal(wav.length, 48);
|
||||
assert.equal(form.get("language"), "de");
|
||||
assert.equal(form.get("model"), "whisper-1");
|
||||
res.writeHead(200, { "Content-Type": "application/json" }).end(JSON.stringify({ text: "Hallo Athena" }));
|
||||
});
|
||||
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));
|
||||
const cfg = { talk: { realtime: { providers: { "athena-talk": {
|
||||
baseUrl: `http://127.0.0.1:${server.address().port}/v1`, apiKey: "test-key", language: "de",
|
||||
} } } } };
|
||||
const providerConfig = transcription.resolveConfig({ cfg, rawConfig: {} });
|
||||
assert.equal(transcription.isConfigured({ cfg, providerConfig }), true);
|
||||
try {
|
||||
const result = new Promise((resolve, reject) => {
|
||||
const session = transcription.createSession({
|
||||
cfg, providerConfig,
|
||||
onTranscript: resolve,
|
||||
onError: reject,
|
||||
});
|
||||
session.connect().then(() => {
|
||||
session.sendAudio(Buffer.from([0xff, 0xff]));
|
||||
session.close();
|
||||
}, reject);
|
||||
});
|
||||
assert.equal(await result, "Hallo Athena");
|
||||
assert.equal(uploads, 1);
|
||||
} finally {
|
||||
server.closeAllConnections();
|
||||
await new Promise((resolve) => server.close(resolve));
|
||||
}
|
||||
});
|
||||
Reference in New Issue
Block a user