Add Athena Whisper dictation provider to OpenClaw Talk plugin

This commit is contained in:
Mikei386
2026-09-16 15:20:38 +02:00
parent 1c93b9c7c1
commit ba68d4e8fb
8 changed files with 257 additions and 3 deletions
+87
View File
@@ -124,6 +124,83 @@ function wavFromPcm16(pcm, sampleRate = 24000) {
header.writeUInt32LE(pcm.length, 40);
return Buffer.concat([header, pcm]);
}
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio) {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
const sample = ~audio[i] & 0xff;
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
}
return wavFromPcm16(pcm, 8000);
}
function resolveTranscriptionConfig(cfg, rawConfig) {
const talkConfig = record(record(record(cfg).talk).realtime);
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
class AthenaTranscriptionSession {
req;
config;
connected = false;
closed = false;
audio = [];
bytes = 0;
constructor(req, config) {
this.req = req;
this.config = config;
}
async connect() { this.connected = true; }
isConnected() { return this.connected && !this.closed; }
sendAudio(audio) {
if (!this.isConnected() || audio.length === 0)
return;
if (this.bytes === 0)
this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
}
close() {
if (this.closed)
return;
this.closed = true;
this.connected = false;
if (!this.bytes)
return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
}
async transcribe(audio) {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok)
throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text)
this.req.onTranscript?.(text);
}
catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
}
}
}
function pcmRms(pcm) {
if (pcm.length < 2)
return 0;
@@ -412,6 +489,16 @@ export default definePluginEntry({
name: "Athena Local Talk",
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
});
api.registerHttpRoute({
path: BROWSER_KEY_PATH,
auth: "plugin",