Add Athena Whisper dictation provider to OpenClaw Talk plugin

This commit is contained in:
Mikei386
2026-09-16 15:20:38 +02:00
parent 1c93b9c7c1
commit ba68d4e8fb
8 changed files with 257 additions and 3 deletions
@@ -145,6 +145,88 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer {
return Buffer.concat([header, pcm]);
}
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio: Buffer): Buffer {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
const sample = ~audio[i] & 0xff;
const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7);
pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2);
}
return wavFromPcm16(pcm, 8000);
}
function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required<ProviderConfig> {
const talkConfig = record(record(record(cfg).talk).realtime);
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
}
class AthenaTranscriptionSession {
private connected = false;
private closed = false;
private readonly audio: Buffer[] = [];
private bytes = 0;
constructor(
private readonly req: {
cfg?: unknown;
providerConfig: Record<string, unknown>;
onSpeechStart?: () => void;
onTranscript?: (text: string) => void;
onError?: (error: Error) => void;
},
private readonly config: Required<ProviderConfig>,
) {}
async connect(): Promise<void> { this.connected = true; }
isConnected(): boolean { return this.connected && !this.closed; }
sendAudio(audio: Buffer): void {
if (!this.isConnected() || audio.length === 0) return;
if (this.bytes === 0) this.req.onSpeechStart?.();
const maxBytes = this.config.maxSpeechSeconds * 8000;
if (this.bytes + audio.length > maxBytes) {
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
this.close();
return;
}
this.audio.push(Buffer.from(audio));
this.bytes += audio.length;
}
close(): void {
if (this.closed) return;
this.closed = true;
this.connected = false;
if (!this.bytes) return;
const audio = Buffer.concat(this.audio);
this.audio.length = 0;
void this.transcribe(audio);
}
private async transcribe(audio: Buffer): Promise<void> {
try {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("language", this.config.language);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
method: "POST",
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
body: form,
signal: AbortSignal.timeout(4500),
});
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
const text = String(record(await response.json()).text || "").trim();
if (text) this.req.onTranscript?.(text);
} catch (error) {
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
}
}
}
function pcmRms(pcm: Buffer): number {
if (pcm.length < 2) return 0;
let sum = 0;
@@ -418,6 +500,16 @@ export default definePluginEntry({
name: "Athena Local Talk",
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)),
});
api.registerHttpRoute({
path: BROWSER_KEY_PATH,
auth: "plugin",