diff --git a/integrations/openclaw-athena-talk/README.md b/integrations/openclaw-athena-talk/README.md index b112190..b91c946 100644 --- a/integrations/openclaw-athena-talk/README.md +++ b/integrations/openclaw-athena-talk/README.md @@ -60,6 +60,18 @@ openclaw plugins install . --force --accept-capabilities openclaw plugins inspect athena-talk --runtime --json ``` +Version 1.2.0 also registers **Athena Whisper (Diktieren)** as a separate +realtime transcription provider through OpenClaw's official plugin API. In the +browser composer, hold the microphone for dictation, then release it to send +the 8 kHz G.711 audio through the Gateway. The plugin converts it to PCM WAV +and calls the same Athena `/audio/transcriptions` endpoint used by Talk. The +transcribed text is returned to the composer; this path does not invoke the +agent or TTS. The transcription provider reuses `talk.realtime.providers.athena-talk` +and the configured model provider for its URL/key, so no second credential is +needed. In `talk.catalog`, it appears under `transcription.providers`. OpenClaw +currently gives a transcription provider five seconds to return its final text +after recording stops; the plugin caps its Whisper request at 4.5 seconds. + Restart the gateway once if the installation does not trigger an automatic reload. The managed plugin copy is stored in OpenClaw's persistent data directory, so normal image updates do not remove it. Hermes remains unchanged; diff --git a/integrations/openclaw-athena-talk/dist/index.js b/integrations/openclaw-athena-talk/dist/index.js index 53aea33..3fb0ec2 100644 --- a/integrations/openclaw-athena-talk/dist/index.js +++ b/integrations/openclaw-athena-talk/dist/index.js @@ -124,6 +124,83 @@ function wavFromPcm16(pcm, sampleRate = 24000) { header.writeUInt32LE(pcm.length, 40); return Buffer.concat([header, pcm]); } +// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's +// existing Whisper endpoint accepts PCM WAV uploads. +function wavFromMulaw8k(audio) { + const pcm = Buffer.allocUnsafe(audio.length * 2); + for (let i = 0; i < audio.length; i += 1) { + const sample = ~audio[i] & 0xff; + const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7); + pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2); + } + return wavFromPcm16(pcm, 8000); +} +function resolveTranscriptionConfig(cfg, rawConfig) { + const talkConfig = record(record(record(cfg).talk).realtime); + const talkProvider = record(record(talkConfig.providers)["athena-talk"]); + return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } }); +} +class AthenaTranscriptionSession { + req; + config; + connected = false; + closed = false; + audio = []; + bytes = 0; + constructor(req, config) { + this.req = req; + this.config = config; + } + async connect() { this.connected = true; } + isConnected() { return this.connected && !this.closed; } + sendAudio(audio) { + if (!this.isConnected() || audio.length === 0) + return; + if (this.bytes === 0) + this.req.onSpeechStart?.(); + const maxBytes = this.config.maxSpeechSeconds * 8000; + if (this.bytes + audio.length > maxBytes) { + this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`)); + this.close(); + return; + } + this.audio.push(Buffer.from(audio)); + this.bytes += audio.length; + } + close() { + if (this.closed) + return; + this.closed = true; + this.connected = false; + if (!this.bytes) + return; + const audio = Buffer.concat(this.audio); + this.audio.length = 0; + void this.transcribe(audio); + } + async transcribe(audio) { + try { + const form = new FormData(); + form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav"); + form.append("model", "whisper-1"); + form.append("language", this.config.language); + const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, { + method: "POST", + headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {}, + body: form, + signal: AbortSignal.timeout(4500), + }); + if (!response.ok) + throw new Error(`Athena STT failed (HTTP ${response.status})`); + const text = String(record(await response.json()).text || "").trim(); + if (text) + this.req.onTranscript?.(text); + } + catch (error) { + this.req.onError?.(error instanceof Error ? error : new Error(String(error))); + } + } +} function pcmRms(pcm) { if (pcm.length < 2) return 0; @@ -412,6 +489,16 @@ export default definePluginEntry({ name: "Athena Local Talk", description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", register(api) { + api.registerRealtimeTranscriptionProvider({ + id: "athena-talk", + label: "Athena Whisper (Diktieren)", + defaultModel: "whisper-1", + models: ["whisper-1"], + autoSelectOrder: 1, + resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig), + isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl), + createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)), + }); api.registerHttpRoute({ path: BROWSER_KEY_PATH, auth: "plugin", diff --git a/integrations/openclaw-athena-talk/index.ts b/integrations/openclaw-athena-talk/index.ts index 8338771..2e16cc0 100644 --- a/integrations/openclaw-athena-talk/index.ts +++ b/integrations/openclaw-athena-talk/index.ts @@ -145,6 +145,88 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer { return Buffer.concat([header, pcm]); } +// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's +// existing Whisper endpoint accepts PCM WAV uploads. +function wavFromMulaw8k(audio: Buffer): Buffer { + const pcm = Buffer.allocUnsafe(audio.length * 2); + for (let i = 0; i < audio.length; i += 1) { + const sample = ~audio[i] & 0xff; + const magnitude = (((sample & 15) << 3) + 132) << ((sample >> 4) & 7); + pcm.writeInt16LE(Math.max(-32768, Math.min(32767, (sample & 128) ? 132 - magnitude : magnitude - 132)), i * 2); + } + return wavFromPcm16(pcm, 8000); +} + +function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required { + const talkConfig = record(record(record(cfg).talk).realtime); + const talkProvider = record(record(talkConfig.providers)["athena-talk"]); + return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } }); +} + +class AthenaTranscriptionSession { + private connected = false; + private closed = false; + private readonly audio: Buffer[] = []; + private bytes = 0; + + constructor( + private readonly req: { + cfg?: unknown; + providerConfig: Record; + onSpeechStart?: () => void; + onTranscript?: (text: string) => void; + onError?: (error: Error) => void; + }, + private readonly config: Required, + ) {} + + async connect(): Promise { this.connected = true; } + isConnected(): boolean { return this.connected && !this.closed; } + + sendAudio(audio: Buffer): void { + if (!this.isConnected() || audio.length === 0) return; + if (this.bytes === 0) this.req.onSpeechStart?.(); + const maxBytes = this.config.maxSpeechSeconds * 8000; + if (this.bytes + audio.length > maxBytes) { + this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`)); + this.close(); + return; + } + this.audio.push(Buffer.from(audio)); + this.bytes += audio.length; + } + + close(): void { + if (this.closed) return; + this.closed = true; + this.connected = false; + if (!this.bytes) return; + const audio = Buffer.concat(this.audio); + this.audio.length = 0; + void this.transcribe(audio); + } + + private async transcribe(audio: Buffer): Promise { + try { + const form = new FormData(); + form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav"); + form.append("model", "whisper-1"); + form.append("language", this.config.language); + const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, { + method: "POST", + headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {}, + body: form, + signal: AbortSignal.timeout(4500), + }); + if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`); + const text = String(record(await response.json()).text || "").trim(); + if (text) this.req.onTranscript?.(text); + } catch (error) { + this.req.onError?.(error instanceof Error ? error : new Error(String(error))); + } + } +} + function pcmRms(pcm: Buffer): number { if (pcm.length < 2) return 0; let sum = 0; @@ -418,6 +500,16 @@ export default definePluginEntry({ name: "Athena Local Talk", description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", register(api) { + api.registerRealtimeTranscriptionProvider({ + id: "athena-talk", + label: "Athena Whisper (Diktieren)", + defaultModel: "whisper-1", + models: ["whisper-1"], + autoSelectOrder: 1, + resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig), + isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl), + createSession: (req) => new AthenaTranscriptionSession(req, resolveTranscriptionConfig(req.cfg, req.providerConfig)), + }); api.registerHttpRoute({ path: BROWSER_KEY_PATH, auth: "plugin", diff --git a/integrations/openclaw-athena-talk/openclaw.plugin.json b/integrations/openclaw-athena-talk/openclaw.plugin.json index bca7b96..abf3267 100644 --- a/integrations/openclaw-athena-talk/openclaw.plugin.json +++ b/integrations/openclaw-athena-talk/openclaw.plugin.json @@ -6,6 +6,9 @@ "onStartup": true }, "contracts": { + "realtimeTranscriptionProviders": [ + "athena-talk" + ], "realtimeVoiceProviders": [ "athena-talk" ] diff --git a/integrations/openclaw-athena-talk/package-lock.json b/integrations/openclaw-athena-talk/package-lock.json index 4a17a69..9434fc9 100644 --- a/integrations/openclaw-athena-talk/package-lock.json +++ b/integrations/openclaw-athena-talk/package-lock.json @@ -1,12 +1,12 @@ { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.1.0", + "version": "1.2.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.1.0", + "version": "1.2.0", "devDependencies": { "@types/node": "^24.0.0", "openclaw": "2026.9.4", diff --git a/integrations/openclaw-athena-talk/package.json b/integrations/openclaw-athena-talk/package.json index df9f660..165aafa 100644 --- a/integrations/openclaw-athena-talk/package.json +++ b/integrations/openclaw-athena-talk/package.json @@ -1,6 +1,6 @@ { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.1.0", + "version": "1.2.0", "private": true, "description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS", "type": "module", diff --git a/integrations/openclaw-athena-talk/test-browser.mjs b/integrations/openclaw-athena-talk/test-browser.mjs index 7c304a8..aff21f4 100644 --- a/integrations/openclaw-athena-talk/test-browser.mjs +++ b/integrations/openclaw-athena-talk/test-browser.mjs @@ -21,6 +21,7 @@ test("browser session signs a short-lived token and proxies the SDP offer", asyn let provider; const routes = []; plugin.register({ + registerRealtimeTranscriptionProvider: () => {}, registerRealtimeVoiceProvider: (value) => { provider = value; }, registerHttpRoute: (value) => { routes.push(value); }, runtime: { config: { current: () => ({ talk: { realtime: { providers: { diff --git a/integrations/openclaw-athena-talk/test-transcription.mjs b/integrations/openclaw-athena-talk/test-transcription.mjs new file mode 100644 index 0000000..7a215c8 --- /dev/null +++ b/integrations/openclaw-athena-talk/test-transcription.mjs @@ -0,0 +1,59 @@ +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import { test } from "node:test"; +import plugin from "./dist/index.js"; + +test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => { + let transcription; + plugin.register({ + registerRealtimeTranscriptionProvider: (value) => { transcription = value; }, + registerRealtimeVoiceProvider: () => {}, + registerHttpRoute: () => {}, + }); + assert.equal(transcription.id, "athena-talk"); + + let uploads = 0; + const server = createServer(async (req, res) => { + uploads += 1; + assert.equal(req.url, "/v1/audio/transcriptions"); + assert.equal(req.headers.authorization, "Bearer test-key"); + const form = await new Request("http://localhost", { + method: "POST", + headers: { "Content-Type": req.headers["content-type"] }, + body: req, + duplex: "half", + }).formData(); + const wav = Buffer.from(await form.get("file").arrayBuffer()); + assert.equal(wav.toString("ascii", 0, 4), "RIFF"); + assert.equal(wav.readUInt32LE(24), 8000); + assert.equal(wav.readUInt16LE(34), 16); + assert.equal(wav.length, 48); + assert.equal(form.get("language"), "de"); + assert.equal(form.get("model"), "whisper-1"); + res.writeHead(200, { "Content-Type": "application/json" }).end(JSON.stringify({ text: "Hallo Athena" })); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const cfg = { talk: { realtime: { providers: { "athena-talk": { + baseUrl: `http://127.0.0.1:${server.address().port}/v1`, apiKey: "test-key", language: "de", + } } } } }; + const providerConfig = transcription.resolveConfig({ cfg, rawConfig: {} }); + assert.equal(transcription.isConfigured({ cfg, providerConfig }), true); + try { + const result = new Promise((resolve, reject) => { + const session = transcription.createSession({ + cfg, providerConfig, + onTranscript: resolve, + onError: reject, + }); + session.connect().then(() => { + session.sendAudio(Buffer.from([0xff, 0xff])); + session.close(); + }, reject); + }); + assert.equal(await result, "Hallo Athena"); + assert.equal(uploads, 1); + } finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + } +});