diff --git a/README.md b/README.md index 61a25ee..edc9675 100644 --- a/README.md +++ b/README.md @@ -103,7 +103,13 @@ Standard, bis Hermes' Sitzungsfehler behoben ist. Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über `/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten -werden lokal auf Athena verarbeitet. +werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale +Realtime-Provider unter +[`integrations/openclaw-athena-talk`](integrations/openclaw-athena-talk). Er +verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives +Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten +bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis +und bleibt deshalb bei normalen Container-Updates bestehen. - Portainer: `https://192.168.1.212:9443` - Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` diff --git a/compose.yaml b/compose.yaml index ebd1d00..1715f9c 100644 --- a/compose.yaml +++ b/compose.yaml @@ -732,14 +732,16 @@ services: WHISPER_HOST: 0.0.0.0 WHISPER_PORT: "8084" WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli - WHISPER_MODEL: /models/ggml-large-v3-turbo.bin - WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin} + WHISPER_MODEL: /models/ggml-small.bin + WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin} + WHISPER_SERVER_PORT: "8085" WHISPER_THREADS: ${WHISPER_THREADS:-8} WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de} networks: [frontend, inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] - cap_add: [CHOWN, SETUID, SETGID] + # The entrypoint supervises whisper-server after dropping it to uid 10004. + cap_add: [CHOWN, SETUID, SETGID, KILL] healthcheck: test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"] interval: 10s diff --git a/integrations/openclaw-athena-talk/README.md b/integrations/openclaw-athena-talk/README.md new file mode 100644 index 0000000..19914fc --- /dev/null +++ b/integrations/openclaw-athena-talk/README.md @@ -0,0 +1,39 @@ +# Athena Local Talk for OpenClaw + +This private OpenClaw provider connects browser/Desktop Talk to the existing +Athena speech stack: + +1. local VAD collects a spoken utterance, +2. Athena Whisper transcribes it, +3. OpenClaw's normal agent-consult path answers with its configured model and + tools, +4. Athena XTTS/Piper returns PCM audio to the Talk client. + +No public speech provider is used. The provider reuses the already configured +`models.providers.athena` base URL and API key; no second key copy is required. + +Recommended `talk.realtime` configuration: + +```json +{ + "provider": "athena-talk", + "model": "athena-local", + "speakerVoice": "alloy", + "language": "de", + "mode": "realtime", + "transport": "gateway-relay", + "brain": "agent-consult", + "providers": { + "athena-talk": { + "vadThreshold": 0.018, + "silenceDurationMs": 750, + "prefixPaddingMs": 300, + "maxSpeechSeconds": 45 + } + } +} +``` + +Install from the OpenClaw container with `openclaw plugins install ` and +restart the gateway once. The plugin is stored in OpenClaw's persistent data +directory, so normal image updates do not remove it. diff --git a/integrations/openclaw-athena-talk/dist/index.js b/integrations/openclaw-athena-talk/dist/index.js new file mode 100644 index 0000000..61742e7 --- /dev/null +++ b/integrations/openclaw-athena-talk/dist/index.js @@ -0,0 +1,302 @@ +import { randomUUID } from "node:crypto"; +import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry"; +const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 }; +function record(value) { + return value && typeof value === "object" && !Array.isArray(value) + ? value + : {}; +} +function resolveConfig(req) { + const raw = record(req.providerConfig); + const modelProviders = record(record(req.cfg).models).providers; + const athena = record(record(modelProviders).athena); + return { + baseUrl: String(raw.baseUrl || athena.baseUrl || "http://192.168.1.212:8081/v1").replace(/\/$/, ""), + apiKey: String(raw.apiKey || athena.apiKey || ""), + voice: String(raw.voice || req.voice || "alloy"), + language: String(raw.language || req.language || "de"), + vadThreshold: Number(raw.vadThreshold ?? 0.018), + silenceDurationMs: Number(raw.silenceDurationMs ?? 750), + prefixPaddingMs: Number(raw.prefixPaddingMs ?? 300), + maxSpeechSeconds: Number(raw.maxSpeechSeconds ?? 45), + }; +} +function wavFromPcm16(pcm, sampleRate = 24000) { + const header = Buffer.alloc(44); + header.write("RIFF", 0); + header.writeUInt32LE(36 + pcm.length, 4); + header.write("WAVEfmt ", 8); + header.writeUInt32LE(16, 16); + header.writeUInt16LE(1, 20); + header.writeUInt16LE(1, 22); + header.writeUInt32LE(sampleRate, 24); + header.writeUInt32LE(sampleRate * 2, 28); + header.writeUInt16LE(2, 32); + header.writeUInt16LE(16, 34); + header.write("data", 36); + header.writeUInt32LE(pcm.length, 40); + return Buffer.concat([header, pcm]); +} +function pcmRms(pcm) { + if (pcm.length < 2) + return 0; + let sum = 0; + const count = Math.floor(pcm.length / 2); + for (let i = 0; i < count; i += 1) { + const value = pcm.readInt16LE(i * 2) / 32768; + sum += value * value; + } + return Math.sqrt(sum / count); +} +function readWavPcm24k(wav) { + if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") { + throw new Error("Athena TTS did not return PCM WAV audio"); + } + let offset = 12; + let sampleRate = 0; + let channels = 0; + let bits = 0; + let data; + while (offset + 8 <= wav.length) { + const id = wav.toString("ascii", offset, offset + 4); + const size = wav.readUInt32LE(offset + 4); + const start = offset + 8; + if (id === "fmt " && size >= 16) { + if (wav.readUInt16LE(start) !== 1) + throw new Error("Athena TTS WAV is not PCM"); + channels = wav.readUInt16LE(start + 2); + sampleRate = wav.readUInt32LE(start + 4); + bits = wav.readUInt16LE(start + 14); + } + else if (id === "data") { + data = wav.subarray(start, Math.min(start + size, wav.length)); + } + offset = start + size + (size % 2); + } + if (!data || !sampleRate || bits !== 16 || channels < 1) { + throw new Error("Unsupported Athena TTS WAV format"); + } + const frames = Math.floor(data.length / (2 * channels)); + const mono = new Int16Array(frames); + for (let i = 0; i < frames; i += 1) + mono[i] = data.readInt16LE(i * channels * 2); + if (sampleRate === 24000) + return Buffer.from(mono.buffer); + const outFrames = Math.max(1, Math.round(frames * 24000 / sampleRate)); + const out = Buffer.alloc(outFrames * 2); + for (let i = 0; i < outFrames; i += 1) { + const source = i * sampleRate / 24000; + const left = Math.min(frames - 1, Math.floor(source)); + const right = Math.min(frames - 1, left + 1); + const fraction = source - left; + const value = Math.round(mono[left] * (1 - fraction) + mono[right] * fraction); + out.writeInt16LE(Math.max(-32768, Math.min(32767, value)), i * 2); + } + return out; +} +class AthenaTalkBridge { + req; + cfg; + supportsToolResultContinuation = false; + supportsToolResultSuppression = false; + connected = false; + closed = false; + speaking = false; + speech = []; + prefix = []; + prefixBytes = 0; + silenceTimer; + active; + generation = 0; + constructor(req, cfg) { + this.req = req; + this.cfg = cfg; + } + async connect() { + this.connected = true; + this.req.onEvent?.({ direction: "server", type: "session.created" }); + this.req.onReady?.(); + } + isConnected() { return this.connected && !this.closed; } + setMediaTimestamp(_timestamp) { } + sendAudio(chunk) { + if (!this.isConnected() || chunk.length === 0) + return; + const voiced = pcmRms(chunk) >= this.cfg.vadThreshold; + const maxPrefix = Math.round(24000 * 2 * this.cfg.prefixPaddingMs / 1000); + if (!this.speaking) { + this.prefix.push(Buffer.from(chunk)); + this.prefixBytes += chunk.length; + while (this.prefixBytes > maxPrefix && this.prefix.length > 1) { + this.prefixBytes -= this.prefix.shift().length; + } + if (!voiced) + return; + if (this.active) { + this.active.abort(); + this.active = undefined; + this.generation += 1; + this.req.onClearAudio?.("barge-in"); + } + this.speaking = true; + this.speech = this.prefix; + this.prefix = []; + this.prefixBytes = 0; + } + else { + this.speech.push(Buffer.from(chunk)); + } + const maxBytes = this.cfg.maxSpeechSeconds * 24000 * 2; + if (this.speech.reduce((sum, part) => sum + part.length, 0) >= maxBytes) { + void this.finishSpeech(); + return; + } + if (voiced) { + if (this.silenceTimer) + clearTimeout(this.silenceTimer); + this.silenceTimer = setTimeout(() => void this.finishSpeech(), this.cfg.silenceDurationMs); + } + } + sendUserMessage(text) { + const trimmed = text.trim(); + if (trimmed) + void this.answer(trimmed); + } + triggerGreeting(instructions) { + void this.answer(instructions?.trim() || "Begrüße mich kurz auf Deutsch."); + } + submitToolResult(_callId, _result) { } + acknowledgeMark(_markName) { } + close() { + this.closed = true; + this.connected = false; + this.generation += 1; + this.active?.abort(); + if (this.silenceTimer) + clearTimeout(this.silenceTimer); + this.req.onClose?.("completed"); + } + async finishSpeech() { + if (!this.speaking) + return; + this.speaking = false; + if (this.silenceTimer) + clearTimeout(this.silenceTimer); + this.silenceTimer = undefined; + const pcm = Buffer.concat(this.speech); + this.speech = []; + if (pcm.length < 24000 * 2 * 0.25) + return; + try { + const text = await this.transcribe(pcm); + if (!text || !this.isConnected()) + return; + this.req.onTranscript?.("user", text, true); + await this.answer(text); + } + catch (error) { + if (error.name !== "AbortError") + this.req.onError?.(error); + } + } + async transcribe(pcm) { + const form = new FormData(); + form.append("file", new Blob([wavFromPcm16(pcm)], { type: "audio/wav" }), "talk.wav"); + form.append("model", "whisper-1"); + form.append("language", this.cfg.language); + const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, { + method: "POST", + headers: this.authHeaders(), + body: form, + }); + if (!response.ok) + throw new Error(`Athena STT failed (HTTP ${response.status})`); + const payload = record(await response.json()); + return String(payload.text || "").trim(); + } + async answer(prompt) { + if (!this.req.runAgentConsult) + throw new Error("OpenClaw agent-consult is unavailable"); + this.active?.abort(); + const controller = new AbortController(); + this.active = controller; + const generation = ++this.generation; + const responseId = `athena-${randomUUID()}`; + try { + const result = await this.req.runAgentConsult({ prompt, signal: controller.signal }); + if (generation !== this.generation || !this.isConnected()) + return; + const text = String(result?.text || "").trim(); + if (!text) + return; + this.req.onTranscript?.("assistant", text, true); + const pcm = await this.synthesize(text, controller.signal); + if (generation !== this.generation || !this.isConnected()) + return; + const frameBytes = 24000 * 2 / 50; + for (let offset = 0; offset < pcm.length; offset += frameBytes) { + this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); + } + this.req.onResponseDone?.({ status: "completed", responseId }); + this.req.onEvent?.({ direction: "server", type: "response.done", responseId }); + } + catch (error) { + if (error.name === "AbortError") { + this.req.onResponseDone?.({ status: "cancelled", responseId }); + } + else { + this.req.onError?.(error); + this.req.onResponseDone?.({ status: "failed", responseId, error: String(error) }); + } + } + finally { + if (this.active === controller) + this.active = undefined; + } + } + async synthesize(text, signal) { + const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, { + method: "POST", + headers: { "Content-Type": "application/json", ...this.authHeaders() }, + body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }), + signal, + }); + if (!response.ok) + throw new Error(`Athena TTS failed (HTTP ${response.status})`); + return readWavPcm24k(Buffer.from(await response.arrayBuffer())); + } + authHeaders() { + return this.cfg.apiKey ? { Authorization: `Bearer ${this.cfg.apiKey}` } : {}; + } +} +export default definePluginEntry({ + id: "athena-talk", + name: "Athena Local Talk", + description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", + register(api) { + api.registerRealtimeVoiceProvider({ + id: "athena-talk", + label: "Athena Local Talk", + aliases: ["athena", "local-athena"], + defaultModel: "athena-local", + voices: ["alloy", "claribel"], + autoSelectOrder: 1, + capabilities: { + transports: ["gateway-relay"], + inputAudioFormats: [AUDIO_FORMAT], + outputAudioFormats: [AUDIO_FORMAT], + supportsBargeIn: true, + handlesInputAudioBargeIn: true, + supportsToolCalls: true, + supportsSessionResumption: false, + }, + resolveConfig: ({ rawConfig }) => record(rawConfig), + isConfigured: ({ providerConfig, cfg }) => { + const raw = record(providerConfig); + const athena = record(record(record(cfg).models).providers).athena; + return Boolean(raw.baseUrl || record(athena).baseUrl); + }, + createBridge: (req) => new AthenaTalkBridge(req, resolveConfig(req)), + }); + }, +}); diff --git a/integrations/openclaw-athena-talk/index.ts b/integrations/openclaw-athena-talk/index.ts new file mode 100644 index 0000000..cbb2bfa --- /dev/null +++ b/integrations/openclaw-athena-talk/index.ts @@ -0,0 +1,309 @@ +import { randomUUID } from "node:crypto"; +import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry"; + +const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 } as const; + +type ProviderConfig = { + baseUrl?: string; + apiKey?: string; + voice?: string; + language?: string; + vadThreshold?: number; + silenceDurationMs?: number; + prefixPaddingMs?: number; + maxSpeechSeconds?: number; +}; + +function record(value: unknown): Record { + return value && typeof value === "object" && !Array.isArray(value) + ? value as Record + : {}; +} + +function resolveConfig(req: any): Required { + const raw = record(req.providerConfig); + const modelProviders = record(record(req.cfg).models).providers; + const athena = record(record(modelProviders).athena); + return { + baseUrl: String(raw.baseUrl || athena.baseUrl || "http://192.168.1.212:8081/v1").replace(/\/$/, ""), + apiKey: String(raw.apiKey || athena.apiKey || ""), + voice: String(raw.voice || req.voice || "alloy"), + language: String(raw.language || req.language || "de"), + vadThreshold: Number(raw.vadThreshold ?? 0.018), + silenceDurationMs: Number(raw.silenceDurationMs ?? 750), + prefixPaddingMs: Number(raw.prefixPaddingMs ?? 300), + maxSpeechSeconds: Number(raw.maxSpeechSeconds ?? 45), + }; +} + +function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer { + const header = Buffer.alloc(44); + header.write("RIFF", 0); + header.writeUInt32LE(36 + pcm.length, 4); + header.write("WAVEfmt ", 8); + header.writeUInt32LE(16, 16); + header.writeUInt16LE(1, 20); + header.writeUInt16LE(1, 22); + header.writeUInt32LE(sampleRate, 24); + header.writeUInt32LE(sampleRate * 2, 28); + header.writeUInt16LE(2, 32); + header.writeUInt16LE(16, 34); + header.write("data", 36); + header.writeUInt32LE(pcm.length, 40); + return Buffer.concat([header, pcm]); +} + +function pcmRms(pcm: Buffer): number { + if (pcm.length < 2) return 0; + let sum = 0; + const count = Math.floor(pcm.length / 2); + for (let i = 0; i < count; i += 1) { + const value = pcm.readInt16LE(i * 2) / 32768; + sum += value * value; + } + return Math.sqrt(sum / count); +} + +function readWavPcm24k(wav: Buffer): Buffer { + if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") { + throw new Error("Athena TTS did not return PCM WAV audio"); + } + let offset = 12; + let sampleRate = 0; + let channels = 0; + let bits = 0; + let data: Buffer | undefined; + while (offset + 8 <= wav.length) { + const id = wav.toString("ascii", offset, offset + 4); + const size = wav.readUInt32LE(offset + 4); + const start = offset + 8; + if (id === "fmt " && size >= 16) { + if (wav.readUInt16LE(start) !== 1) throw new Error("Athena TTS WAV is not PCM"); + channels = wav.readUInt16LE(start + 2); + sampleRate = wav.readUInt32LE(start + 4); + bits = wav.readUInt16LE(start + 14); + } else if (id === "data") { + data = wav.subarray(start, Math.min(start + size, wav.length)); + } + offset = start + size + (size % 2); + } + if (!data || !sampleRate || bits !== 16 || channels < 1) { + throw new Error("Unsupported Athena TTS WAV format"); + } + + const frames = Math.floor(data.length / (2 * channels)); + const mono = new Int16Array(frames); + for (let i = 0; i < frames; i += 1) mono[i] = data.readInt16LE(i * channels * 2); + if (sampleRate === 24000) return Buffer.from(mono.buffer); + + const outFrames = Math.max(1, Math.round(frames * 24000 / sampleRate)); + const out = Buffer.alloc(outFrames * 2); + for (let i = 0; i < outFrames; i += 1) { + const source = i * sampleRate / 24000; + const left = Math.min(frames - 1, Math.floor(source)); + const right = Math.min(frames - 1, left + 1); + const fraction = source - left; + const value = Math.round(mono[left] * (1 - fraction) + mono[right] * fraction); + out.writeInt16LE(Math.max(-32768, Math.min(32767, value)), i * 2); + } + return out; +} + +class AthenaTalkBridge { + readonly supportsToolResultContinuation = false; + readonly supportsToolResultSuppression = false; + private connected = false; + private closed = false; + private speaking = false; + private speech: Buffer[] = []; + private prefix: Buffer[] = []; + private prefixBytes = 0; + private silenceTimer?: NodeJS.Timeout; + private active?: AbortController; + private generation = 0; + + constructor(private readonly req: any, private readonly cfg: Required) {} + + async connect(): Promise { + this.connected = true; + this.req.onEvent?.({ direction: "server", type: "session.created" }); + this.req.onReady?.(); + } + + isConnected(): boolean { return this.connected && !this.closed; } + + setMediaTimestamp(_timestamp: number): void {} + + sendAudio(chunk: Buffer): void { + if (!this.isConnected() || chunk.length === 0) return; + const voiced = pcmRms(chunk) >= this.cfg.vadThreshold; + const maxPrefix = Math.round(24000 * 2 * this.cfg.prefixPaddingMs / 1000); + + if (!this.speaking) { + this.prefix.push(Buffer.from(chunk)); + this.prefixBytes += chunk.length; + while (this.prefixBytes > maxPrefix && this.prefix.length > 1) { + this.prefixBytes -= this.prefix.shift()!.length; + } + if (!voiced) return; + if (this.active) { + this.active.abort(); + this.active = undefined; + this.generation += 1; + this.req.onClearAudio?.("barge-in"); + } + this.speaking = true; + this.speech = this.prefix; + this.prefix = []; + this.prefixBytes = 0; + } else { + this.speech.push(Buffer.from(chunk)); + } + + const maxBytes = this.cfg.maxSpeechSeconds * 24000 * 2; + if (this.speech.reduce((sum, part) => sum + part.length, 0) >= maxBytes) { + void this.finishSpeech(); + return; + } + if (voiced) { + if (this.silenceTimer) clearTimeout(this.silenceTimer); + this.silenceTimer = setTimeout(() => void this.finishSpeech(), this.cfg.silenceDurationMs); + } + } + + sendUserMessage(text: string): void { + const trimmed = text.trim(); + if (trimmed) void this.answer(trimmed); + } + + triggerGreeting(instructions?: string): void { + void this.answer(instructions?.trim() || "Begrüße mich kurz auf Deutsch."); + } + + submitToolResult(_callId: string, _result: unknown): void {} + acknowledgeMark(_markName: string): void {} + + close(): void { + this.closed = true; + this.connected = false; + this.generation += 1; + this.active?.abort(); + if (this.silenceTimer) clearTimeout(this.silenceTimer); + this.req.onClose?.("completed"); + } + + private async finishSpeech(): Promise { + if (!this.speaking) return; + this.speaking = false; + if (this.silenceTimer) clearTimeout(this.silenceTimer); + this.silenceTimer = undefined; + const pcm = Buffer.concat(this.speech); + this.speech = []; + if (pcm.length < 24000 * 2 * 0.25) return; + + try { + const text = await this.transcribe(pcm); + if (!text || !this.isConnected()) return; + this.req.onTranscript?.("user", text, true); + await this.answer(text); + } catch (error) { + if ((error as Error).name !== "AbortError") this.req.onError?.(error as Error); + } + } + + private async transcribe(pcm: Buffer): Promise { + const form = new FormData(); + form.append("file", new Blob([wavFromPcm16(pcm)], { type: "audio/wav" }), "talk.wav"); + form.append("model", "whisper-1"); + form.append("language", this.cfg.language); + const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, { + method: "POST", + headers: this.authHeaders(), + body: form, + }); + if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`); + const payload = record(await response.json()); + return String(payload.text || "").trim(); + } + + private async answer(prompt: string): Promise { + if (!this.req.runAgentConsult) throw new Error("OpenClaw agent-consult is unavailable"); + this.active?.abort(); + const controller = new AbortController(); + this.active = controller; + const generation = ++this.generation; + const responseId = `athena-${randomUUID()}`; + try { + const result = await this.req.runAgentConsult({ prompt, signal: controller.signal }); + if (generation !== this.generation || !this.isConnected()) return; + const text = String(result?.text || "").trim(); + if (!text) return; + this.req.onTranscript?.("assistant", text, true); + const pcm = await this.synthesize(text, controller.signal); + if (generation !== this.generation || !this.isConnected()) return; + const frameBytes = 24000 * 2 / 50; + for (let offset = 0; offset < pcm.length; offset += frameBytes) { + this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length))); + } + this.req.onResponseDone?.({ status: "completed", responseId }); + this.req.onEvent?.({ direction: "server", type: "response.done", responseId }); + } catch (error) { + if ((error as Error).name === "AbortError") { + this.req.onResponseDone?.({ status: "cancelled", responseId }); + } else { + this.req.onError?.(error as Error); + this.req.onResponseDone?.({ status: "failed", responseId, error: String(error) }); + } + } finally { + if (this.active === controller) this.active = undefined; + } + } + + private async synthesize(text: string, signal: AbortSignal): Promise { + const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, { + method: "POST", + headers: { "Content-Type": "application/json", ...this.authHeaders() }, + // Let Athena select its currently active local backend (XTTS or Piper). + body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }), + signal, + }); + if (!response.ok) throw new Error(`Athena TTS failed (HTTP ${response.status})`); + return readWavPcm24k(Buffer.from(await response.arrayBuffer())); + } + + private authHeaders(): Record { + return this.cfg.apiKey ? { Authorization: `Bearer ${this.cfg.apiKey}` } : {}; + } +} + +export default definePluginEntry({ + id: "athena-talk", + name: "Athena Local Talk", + description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.", + register(api) { + api.registerRealtimeVoiceProvider({ + id: "athena-talk", + label: "Athena Local Talk", + aliases: ["athena", "local-athena"], + defaultModel: "athena-local", + voices: ["alloy", "claribel"], + autoSelectOrder: 1, + capabilities: { + transports: ["gateway-relay"], + inputAudioFormats: [AUDIO_FORMAT], + outputAudioFormats: [AUDIO_FORMAT], + supportsBargeIn: true, + handlesInputAudioBargeIn: true, + supportsToolCalls: true, + supportsSessionResumption: false, + }, + resolveConfig: ({ rawConfig }) => record(rawConfig), + isConfigured: ({ providerConfig, cfg }) => { + const raw = record(providerConfig); + const athena = record(record(record(cfg).models).providers).athena; + return Boolean(raw.baseUrl || record(athena).baseUrl); + }, + createBridge: (req) => new AthenaTalkBridge(req, resolveConfig(req)), + } as any); + }, +}); diff --git a/integrations/openclaw-athena-talk/openclaw.plugin.json b/integrations/openclaw-athena-talk/openclaw.plugin.json new file mode 100644 index 0000000..c5e8a49 --- /dev/null +++ b/integrations/openclaw-athena-talk/openclaw.plugin.json @@ -0,0 +1,23 @@ +{ + "id": "athena-talk", + "name": "Athena Local Talk", + "description": "Local German voice conversations through Athena without a public speech provider.", + "version": "0.1.0", + "enabledByDefault": true, + "activation": { + "onStartup": true, + "capabilities": [ + "provider" + ] + }, + "contracts": { + "realtimeVoiceProviders": [ + "athena-talk" + ] + }, + "configSchema": { + "type": "object", + "additionalProperties": false, + "properties": {} + } +} diff --git a/integrations/openclaw-athena-talk/package.json b/integrations/openclaw-athena-talk/package.json new file mode 100644 index 0000000..8333bc9 --- /dev/null +++ b/integrations/openclaw-athena-talk/package.json @@ -0,0 +1,23 @@ +{ + "name": "openclaw-plugin-athena-talk", + "version": "0.1.0", + "description": "Private realtime Talk bridge for Athena STT, OpenClaw agent consult, and Athena TTS.", + "type": "module", + "private": true, + "peerDependencies": { + "openclaw": ">=2026.8.2" + }, + "peerDependenciesMeta": { + "openclaw": { + "optional": true + } + }, + "openclaw": { + "extensions": [ + "./dist/index.js" + ], + "compat": { + "pluginApi": ">=2026.8.2" + } + } +} diff --git a/platform/docker/whisper/Dockerfile b/platform/docker/whisper/Dockerfile index 8c2925f..9268e8c 100644 --- a/platform/docker/whisper/Dockerfile +++ b/platform/docker/whisper/Dockerfile @@ -11,7 +11,7 @@ RUN apt-get update \ -DCMAKE_BUILD_TYPE=Release \ -DGGML_NATIVE=ON \ -DWHISPER_BUILD_TESTS=OFF \ - -DWHISPER_BUILD_SERVER=OFF \ + -DWHISPER_BUILD_SERVER=ON \ && cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \ && apt-get purge -y --auto-remove build-essential cmake git \ && rm -rf /var/lib/apt/lists/* /root/.cache \ @@ -25,7 +25,8 @@ RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint ENV WHISPER_HOST=0.0.0.0 \ WHISPER_PORT=8084 \ WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \ - WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \ + WHISPER_MODEL=/models/ggml-small.bin \ + WHISPER_SERVER_URL=http://127.0.0.1:8085 \ WHISPER_THREADS=8 \ WHISPER_LANGUAGE=de \ FFMPEG_BIN=/usr/bin/ffmpeg diff --git a/platform/docker/whisper/entrypoint.sh b/platform/docker/whisper/entrypoint.sh index 2bf53eb..25cfcd6 100644 --- a/platform/docker/whisper/entrypoint.sh +++ b/platform/docker/whisper/entrypoint.sh @@ -1,8 +1,8 @@ #!/bin/sh set -eu -model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin} -model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin} +model=${WHISPER_MODEL:-/models/ggml-small.bin} +model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin} model_dir=$(dirname "$model") mkdir -p "$model_dir" @@ -22,4 +22,44 @@ if [ ! -s "$model" ]; then gosu whisper mv "$partial" "$model" fi -exec gosu whisper python /app/stt_worker.py +server_port=${WHISPER_SERVER_PORT:-8085} +server_log=/tmp/whisper-server.log + +gosu whisper /opt/whisper.cpp/build/bin/whisper-server \ + --model "$model" \ + --threads "${WHISPER_THREADS:-8}" \ + --language "${WHISPER_LANGUAGE:-de}" \ + --host 127.0.0.1 \ + --port "$server_port" \ + --inference-path /inference \ + --no-gpu >"$server_log" 2>&1 & + +server_pid=$! +worker_pid="" +cleanup() { + [ -z "$worker_pid" ] || kill "$worker_pid" 2>/dev/null || true + kill "$server_pid" 2>/dev/null || true +} +trap cleanup EXIT INT TERM + +# Der Modell-Load dauert beim Containerstart kurz. Der HTTP-Wrapper wird erst +# freigegeben, wenn whisper-server tatsächlich Antworten annimmt. +i=0 +until curl --fail --silent "http://127.0.0.1:${server_port}/" >/dev/null; do + i=$((i + 1)) + if ! kill -0 "$server_pid" 2>/dev/null; then + cat "$server_log" + exit 1 + fi + if [ "$i" -ge 180 ]; then + echo "whisper-server did not become ready" >&2 + cat "$server_log" + exit 1 + fi + sleep 1 +done + +export WHISPER_SERVER_URL="http://127.0.0.1:${server_port}" +gosu whisper python /app/stt_worker.py & +worker_pid=$! +wait "$worker_pid" diff --git a/router/stt_worker.py b/router/stt_worker.py index 2cc856c..60e6c88 100644 --- a/router/stt_worker.py +++ b/router/stt_worker.py @@ -3,7 +3,8 @@ STT-Worker – langlebiger Whisper-Transkriptions-Service (CPU-only). Liest Audio-Dateien (WAV, MP3, OGG, FLAC, WebM/Opus via ffmpeg), -transkribiert sie mit whisper.cpp (whisper-cli) und liefert JSON-Text. +transkribiert sie mit einem dauerhaft geladenen whisper.cpp-Server (mit +whisper-cli als Rückfallweg) und liefert JSON-Text. Konfiguration über Umgebungsvariablen: WHISPER_HOST Bind-Adresse (Default: 127.0.0.1) @@ -28,6 +29,7 @@ import sys import tempfile import time import uuid +import urllib.request from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer # --------------------------------------------------------------------------- @@ -45,6 +47,7 @@ WHISPER_MODEL = os.environ.get( ) WHISPER_THREADS = int(os.environ.get("WHISPER_THREADS", "8")) WHISPER_LANGUAGE = os.environ.get("WHISPER_LANGUAGE", "de") +WHISPER_SERVER_URL = os.environ.get("WHISPER_SERVER_URL", "").rstrip("/") FFMPEG_BIN = os.environ.get("FFMPEG_BIN", "/usr/bin/ffmpeg") LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO") @@ -139,13 +142,22 @@ def transcribe( temperature: float | None = None, ) -> dict: """ - Führt die Transkription mit whisper-cli aus. + Führt die Transkription bevorzugt über den persistenten whisper.cpp- + Server aus. Dadurch wird das Modell nicht pro Aufnahme neu geladen. Liefert dict mit 'text' und Metadaten. """ lang = language or WHISPER_LANGUAGE if lang == "auto": lang = "auto" + if WHISPER_SERVER_URL: + return _transcribe_via_server( + audio_path, + language=lang, + prompt=prompt, + temperature=temperature, + ) + out_prefix = f"/tmp/stt_{uuid.uuid4().hex[:12]}" out_json = out_prefix + ".json" @@ -217,6 +229,65 @@ def transcribe( return result +def _transcribe_via_server( + audio_path: str, + language: str, + prompt: str | None = None, + temperature: float | None = None, +) -> dict: + """Sendet eine Aufnahme an den bereits geladenen whisper-server.""" + boundary = f"----athena-whisper-{uuid.uuid4().hex}" + chunks: list[bytes] = [] + + def add_field(name: str, value: str) -> None: + chunks.extend([ + f"--{boundary}\r\n".encode(), + f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(), + value.encode("utf-8"), + b"\r\n", + ]) + + with open(audio_path, "rb") as f: + audio = f.read() + chunks.extend([ + f"--{boundary}\r\n".encode(), + b'Content-Disposition: form-data; name="file"; filename="audio.wav"\r\n', + b"Content-Type: audio/wav\r\n\r\n", + audio, + b"\r\n", + ]) + add_field("response_format", "json") + add_field("language", language) + if prompt: + add_field("prompt", prompt) + if temperature is not None: + add_field("temperature", str(temperature)) + chunks.append(f"--{boundary}--\r\n".encode()) + + request = urllib.request.Request( + f"{WHISPER_SERVER_URL}/inference", + data=b"".join(chunks), + headers={"Content-Type": f"multipart/form-data; boundary={boundary}"}, + method="POST", + ) + t0 = time.monotonic() + with urllib.request.urlopen(request, timeout=300) as response: + payload = json.loads(response.read().decode("utf-8")) + elapsed = time.monotonic() - t0 + text = payload.get("text", "") if isinstance(payload, dict) else "" + result = { + "text": text.strip(), + "language": language, + "duration_ms": int(elapsed * 1000), + "engine": "whisper-server", + } + log.info( + "Transkription (persistent): %d ms, %d Zeichen, Sprache=%s", + result["duration_ms"], len(result["text"]), language, + ) + return result + + # --------------------------------------------------------------------------- # HTTP-Handler # --------------------------------------------------------------------------- @@ -300,6 +371,8 @@ class STTHandler(BaseHTTPRequestHandler): "whisper_cli_exists": cli_ok, "threads": WHISPER_THREADS, "language": WHISPER_LANGUAGE, + "server_url": WHISPER_SERVER_URL or None, + "persistent_model": bool(WHISPER_SERVER_URL), "ffmpeg": FFMPEG_BIN, "ffmpeg_exists": os.path.isfile(FFMPEG_BIN), })