Add private realtime Talk for OpenClaw
This commit is contained in:
@@ -103,7 +103,13 @@ Standard, bis Hermes' Sitzungsfehler behoben ist.
|
|||||||
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
||||||
`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das
|
`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das
|
||||||
Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
|
Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
|
||||||
werden lokal auf Athena verarbeitet.
|
werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale
|
||||||
|
Realtime-Provider unter
|
||||||
|
[`integrations/openclaw-athena-talk`](integrations/openclaw-athena-talk). Er
|
||||||
|
verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives
|
||||||
|
Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten
|
||||||
|
bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis
|
||||||
|
und bleibt deshalb bei normalen Container-Updates bestehen.
|
||||||
- Portainer: `https://192.168.1.212:9443`
|
- Portainer: `https://192.168.1.212:9443`
|
||||||
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
||||||
|
|
||||||
|
|||||||
+5
-3
@@ -732,14 +732,16 @@ services:
|
|||||||
WHISPER_HOST: 0.0.0.0
|
WHISPER_HOST: 0.0.0.0
|
||||||
WHISPER_PORT: "8084"
|
WHISPER_PORT: "8084"
|
||||||
WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
|
WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
|
||||||
WHISPER_MODEL: /models/ggml-large-v3-turbo.bin
|
WHISPER_MODEL: /models/ggml-small.bin
|
||||||
WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin}
|
||||||
|
WHISPER_SERVER_PORT: "8085"
|
||||||
WHISPER_THREADS: ${WHISPER_THREADS:-8}
|
WHISPER_THREADS: ${WHISPER_THREADS:-8}
|
||||||
WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
|
WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
|
||||||
networks: [frontend, inference]
|
networks: [frontend, inference]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
cap_drop: [ALL]
|
cap_drop: [ALL]
|
||||||
cap_add: [CHOWN, SETUID, SETGID]
|
# The entrypoint supervises whisper-server after dropping it to uid 10004.
|
||||||
|
cap_add: [CHOWN, SETUID, SETGID, KILL]
|
||||||
healthcheck:
|
healthcheck:
|
||||||
test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
|
||||||
interval: 10s
|
interval: 10s
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
# Athena Local Talk for OpenClaw
|
||||||
|
|
||||||
|
This private OpenClaw provider connects browser/Desktop Talk to the existing
|
||||||
|
Athena speech stack:
|
||||||
|
|
||||||
|
1. local VAD collects a spoken utterance,
|
||||||
|
2. Athena Whisper transcribes it,
|
||||||
|
3. OpenClaw's normal agent-consult path answers with its configured model and
|
||||||
|
tools,
|
||||||
|
4. Athena XTTS/Piper returns PCM audio to the Talk client.
|
||||||
|
|
||||||
|
No public speech provider is used. The provider reuses the already configured
|
||||||
|
`models.providers.athena` base URL and API key; no second key copy is required.
|
||||||
|
|
||||||
|
Recommended `talk.realtime` configuration:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"provider": "athena-talk",
|
||||||
|
"model": "athena-local",
|
||||||
|
"speakerVoice": "alloy",
|
||||||
|
"language": "de",
|
||||||
|
"mode": "realtime",
|
||||||
|
"transport": "gateway-relay",
|
||||||
|
"brain": "agent-consult",
|
||||||
|
"providers": {
|
||||||
|
"athena-talk": {
|
||||||
|
"vadThreshold": 0.018,
|
||||||
|
"silenceDurationMs": 750,
|
||||||
|
"prefixPaddingMs": 300,
|
||||||
|
"maxSpeechSeconds": 45
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Install from the OpenClaw container with `openclaw plugins install <path>` and
|
||||||
|
restart the gateway once. The plugin is stored in OpenClaw's persistent data
|
||||||
|
directory, so normal image updates do not remove it.
|
||||||
+302
@@ -0,0 +1,302 @@
|
|||||||
|
import { randomUUID } from "node:crypto";
|
||||||
|
import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry";
|
||||||
|
const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 };
|
||||||
|
function record(value) {
|
||||||
|
return value && typeof value === "object" && !Array.isArray(value)
|
||||||
|
? value
|
||||||
|
: {};
|
||||||
|
}
|
||||||
|
function resolveConfig(req) {
|
||||||
|
const raw = record(req.providerConfig);
|
||||||
|
const modelProviders = record(record(req.cfg).models).providers;
|
||||||
|
const athena = record(record(modelProviders).athena);
|
||||||
|
return {
|
||||||
|
baseUrl: String(raw.baseUrl || athena.baseUrl || "http://192.168.1.212:8081/v1").replace(/\/$/, ""),
|
||||||
|
apiKey: String(raw.apiKey || athena.apiKey || ""),
|
||||||
|
voice: String(raw.voice || req.voice || "alloy"),
|
||||||
|
language: String(raw.language || req.language || "de"),
|
||||||
|
vadThreshold: Number(raw.vadThreshold ?? 0.018),
|
||||||
|
silenceDurationMs: Number(raw.silenceDurationMs ?? 750),
|
||||||
|
prefixPaddingMs: Number(raw.prefixPaddingMs ?? 300),
|
||||||
|
maxSpeechSeconds: Number(raw.maxSpeechSeconds ?? 45),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
function wavFromPcm16(pcm, sampleRate = 24000) {
|
||||||
|
const header = Buffer.alloc(44);
|
||||||
|
header.write("RIFF", 0);
|
||||||
|
header.writeUInt32LE(36 + pcm.length, 4);
|
||||||
|
header.write("WAVEfmt ", 8);
|
||||||
|
header.writeUInt32LE(16, 16);
|
||||||
|
header.writeUInt16LE(1, 20);
|
||||||
|
header.writeUInt16LE(1, 22);
|
||||||
|
header.writeUInt32LE(sampleRate, 24);
|
||||||
|
header.writeUInt32LE(sampleRate * 2, 28);
|
||||||
|
header.writeUInt16LE(2, 32);
|
||||||
|
header.writeUInt16LE(16, 34);
|
||||||
|
header.write("data", 36);
|
||||||
|
header.writeUInt32LE(pcm.length, 40);
|
||||||
|
return Buffer.concat([header, pcm]);
|
||||||
|
}
|
||||||
|
function pcmRms(pcm) {
|
||||||
|
if (pcm.length < 2)
|
||||||
|
return 0;
|
||||||
|
let sum = 0;
|
||||||
|
const count = Math.floor(pcm.length / 2);
|
||||||
|
for (let i = 0; i < count; i += 1) {
|
||||||
|
const value = pcm.readInt16LE(i * 2) / 32768;
|
||||||
|
sum += value * value;
|
||||||
|
}
|
||||||
|
return Math.sqrt(sum / count);
|
||||||
|
}
|
||||||
|
function readWavPcm24k(wav) {
|
||||||
|
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
||||||
|
throw new Error("Athena TTS did not return PCM WAV audio");
|
||||||
|
}
|
||||||
|
let offset = 12;
|
||||||
|
let sampleRate = 0;
|
||||||
|
let channels = 0;
|
||||||
|
let bits = 0;
|
||||||
|
let data;
|
||||||
|
while (offset + 8 <= wav.length) {
|
||||||
|
const id = wav.toString("ascii", offset, offset + 4);
|
||||||
|
const size = wav.readUInt32LE(offset + 4);
|
||||||
|
const start = offset + 8;
|
||||||
|
if (id === "fmt " && size >= 16) {
|
||||||
|
if (wav.readUInt16LE(start) !== 1)
|
||||||
|
throw new Error("Athena TTS WAV is not PCM");
|
||||||
|
channels = wav.readUInt16LE(start + 2);
|
||||||
|
sampleRate = wav.readUInt32LE(start + 4);
|
||||||
|
bits = wav.readUInt16LE(start + 14);
|
||||||
|
}
|
||||||
|
else if (id === "data") {
|
||||||
|
data = wav.subarray(start, Math.min(start + size, wav.length));
|
||||||
|
}
|
||||||
|
offset = start + size + (size % 2);
|
||||||
|
}
|
||||||
|
if (!data || !sampleRate || bits !== 16 || channels < 1) {
|
||||||
|
throw new Error("Unsupported Athena TTS WAV format");
|
||||||
|
}
|
||||||
|
const frames = Math.floor(data.length / (2 * channels));
|
||||||
|
const mono = new Int16Array(frames);
|
||||||
|
for (let i = 0; i < frames; i += 1)
|
||||||
|
mono[i] = data.readInt16LE(i * channels * 2);
|
||||||
|
if (sampleRate === 24000)
|
||||||
|
return Buffer.from(mono.buffer);
|
||||||
|
const outFrames = Math.max(1, Math.round(frames * 24000 / sampleRate));
|
||||||
|
const out = Buffer.alloc(outFrames * 2);
|
||||||
|
for (let i = 0; i < outFrames; i += 1) {
|
||||||
|
const source = i * sampleRate / 24000;
|
||||||
|
const left = Math.min(frames - 1, Math.floor(source));
|
||||||
|
const right = Math.min(frames - 1, left + 1);
|
||||||
|
const fraction = source - left;
|
||||||
|
const value = Math.round(mono[left] * (1 - fraction) + mono[right] * fraction);
|
||||||
|
out.writeInt16LE(Math.max(-32768, Math.min(32767, value)), i * 2);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
class AthenaTalkBridge {
|
||||||
|
req;
|
||||||
|
cfg;
|
||||||
|
supportsToolResultContinuation = false;
|
||||||
|
supportsToolResultSuppression = false;
|
||||||
|
connected = false;
|
||||||
|
closed = false;
|
||||||
|
speaking = false;
|
||||||
|
speech = [];
|
||||||
|
prefix = [];
|
||||||
|
prefixBytes = 0;
|
||||||
|
silenceTimer;
|
||||||
|
active;
|
||||||
|
generation = 0;
|
||||||
|
constructor(req, cfg) {
|
||||||
|
this.req = req;
|
||||||
|
this.cfg = cfg;
|
||||||
|
}
|
||||||
|
async connect() {
|
||||||
|
this.connected = true;
|
||||||
|
this.req.onEvent?.({ direction: "server", type: "session.created" });
|
||||||
|
this.req.onReady?.();
|
||||||
|
}
|
||||||
|
isConnected() { return this.connected && !this.closed; }
|
||||||
|
setMediaTimestamp(_timestamp) { }
|
||||||
|
sendAudio(chunk) {
|
||||||
|
if (!this.isConnected() || chunk.length === 0)
|
||||||
|
return;
|
||||||
|
const voiced = pcmRms(chunk) >= this.cfg.vadThreshold;
|
||||||
|
const maxPrefix = Math.round(24000 * 2 * this.cfg.prefixPaddingMs / 1000);
|
||||||
|
if (!this.speaking) {
|
||||||
|
this.prefix.push(Buffer.from(chunk));
|
||||||
|
this.prefixBytes += chunk.length;
|
||||||
|
while (this.prefixBytes > maxPrefix && this.prefix.length > 1) {
|
||||||
|
this.prefixBytes -= this.prefix.shift().length;
|
||||||
|
}
|
||||||
|
if (!voiced)
|
||||||
|
return;
|
||||||
|
if (this.active) {
|
||||||
|
this.active.abort();
|
||||||
|
this.active = undefined;
|
||||||
|
this.generation += 1;
|
||||||
|
this.req.onClearAudio?.("barge-in");
|
||||||
|
}
|
||||||
|
this.speaking = true;
|
||||||
|
this.speech = this.prefix;
|
||||||
|
this.prefix = [];
|
||||||
|
this.prefixBytes = 0;
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
this.speech.push(Buffer.from(chunk));
|
||||||
|
}
|
||||||
|
const maxBytes = this.cfg.maxSpeechSeconds * 24000 * 2;
|
||||||
|
if (this.speech.reduce((sum, part) => sum + part.length, 0) >= maxBytes) {
|
||||||
|
void this.finishSpeech();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (voiced) {
|
||||||
|
if (this.silenceTimer)
|
||||||
|
clearTimeout(this.silenceTimer);
|
||||||
|
this.silenceTimer = setTimeout(() => void this.finishSpeech(), this.cfg.silenceDurationMs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sendUserMessage(text) {
|
||||||
|
const trimmed = text.trim();
|
||||||
|
if (trimmed)
|
||||||
|
void this.answer(trimmed);
|
||||||
|
}
|
||||||
|
triggerGreeting(instructions) {
|
||||||
|
void this.answer(instructions?.trim() || "Begrüße mich kurz auf Deutsch.");
|
||||||
|
}
|
||||||
|
submitToolResult(_callId, _result) { }
|
||||||
|
acknowledgeMark(_markName) { }
|
||||||
|
close() {
|
||||||
|
this.closed = true;
|
||||||
|
this.connected = false;
|
||||||
|
this.generation += 1;
|
||||||
|
this.active?.abort();
|
||||||
|
if (this.silenceTimer)
|
||||||
|
clearTimeout(this.silenceTimer);
|
||||||
|
this.req.onClose?.("completed");
|
||||||
|
}
|
||||||
|
async finishSpeech() {
|
||||||
|
if (!this.speaking)
|
||||||
|
return;
|
||||||
|
this.speaking = false;
|
||||||
|
if (this.silenceTimer)
|
||||||
|
clearTimeout(this.silenceTimer);
|
||||||
|
this.silenceTimer = undefined;
|
||||||
|
const pcm = Buffer.concat(this.speech);
|
||||||
|
this.speech = [];
|
||||||
|
if (pcm.length < 24000 * 2 * 0.25)
|
||||||
|
return;
|
||||||
|
try {
|
||||||
|
const text = await this.transcribe(pcm);
|
||||||
|
if (!text || !this.isConnected())
|
||||||
|
return;
|
||||||
|
this.req.onTranscript?.("user", text, true);
|
||||||
|
await this.answer(text);
|
||||||
|
}
|
||||||
|
catch (error) {
|
||||||
|
if (error.name !== "AbortError")
|
||||||
|
this.req.onError?.(error);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
async transcribe(pcm) {
|
||||||
|
const form = new FormData();
|
||||||
|
form.append("file", new Blob([wavFromPcm16(pcm)], { type: "audio/wav" }), "talk.wav");
|
||||||
|
form.append("model", "whisper-1");
|
||||||
|
form.append("language", this.cfg.language);
|
||||||
|
const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: this.authHeaders(),
|
||||||
|
body: form,
|
||||||
|
});
|
||||||
|
if (!response.ok)
|
||||||
|
throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||||
|
const payload = record(await response.json());
|
||||||
|
return String(payload.text || "").trim();
|
||||||
|
}
|
||||||
|
async answer(prompt) {
|
||||||
|
if (!this.req.runAgentConsult)
|
||||||
|
throw new Error("OpenClaw agent-consult is unavailable");
|
||||||
|
this.active?.abort();
|
||||||
|
const controller = new AbortController();
|
||||||
|
this.active = controller;
|
||||||
|
const generation = ++this.generation;
|
||||||
|
const responseId = `athena-${randomUUID()}`;
|
||||||
|
try {
|
||||||
|
const result = await this.req.runAgentConsult({ prompt, signal: controller.signal });
|
||||||
|
if (generation !== this.generation || !this.isConnected())
|
||||||
|
return;
|
||||||
|
const text = String(result?.text || "").trim();
|
||||||
|
if (!text)
|
||||||
|
return;
|
||||||
|
this.req.onTranscript?.("assistant", text, true);
|
||||||
|
const pcm = await this.synthesize(text, controller.signal);
|
||||||
|
if (generation !== this.generation || !this.isConnected())
|
||||||
|
return;
|
||||||
|
const frameBytes = 24000 * 2 / 50;
|
||||||
|
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
||||||
|
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
||||||
|
}
|
||||||
|
this.req.onResponseDone?.({ status: "completed", responseId });
|
||||||
|
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
||||||
|
}
|
||||||
|
catch (error) {
|
||||||
|
if (error.name === "AbortError") {
|
||||||
|
this.req.onResponseDone?.({ status: "cancelled", responseId });
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
this.req.onError?.(error);
|
||||||
|
this.req.onResponseDone?.({ status: "failed", responseId, error: String(error) });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
finally {
|
||||||
|
if (this.active === controller)
|
||||||
|
this.active = undefined;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
async synthesize(text, signal) {
|
||||||
|
const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: { "Content-Type": "application/json", ...this.authHeaders() },
|
||||||
|
body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }),
|
||||||
|
signal,
|
||||||
|
});
|
||||||
|
if (!response.ok)
|
||||||
|
throw new Error(`Athena TTS failed (HTTP ${response.status})`);
|
||||||
|
return readWavPcm24k(Buffer.from(await response.arrayBuffer()));
|
||||||
|
}
|
||||||
|
authHeaders() {
|
||||||
|
return this.cfg.apiKey ? { Authorization: `Bearer ${this.cfg.apiKey}` } : {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
export default definePluginEntry({
|
||||||
|
id: "athena-talk",
|
||||||
|
name: "Athena Local Talk",
|
||||||
|
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
|
||||||
|
register(api) {
|
||||||
|
api.registerRealtimeVoiceProvider({
|
||||||
|
id: "athena-talk",
|
||||||
|
label: "Athena Local Talk",
|
||||||
|
aliases: ["athena", "local-athena"],
|
||||||
|
defaultModel: "athena-local",
|
||||||
|
voices: ["alloy", "claribel"],
|
||||||
|
autoSelectOrder: 1,
|
||||||
|
capabilities: {
|
||||||
|
transports: ["gateway-relay"],
|
||||||
|
inputAudioFormats: [AUDIO_FORMAT],
|
||||||
|
outputAudioFormats: [AUDIO_FORMAT],
|
||||||
|
supportsBargeIn: true,
|
||||||
|
handlesInputAudioBargeIn: true,
|
||||||
|
supportsToolCalls: true,
|
||||||
|
supportsSessionResumption: false,
|
||||||
|
},
|
||||||
|
resolveConfig: ({ rawConfig }) => record(rawConfig),
|
||||||
|
isConfigured: ({ providerConfig, cfg }) => {
|
||||||
|
const raw = record(providerConfig);
|
||||||
|
const athena = record(record(record(cfg).models).providers).athena;
|
||||||
|
return Boolean(raw.baseUrl || record(athena).baseUrl);
|
||||||
|
},
|
||||||
|
createBridge: (req) => new AthenaTalkBridge(req, resolveConfig(req)),
|
||||||
|
});
|
||||||
|
},
|
||||||
|
});
|
||||||
@@ -0,0 +1,309 @@
|
|||||||
|
import { randomUUID } from "node:crypto";
|
||||||
|
import { definePluginEntry } from "openclaw/plugin-sdk/plugin-entry";
|
||||||
|
|
||||||
|
const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 } as const;
|
||||||
|
|
||||||
|
type ProviderConfig = {
|
||||||
|
baseUrl?: string;
|
||||||
|
apiKey?: string;
|
||||||
|
voice?: string;
|
||||||
|
language?: string;
|
||||||
|
vadThreshold?: number;
|
||||||
|
silenceDurationMs?: number;
|
||||||
|
prefixPaddingMs?: number;
|
||||||
|
maxSpeechSeconds?: number;
|
||||||
|
};
|
||||||
|
|
||||||
|
function record(value: unknown): Record<string, any> {
|
||||||
|
return value && typeof value === "object" && !Array.isArray(value)
|
||||||
|
? value as Record<string, any>
|
||||||
|
: {};
|
||||||
|
}
|
||||||
|
|
||||||
|
function resolveConfig(req: any): Required<ProviderConfig> {
|
||||||
|
const raw = record(req.providerConfig);
|
||||||
|
const modelProviders = record(record(req.cfg).models).providers;
|
||||||
|
const athena = record(record(modelProviders).athena);
|
||||||
|
return {
|
||||||
|
baseUrl: String(raw.baseUrl || athena.baseUrl || "http://192.168.1.212:8081/v1").replace(/\/$/, ""),
|
||||||
|
apiKey: String(raw.apiKey || athena.apiKey || ""),
|
||||||
|
voice: String(raw.voice || req.voice || "alloy"),
|
||||||
|
language: String(raw.language || req.language || "de"),
|
||||||
|
vadThreshold: Number(raw.vadThreshold ?? 0.018),
|
||||||
|
silenceDurationMs: Number(raw.silenceDurationMs ?? 750),
|
||||||
|
prefixPaddingMs: Number(raw.prefixPaddingMs ?? 300),
|
||||||
|
maxSpeechSeconds: Number(raw.maxSpeechSeconds ?? 45),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer {
|
||||||
|
const header = Buffer.alloc(44);
|
||||||
|
header.write("RIFF", 0);
|
||||||
|
header.writeUInt32LE(36 + pcm.length, 4);
|
||||||
|
header.write("WAVEfmt ", 8);
|
||||||
|
header.writeUInt32LE(16, 16);
|
||||||
|
header.writeUInt16LE(1, 20);
|
||||||
|
header.writeUInt16LE(1, 22);
|
||||||
|
header.writeUInt32LE(sampleRate, 24);
|
||||||
|
header.writeUInt32LE(sampleRate * 2, 28);
|
||||||
|
header.writeUInt16LE(2, 32);
|
||||||
|
header.writeUInt16LE(16, 34);
|
||||||
|
header.write("data", 36);
|
||||||
|
header.writeUInt32LE(pcm.length, 40);
|
||||||
|
return Buffer.concat([header, pcm]);
|
||||||
|
}
|
||||||
|
|
||||||
|
function pcmRms(pcm: Buffer): number {
|
||||||
|
if (pcm.length < 2) return 0;
|
||||||
|
let sum = 0;
|
||||||
|
const count = Math.floor(pcm.length / 2);
|
||||||
|
for (let i = 0; i < count; i += 1) {
|
||||||
|
const value = pcm.readInt16LE(i * 2) / 32768;
|
||||||
|
sum += value * value;
|
||||||
|
}
|
||||||
|
return Math.sqrt(sum / count);
|
||||||
|
}
|
||||||
|
|
||||||
|
function readWavPcm24k(wav: Buffer): Buffer {
|
||||||
|
if (wav.length < 44 || wav.toString("ascii", 0, 4) !== "RIFF") {
|
||||||
|
throw new Error("Athena TTS did not return PCM WAV audio");
|
||||||
|
}
|
||||||
|
let offset = 12;
|
||||||
|
let sampleRate = 0;
|
||||||
|
let channels = 0;
|
||||||
|
let bits = 0;
|
||||||
|
let data: Buffer | undefined;
|
||||||
|
while (offset + 8 <= wav.length) {
|
||||||
|
const id = wav.toString("ascii", offset, offset + 4);
|
||||||
|
const size = wav.readUInt32LE(offset + 4);
|
||||||
|
const start = offset + 8;
|
||||||
|
if (id === "fmt " && size >= 16) {
|
||||||
|
if (wav.readUInt16LE(start) !== 1) throw new Error("Athena TTS WAV is not PCM");
|
||||||
|
channels = wav.readUInt16LE(start + 2);
|
||||||
|
sampleRate = wav.readUInt32LE(start + 4);
|
||||||
|
bits = wav.readUInt16LE(start + 14);
|
||||||
|
} else if (id === "data") {
|
||||||
|
data = wav.subarray(start, Math.min(start + size, wav.length));
|
||||||
|
}
|
||||||
|
offset = start + size + (size % 2);
|
||||||
|
}
|
||||||
|
if (!data || !sampleRate || bits !== 16 || channels < 1) {
|
||||||
|
throw new Error("Unsupported Athena TTS WAV format");
|
||||||
|
}
|
||||||
|
|
||||||
|
const frames = Math.floor(data.length / (2 * channels));
|
||||||
|
const mono = new Int16Array(frames);
|
||||||
|
for (let i = 0; i < frames; i += 1) mono[i] = data.readInt16LE(i * channels * 2);
|
||||||
|
if (sampleRate === 24000) return Buffer.from(mono.buffer);
|
||||||
|
|
||||||
|
const outFrames = Math.max(1, Math.round(frames * 24000 / sampleRate));
|
||||||
|
const out = Buffer.alloc(outFrames * 2);
|
||||||
|
for (let i = 0; i < outFrames; i += 1) {
|
||||||
|
const source = i * sampleRate / 24000;
|
||||||
|
const left = Math.min(frames - 1, Math.floor(source));
|
||||||
|
const right = Math.min(frames - 1, left + 1);
|
||||||
|
const fraction = source - left;
|
||||||
|
const value = Math.round(mono[left] * (1 - fraction) + mono[right] * fraction);
|
||||||
|
out.writeInt16LE(Math.max(-32768, Math.min(32767, value)), i * 2);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
class AthenaTalkBridge {
|
||||||
|
readonly supportsToolResultContinuation = false;
|
||||||
|
readonly supportsToolResultSuppression = false;
|
||||||
|
private connected = false;
|
||||||
|
private closed = false;
|
||||||
|
private speaking = false;
|
||||||
|
private speech: Buffer[] = [];
|
||||||
|
private prefix: Buffer[] = [];
|
||||||
|
private prefixBytes = 0;
|
||||||
|
private silenceTimer?: NodeJS.Timeout;
|
||||||
|
private active?: AbortController;
|
||||||
|
private generation = 0;
|
||||||
|
|
||||||
|
constructor(private readonly req: any, private readonly cfg: Required<ProviderConfig>) {}
|
||||||
|
|
||||||
|
async connect(): Promise<void> {
|
||||||
|
this.connected = true;
|
||||||
|
this.req.onEvent?.({ direction: "server", type: "session.created" });
|
||||||
|
this.req.onReady?.();
|
||||||
|
}
|
||||||
|
|
||||||
|
isConnected(): boolean { return this.connected && !this.closed; }
|
||||||
|
|
||||||
|
setMediaTimestamp(_timestamp: number): void {}
|
||||||
|
|
||||||
|
sendAudio(chunk: Buffer): void {
|
||||||
|
if (!this.isConnected() || chunk.length === 0) return;
|
||||||
|
const voiced = pcmRms(chunk) >= this.cfg.vadThreshold;
|
||||||
|
const maxPrefix = Math.round(24000 * 2 * this.cfg.prefixPaddingMs / 1000);
|
||||||
|
|
||||||
|
if (!this.speaking) {
|
||||||
|
this.prefix.push(Buffer.from(chunk));
|
||||||
|
this.prefixBytes += chunk.length;
|
||||||
|
while (this.prefixBytes > maxPrefix && this.prefix.length > 1) {
|
||||||
|
this.prefixBytes -= this.prefix.shift()!.length;
|
||||||
|
}
|
||||||
|
if (!voiced) return;
|
||||||
|
if (this.active) {
|
||||||
|
this.active.abort();
|
||||||
|
this.active = undefined;
|
||||||
|
this.generation += 1;
|
||||||
|
this.req.onClearAudio?.("barge-in");
|
||||||
|
}
|
||||||
|
this.speaking = true;
|
||||||
|
this.speech = this.prefix;
|
||||||
|
this.prefix = [];
|
||||||
|
this.prefixBytes = 0;
|
||||||
|
} else {
|
||||||
|
this.speech.push(Buffer.from(chunk));
|
||||||
|
}
|
||||||
|
|
||||||
|
const maxBytes = this.cfg.maxSpeechSeconds * 24000 * 2;
|
||||||
|
if (this.speech.reduce((sum, part) => sum + part.length, 0) >= maxBytes) {
|
||||||
|
void this.finishSpeech();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (voiced) {
|
||||||
|
if (this.silenceTimer) clearTimeout(this.silenceTimer);
|
||||||
|
this.silenceTimer = setTimeout(() => void this.finishSpeech(), this.cfg.silenceDurationMs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
sendUserMessage(text: string): void {
|
||||||
|
const trimmed = text.trim();
|
||||||
|
if (trimmed) void this.answer(trimmed);
|
||||||
|
}
|
||||||
|
|
||||||
|
triggerGreeting(instructions?: string): void {
|
||||||
|
void this.answer(instructions?.trim() || "Begrüße mich kurz auf Deutsch.");
|
||||||
|
}
|
||||||
|
|
||||||
|
submitToolResult(_callId: string, _result: unknown): void {}
|
||||||
|
acknowledgeMark(_markName: string): void {}
|
||||||
|
|
||||||
|
close(): void {
|
||||||
|
this.closed = true;
|
||||||
|
this.connected = false;
|
||||||
|
this.generation += 1;
|
||||||
|
this.active?.abort();
|
||||||
|
if (this.silenceTimer) clearTimeout(this.silenceTimer);
|
||||||
|
this.req.onClose?.("completed");
|
||||||
|
}
|
||||||
|
|
||||||
|
private async finishSpeech(): Promise<void> {
|
||||||
|
if (!this.speaking) return;
|
||||||
|
this.speaking = false;
|
||||||
|
if (this.silenceTimer) clearTimeout(this.silenceTimer);
|
||||||
|
this.silenceTimer = undefined;
|
||||||
|
const pcm = Buffer.concat(this.speech);
|
||||||
|
this.speech = [];
|
||||||
|
if (pcm.length < 24000 * 2 * 0.25) return;
|
||||||
|
|
||||||
|
try {
|
||||||
|
const text = await this.transcribe(pcm);
|
||||||
|
if (!text || !this.isConnected()) return;
|
||||||
|
this.req.onTranscript?.("user", text, true);
|
||||||
|
await this.answer(text);
|
||||||
|
} catch (error) {
|
||||||
|
if ((error as Error).name !== "AbortError") this.req.onError?.(error as Error);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private async transcribe(pcm: Buffer): Promise<string> {
|
||||||
|
const form = new FormData();
|
||||||
|
form.append("file", new Blob([wavFromPcm16(pcm)], { type: "audio/wav" }), "talk.wav");
|
||||||
|
form.append("model", "whisper-1");
|
||||||
|
form.append("language", this.cfg.language);
|
||||||
|
const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: this.authHeaders(),
|
||||||
|
body: form,
|
||||||
|
});
|
||||||
|
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||||
|
const payload = record(await response.json());
|
||||||
|
return String(payload.text || "").trim();
|
||||||
|
}
|
||||||
|
|
||||||
|
private async answer(prompt: string): Promise<void> {
|
||||||
|
if (!this.req.runAgentConsult) throw new Error("OpenClaw agent-consult is unavailable");
|
||||||
|
this.active?.abort();
|
||||||
|
const controller = new AbortController();
|
||||||
|
this.active = controller;
|
||||||
|
const generation = ++this.generation;
|
||||||
|
const responseId = `athena-${randomUUID()}`;
|
||||||
|
try {
|
||||||
|
const result = await this.req.runAgentConsult({ prompt, signal: controller.signal });
|
||||||
|
if (generation !== this.generation || !this.isConnected()) return;
|
||||||
|
const text = String(result?.text || "").trim();
|
||||||
|
if (!text) return;
|
||||||
|
this.req.onTranscript?.("assistant", text, true);
|
||||||
|
const pcm = await this.synthesize(text, controller.signal);
|
||||||
|
if (generation !== this.generation || !this.isConnected()) return;
|
||||||
|
const frameBytes = 24000 * 2 / 50;
|
||||||
|
for (let offset = 0; offset < pcm.length; offset += frameBytes) {
|
||||||
|
this.req.onAudio?.(pcm.subarray(offset, Math.min(offset + frameBytes, pcm.length)));
|
||||||
|
}
|
||||||
|
this.req.onResponseDone?.({ status: "completed", responseId });
|
||||||
|
this.req.onEvent?.({ direction: "server", type: "response.done", responseId });
|
||||||
|
} catch (error) {
|
||||||
|
if ((error as Error).name === "AbortError") {
|
||||||
|
this.req.onResponseDone?.({ status: "cancelled", responseId });
|
||||||
|
} else {
|
||||||
|
this.req.onError?.(error as Error);
|
||||||
|
this.req.onResponseDone?.({ status: "failed", responseId, error: String(error) });
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
if (this.active === controller) this.active = undefined;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private async synthesize(text: string, signal: AbortSignal): Promise<Buffer> {
|
||||||
|
const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: { "Content-Type": "application/json", ...this.authHeaders() },
|
||||||
|
// Let Athena select its currently active local backend (XTTS or Piper).
|
||||||
|
body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }),
|
||||||
|
signal,
|
||||||
|
});
|
||||||
|
if (!response.ok) throw new Error(`Athena TTS failed (HTTP ${response.status})`);
|
||||||
|
return readWavPcm24k(Buffer.from(await response.arrayBuffer()));
|
||||||
|
}
|
||||||
|
|
||||||
|
private authHeaders(): Record<string, string> {
|
||||||
|
return this.cfg.apiKey ? { Authorization: `Bearer ${this.cfg.apiKey}` } : {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export default definePluginEntry({
|
||||||
|
id: "athena-talk",
|
||||||
|
name: "Athena Local Talk",
|
||||||
|
description: "Private voice loop using Athena STT and TTS with the normal OpenClaw agent.",
|
||||||
|
register(api) {
|
||||||
|
api.registerRealtimeVoiceProvider({
|
||||||
|
id: "athena-talk",
|
||||||
|
label: "Athena Local Talk",
|
||||||
|
aliases: ["athena", "local-athena"],
|
||||||
|
defaultModel: "athena-local",
|
||||||
|
voices: ["alloy", "claribel"],
|
||||||
|
autoSelectOrder: 1,
|
||||||
|
capabilities: {
|
||||||
|
transports: ["gateway-relay"],
|
||||||
|
inputAudioFormats: [AUDIO_FORMAT],
|
||||||
|
outputAudioFormats: [AUDIO_FORMAT],
|
||||||
|
supportsBargeIn: true,
|
||||||
|
handlesInputAudioBargeIn: true,
|
||||||
|
supportsToolCalls: true,
|
||||||
|
supportsSessionResumption: false,
|
||||||
|
},
|
||||||
|
resolveConfig: ({ rawConfig }) => record(rawConfig),
|
||||||
|
isConfigured: ({ providerConfig, cfg }) => {
|
||||||
|
const raw = record(providerConfig);
|
||||||
|
const athena = record(record(record(cfg).models).providers).athena;
|
||||||
|
return Boolean(raw.baseUrl || record(athena).baseUrl);
|
||||||
|
},
|
||||||
|
createBridge: (req) => new AthenaTalkBridge(req, resolveConfig(req)),
|
||||||
|
} as any);
|
||||||
|
},
|
||||||
|
});
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
{
|
||||||
|
"id": "athena-talk",
|
||||||
|
"name": "Athena Local Talk",
|
||||||
|
"description": "Local German voice conversations through Athena without a public speech provider.",
|
||||||
|
"version": "0.1.0",
|
||||||
|
"enabledByDefault": true,
|
||||||
|
"activation": {
|
||||||
|
"onStartup": true,
|
||||||
|
"capabilities": [
|
||||||
|
"provider"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"contracts": {
|
||||||
|
"realtimeVoiceProviders": [
|
||||||
|
"athena-talk"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"configSchema": {
|
||||||
|
"type": "object",
|
||||||
|
"additionalProperties": false,
|
||||||
|
"properties": {}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
{
|
||||||
|
"name": "openclaw-plugin-athena-talk",
|
||||||
|
"version": "0.1.0",
|
||||||
|
"description": "Private realtime Talk bridge for Athena STT, OpenClaw agent consult, and Athena TTS.",
|
||||||
|
"type": "module",
|
||||||
|
"private": true,
|
||||||
|
"peerDependencies": {
|
||||||
|
"openclaw": ">=2026.8.2"
|
||||||
|
},
|
||||||
|
"peerDependenciesMeta": {
|
||||||
|
"openclaw": {
|
||||||
|
"optional": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"openclaw": {
|
||||||
|
"extensions": [
|
||||||
|
"./dist/index.js"
|
||||||
|
],
|
||||||
|
"compat": {
|
||||||
|
"pluginApi": ">=2026.8.2"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,7 +11,7 @@ RUN apt-get update \
|
|||||||
-DCMAKE_BUILD_TYPE=Release \
|
-DCMAKE_BUILD_TYPE=Release \
|
||||||
-DGGML_NATIVE=ON \
|
-DGGML_NATIVE=ON \
|
||||||
-DWHISPER_BUILD_TESTS=OFF \
|
-DWHISPER_BUILD_TESTS=OFF \
|
||||||
-DWHISPER_BUILD_SERVER=OFF \
|
-DWHISPER_BUILD_SERVER=ON \
|
||||||
&& cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \
|
&& cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \
|
||||||
&& apt-get purge -y --auto-remove build-essential cmake git \
|
&& apt-get purge -y --auto-remove build-essential cmake git \
|
||||||
&& rm -rf /var/lib/apt/lists/* /root/.cache \
|
&& rm -rf /var/lib/apt/lists/* /root/.cache \
|
||||||
@@ -25,7 +25,8 @@ RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint
|
|||||||
ENV WHISPER_HOST=0.0.0.0 \
|
ENV WHISPER_HOST=0.0.0.0 \
|
||||||
WHISPER_PORT=8084 \
|
WHISPER_PORT=8084 \
|
||||||
WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \
|
WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \
|
||||||
WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \
|
WHISPER_MODEL=/models/ggml-small.bin \
|
||||||
|
WHISPER_SERVER_URL=http://127.0.0.1:8085 \
|
||||||
WHISPER_THREADS=8 \
|
WHISPER_THREADS=8 \
|
||||||
WHISPER_LANGUAGE=de \
|
WHISPER_LANGUAGE=de \
|
||||||
FFMPEG_BIN=/usr/bin/ffmpeg
|
FFMPEG_BIN=/usr/bin/ffmpeg
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin}
|
model=${WHISPER_MODEL:-/models/ggml-small.bin}
|
||||||
model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin}
|
||||||
model_dir=$(dirname "$model")
|
model_dir=$(dirname "$model")
|
||||||
|
|
||||||
mkdir -p "$model_dir"
|
mkdir -p "$model_dir"
|
||||||
@@ -22,4 +22,44 @@ if [ ! -s "$model" ]; then
|
|||||||
gosu whisper mv "$partial" "$model"
|
gosu whisper mv "$partial" "$model"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
exec gosu whisper python /app/stt_worker.py
|
server_port=${WHISPER_SERVER_PORT:-8085}
|
||||||
|
server_log=/tmp/whisper-server.log
|
||||||
|
|
||||||
|
gosu whisper /opt/whisper.cpp/build/bin/whisper-server \
|
||||||
|
--model "$model" \
|
||||||
|
--threads "${WHISPER_THREADS:-8}" \
|
||||||
|
--language "${WHISPER_LANGUAGE:-de}" \
|
||||||
|
--host 127.0.0.1 \
|
||||||
|
--port "$server_port" \
|
||||||
|
--inference-path /inference \
|
||||||
|
--no-gpu >"$server_log" 2>&1 &
|
||||||
|
|
||||||
|
server_pid=$!
|
||||||
|
worker_pid=""
|
||||||
|
cleanup() {
|
||||||
|
[ -z "$worker_pid" ] || kill "$worker_pid" 2>/dev/null || true
|
||||||
|
kill "$server_pid" 2>/dev/null || true
|
||||||
|
}
|
||||||
|
trap cleanup EXIT INT TERM
|
||||||
|
|
||||||
|
# Der Modell-Load dauert beim Containerstart kurz. Der HTTP-Wrapper wird erst
|
||||||
|
# freigegeben, wenn whisper-server tatsächlich Antworten annimmt.
|
||||||
|
i=0
|
||||||
|
until curl --fail --silent "http://127.0.0.1:${server_port}/" >/dev/null; do
|
||||||
|
i=$((i + 1))
|
||||||
|
if ! kill -0 "$server_pid" 2>/dev/null; then
|
||||||
|
cat "$server_log"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [ "$i" -ge 180 ]; then
|
||||||
|
echo "whisper-server did not become ready" >&2
|
||||||
|
cat "$server_log"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
export WHISPER_SERVER_URL="http://127.0.0.1:${server_port}"
|
||||||
|
gosu whisper python /app/stt_worker.py &
|
||||||
|
worker_pid=$!
|
||||||
|
wait "$worker_pid"
|
||||||
|
|||||||
+75
-2
@@ -3,7 +3,8 @@
|
|||||||
STT-Worker – langlebiger Whisper-Transkriptions-Service (CPU-only).
|
STT-Worker – langlebiger Whisper-Transkriptions-Service (CPU-only).
|
||||||
|
|
||||||
Liest Audio-Dateien (WAV, MP3, OGG, FLAC, WebM/Opus via ffmpeg),
|
Liest Audio-Dateien (WAV, MP3, OGG, FLAC, WebM/Opus via ffmpeg),
|
||||||
transkribiert sie mit whisper.cpp (whisper-cli) und liefert JSON-Text.
|
transkribiert sie mit einem dauerhaft geladenen whisper.cpp-Server (mit
|
||||||
|
whisper-cli als Rückfallweg) und liefert JSON-Text.
|
||||||
|
|
||||||
Konfiguration über Umgebungsvariablen:
|
Konfiguration über Umgebungsvariablen:
|
||||||
WHISPER_HOST Bind-Adresse (Default: 127.0.0.1)
|
WHISPER_HOST Bind-Adresse (Default: 127.0.0.1)
|
||||||
@@ -28,6 +29,7 @@ import sys
|
|||||||
import tempfile
|
import tempfile
|
||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
|
import urllib.request
|
||||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -45,6 +47,7 @@ WHISPER_MODEL = os.environ.get(
|
|||||||
)
|
)
|
||||||
WHISPER_THREADS = int(os.environ.get("WHISPER_THREADS", "8"))
|
WHISPER_THREADS = int(os.environ.get("WHISPER_THREADS", "8"))
|
||||||
WHISPER_LANGUAGE = os.environ.get("WHISPER_LANGUAGE", "de")
|
WHISPER_LANGUAGE = os.environ.get("WHISPER_LANGUAGE", "de")
|
||||||
|
WHISPER_SERVER_URL = os.environ.get("WHISPER_SERVER_URL", "").rstrip("/")
|
||||||
FFMPEG_BIN = os.environ.get("FFMPEG_BIN", "/usr/bin/ffmpeg")
|
FFMPEG_BIN = os.environ.get("FFMPEG_BIN", "/usr/bin/ffmpeg")
|
||||||
LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO")
|
LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO")
|
||||||
|
|
||||||
@@ -139,13 +142,22 @@ def transcribe(
|
|||||||
temperature: float | None = None,
|
temperature: float | None = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""
|
"""
|
||||||
Führt die Transkription mit whisper-cli aus.
|
Führt die Transkription bevorzugt über den persistenten whisper.cpp-
|
||||||
|
Server aus. Dadurch wird das Modell nicht pro Aufnahme neu geladen.
|
||||||
Liefert dict mit 'text' und Metadaten.
|
Liefert dict mit 'text' und Metadaten.
|
||||||
"""
|
"""
|
||||||
lang = language or WHISPER_LANGUAGE
|
lang = language or WHISPER_LANGUAGE
|
||||||
if lang == "auto":
|
if lang == "auto":
|
||||||
lang = "auto"
|
lang = "auto"
|
||||||
|
|
||||||
|
if WHISPER_SERVER_URL:
|
||||||
|
return _transcribe_via_server(
|
||||||
|
audio_path,
|
||||||
|
language=lang,
|
||||||
|
prompt=prompt,
|
||||||
|
temperature=temperature,
|
||||||
|
)
|
||||||
|
|
||||||
out_prefix = f"/tmp/stt_{uuid.uuid4().hex[:12]}"
|
out_prefix = f"/tmp/stt_{uuid.uuid4().hex[:12]}"
|
||||||
out_json = out_prefix + ".json"
|
out_json = out_prefix + ".json"
|
||||||
|
|
||||||
@@ -217,6 +229,65 @@ def transcribe(
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _transcribe_via_server(
|
||||||
|
audio_path: str,
|
||||||
|
language: str,
|
||||||
|
prompt: str | None = None,
|
||||||
|
temperature: float | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""Sendet eine Aufnahme an den bereits geladenen whisper-server."""
|
||||||
|
boundary = f"----athena-whisper-{uuid.uuid4().hex}"
|
||||||
|
chunks: list[bytes] = []
|
||||||
|
|
||||||
|
def add_field(name: str, value: str) -> None:
|
||||||
|
chunks.extend([
|
||||||
|
f"--{boundary}\r\n".encode(),
|
||||||
|
f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(),
|
||||||
|
value.encode("utf-8"),
|
||||||
|
b"\r\n",
|
||||||
|
])
|
||||||
|
|
||||||
|
with open(audio_path, "rb") as f:
|
||||||
|
audio = f.read()
|
||||||
|
chunks.extend([
|
||||||
|
f"--{boundary}\r\n".encode(),
|
||||||
|
b'Content-Disposition: form-data; name="file"; filename="audio.wav"\r\n',
|
||||||
|
b"Content-Type: audio/wav\r\n\r\n",
|
||||||
|
audio,
|
||||||
|
b"\r\n",
|
||||||
|
])
|
||||||
|
add_field("response_format", "json")
|
||||||
|
add_field("language", language)
|
||||||
|
if prompt:
|
||||||
|
add_field("prompt", prompt)
|
||||||
|
if temperature is not None:
|
||||||
|
add_field("temperature", str(temperature))
|
||||||
|
chunks.append(f"--{boundary}--\r\n".encode())
|
||||||
|
|
||||||
|
request = urllib.request.Request(
|
||||||
|
f"{WHISPER_SERVER_URL}/inference",
|
||||||
|
data=b"".join(chunks),
|
||||||
|
headers={"Content-Type": f"multipart/form-data; boundary={boundary}"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
t0 = time.monotonic()
|
||||||
|
with urllib.request.urlopen(request, timeout=300) as response:
|
||||||
|
payload = json.loads(response.read().decode("utf-8"))
|
||||||
|
elapsed = time.monotonic() - t0
|
||||||
|
text = payload.get("text", "") if isinstance(payload, dict) else ""
|
||||||
|
result = {
|
||||||
|
"text": text.strip(),
|
||||||
|
"language": language,
|
||||||
|
"duration_ms": int(elapsed * 1000),
|
||||||
|
"engine": "whisper-server",
|
||||||
|
}
|
||||||
|
log.info(
|
||||||
|
"Transkription (persistent): %d ms, %d Zeichen, Sprache=%s",
|
||||||
|
result["duration_ms"], len(result["text"]), language,
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# HTTP-Handler
|
# HTTP-Handler
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -300,6 +371,8 @@ class STTHandler(BaseHTTPRequestHandler):
|
|||||||
"whisper_cli_exists": cli_ok,
|
"whisper_cli_exists": cli_ok,
|
||||||
"threads": WHISPER_THREADS,
|
"threads": WHISPER_THREADS,
|
||||||
"language": WHISPER_LANGUAGE,
|
"language": WHISPER_LANGUAGE,
|
||||||
|
"server_url": WHISPER_SERVER_URL or None,
|
||||||
|
"persistent_model": bool(WHISPER_SERVER_URL),
|
||||||
"ffmpeg": FFMPEG_BIN,
|
"ffmpeg": FFMPEG_BIN,
|
||||||
"ffmpeg_exists": os.path.isfile(FFMPEG_BIN),
|
"ffmpeg_exists": os.path.isfile(FFMPEG_BIN),
|
||||||
})
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user