Stream long OpenClaw dictation in segments
This commit is contained in:
1 parent
e577c55489
commit
46e5bbdf7f
10 files changed
+313
-72
No files matched your search
+95
-28
@@ -4,6 +4,10 @@ const AUDIO_FORMAT = { encoding: "pcm16", sampleRateHz: 24000, channels: 1 };
|
||||
const BROWSER_OFFER_PATH = "/plugins/athena-talk/realtime/calls";
|
||||
const BROWSER_KEY_PATH = "/plugins/athena-talk/realtime/public-key";
|
||||
const MAX_OFFER_BYTES = 64 * 1024;
|
||||
const DICTATION_SAMPLE_RATE_HZ = 8000;
|
||||
const DICTATION_SEGMENT_BYTES = DICTATION_SAMPLE_RATE_HZ * 6;
|
||||
const DICTATION_OVERLAP_BYTES = DICTATION_SAMPLE_RATE_HZ / 2;
|
||||
const DICTATION_REQUEST_TIMEOUT_MS = 4500;
|
||||
const browserKeys = generateKeyPairSync("ed25519");
|
||||
const publicKeyPem = browserKeys.publicKey.export({ format: "pem", type: "spki" }).toString();
|
||||
function base64url(value) {
|
||||
@@ -149,13 +153,38 @@ function resolveTranscriptionConfig(cfg, rawConfig) {
|
||||
const talkProvider = record(record(talkConfig.providers)["athena-talk"]);
|
||||
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
|
||||
}
|
||||
function comparableWord(value) {
|
||||
return value.toLocaleLowerCase("de-DE").replace(/[^\p{L}\p{N}]+/gu, "");
|
||||
}
|
||||
function removeTranscriptOverlap(previous, current) {
|
||||
const priorWords = previous.trim().split(/\s+/).filter(Boolean);
|
||||
const currentWords = current.trim().split(/\s+/).filter(Boolean);
|
||||
const maximum = Math.min(12, priorWords.length, currentWords.length);
|
||||
for (let count = maximum; count >= 1; count -= 1) {
|
||||
const left = priorWords.slice(-count).map(comparableWord);
|
||||
const right = currentWords.slice(0, count).map(comparableWord);
|
||||
if (!left.every((word, index) => word && word === right[index]))
|
||||
continue;
|
||||
// A single short word is too ambiguous to remove safely. Longer words are
|
||||
// sufficient because the audio overlap is only half a second.
|
||||
if (count === 1 && left[0].length < 5)
|
||||
continue;
|
||||
return currentWords.slice(count).join(" ");
|
||||
}
|
||||
return currentWords.join(" ");
|
||||
}
|
||||
class AthenaTranscriptionSession {
|
||||
req;
|
||||
config;
|
||||
connected = false;
|
||||
closed = false;
|
||||
audio = [];
|
||||
bytes = 0;
|
||||
bufferedBytes = 0;
|
||||
totalBytes = 0;
|
||||
processing = Promise.resolve();
|
||||
completedTranscripts = [];
|
||||
emittedTranscripts = 0;
|
||||
processingError = null;
|
||||
constructor(req, config) {
|
||||
this.req = req;
|
||||
this.config = config;
|
||||
@@ -165,50 +194,88 @@ class AthenaTranscriptionSession {
|
||||
sendAudio(audio) {
|
||||
if (!this.isConnected() || audio.length === 0)
|
||||
return;
|
||||
if (this.bytes === 0)
|
||||
if (this.totalBytes === 0)
|
||||
this.req.onSpeechStart?.();
|
||||
const maxBytes = this.config.maxSpeechSeconds * 8000;
|
||||
if (this.bytes + audio.length > maxBytes) {
|
||||
if (this.totalBytes + audio.length > maxBytes) {
|
||||
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
|
||||
this.close();
|
||||
return;
|
||||
}
|
||||
this.audio.push(Buffer.from(audio));
|
||||
this.bytes += audio.length;
|
||||
this.bufferedBytes += audio.length;
|
||||
this.totalBytes += audio.length;
|
||||
while (this.bufferedBytes >= DICTATION_SEGMENT_BYTES) {
|
||||
this.queueFullSegment();
|
||||
}
|
||||
}
|
||||
close() {
|
||||
if (this.closed)
|
||||
return;
|
||||
this.closed = true;
|
||||
this.connected = false;
|
||||
if (!this.bytes)
|
||||
if (!this.totalBytes)
|
||||
return;
|
||||
const audio = Buffer.concat(this.audio);
|
||||
this.audio.length = 0;
|
||||
void this.transcribe(audio);
|
||||
this.emitCompletedTranscripts();
|
||||
const tail = Buffer.concat(this.audio);
|
||||
this.audio = [];
|
||||
this.bufferedBytes = 0;
|
||||
if (tail.length > DICTATION_OVERLAP_BYTES || this.completedTranscripts.length === 0) {
|
||||
this.queueTranscription(tail);
|
||||
}
|
||||
void this.processing.finally(() => {
|
||||
this.emitCompletedTranscripts();
|
||||
if (this.processingError && this.completedTranscripts.length === 0) {
|
||||
this.req.onError?.(this.processingError);
|
||||
}
|
||||
});
|
||||
}
|
||||
async transcribe(audio) {
|
||||
try {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(4500),
|
||||
});
|
||||
if (!response.ok)
|
||||
throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
const text = String(record(await response.json()).text || "").trim();
|
||||
if (text)
|
||||
this.req.onTranscript?.(text);
|
||||
}
|
||||
catch (error) {
|
||||
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
queueFullSegment() {
|
||||
const buffered = Buffer.concat(this.audio);
|
||||
const segment = Buffer.from(buffered.subarray(0, DICTATION_SEGMENT_BYTES));
|
||||
const retained = Buffer.from(buffered.subarray(DICTATION_SEGMENT_BYTES - DICTATION_OVERLAP_BYTES));
|
||||
this.audio = retained.length ? [retained] : [];
|
||||
this.bufferedBytes = retained.length;
|
||||
this.queueTranscription(segment);
|
||||
}
|
||||
queueTranscription(audio) {
|
||||
if (!audio.length)
|
||||
return;
|
||||
this.processing = this.processing.then(async () => {
|
||||
const previous = this.completedTranscripts.join(" ");
|
||||
const text = await this.transcribe(audio, previous.slice(-240));
|
||||
const novel = removeTranscriptOverlap(previous, text);
|
||||
if (novel)
|
||||
this.completedTranscripts.push(novel);
|
||||
if (this.closed)
|
||||
this.emitCompletedTranscripts();
|
||||
}).catch((error) => {
|
||||
this.processingError = error instanceof Error ? error : new Error(String(error));
|
||||
});
|
||||
}
|
||||
emitCompletedTranscripts() {
|
||||
while (this.emittedTranscripts < this.completedTranscripts.length) {
|
||||
this.req.onTranscript?.(this.completedTranscripts[this.emittedTranscripts]);
|
||||
this.emittedTranscripts += 1;
|
||||
}
|
||||
}
|
||||
async transcribe(audio, prompt) {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
if (prompt)
|
||||
form.append("prompt", prompt);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(DICTATION_REQUEST_TIMEOUT_MS),
|
||||
});
|
||||
if (!response.ok)
|
||||
throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
return String(record(await response.json()).text || "").trim();
|
||||
}
|
||||
}
|
||||
function pcmRms(pcm) {
|
||||
if (pcm.length < 2)
|
||||
|
||||
Reference in new issue
Block a user