Stream long OpenClaw dictation in segments
This commit is contained in:
1 parent
e577c55489
commit
46e5bbdf7f
10 files changed
+313
-72
No files matched your search
@@ -20,6 +20,10 @@ type ProviderConfig = {
|
||||
const BROWSER_OFFER_PATH = "/plugins/athena-talk/realtime/calls";
|
||||
const BROWSER_KEY_PATH = "/plugins/athena-talk/realtime/public-key";
|
||||
const MAX_OFFER_BYTES = 64 * 1024;
|
||||
const DICTATION_SAMPLE_RATE_HZ = 8000;
|
||||
const DICTATION_SEGMENT_BYTES = DICTATION_SAMPLE_RATE_HZ * 6;
|
||||
const DICTATION_OVERLAP_BYTES = DICTATION_SAMPLE_RATE_HZ / 2;
|
||||
const DICTATION_REQUEST_TIMEOUT_MS = 4500;
|
||||
const browserKeys = generateKeyPairSync("ed25519");
|
||||
const publicKeyPem = browserKeys.publicKey.export({ format: "pem", type: "spki" }).toString();
|
||||
|
||||
@@ -171,11 +175,36 @@ function resolveTranscriptionConfig(cfg: unknown, rawConfig: unknown): Required<
|
||||
return resolveConfig({ cfg, providerConfig: { ...talkProvider, ...record(rawConfig) } });
|
||||
}
|
||||
|
||||
function comparableWord(value: string): string {
|
||||
return value.toLocaleLowerCase("de-DE").replace(/[^\p{L}\p{N}]+/gu, "");
|
||||
}
|
||||
|
||||
function removeTranscriptOverlap(previous: string, current: string): string {
|
||||
const priorWords = previous.trim().split(/\s+/).filter(Boolean);
|
||||
const currentWords = current.trim().split(/\s+/).filter(Boolean);
|
||||
const maximum = Math.min(12, priorWords.length, currentWords.length);
|
||||
for (let count = maximum; count >= 1; count -= 1) {
|
||||
const left = priorWords.slice(-count).map(comparableWord);
|
||||
const right = currentWords.slice(0, count).map(comparableWord);
|
||||
if (!left.every((word, index) => word && word === right[index])) continue;
|
||||
// A single short word is too ambiguous to remove safely. Longer words are
|
||||
// sufficient because the audio overlap is only half a second.
|
||||
if (count === 1 && left[0].length < 5) continue;
|
||||
return currentWords.slice(count).join(" ");
|
||||
}
|
||||
return currentWords.join(" ");
|
||||
}
|
||||
|
||||
class AthenaTranscriptionSession {
|
||||
private connected = false;
|
||||
private closed = false;
|
||||
private readonly audio: Buffer[] = [];
|
||||
private bytes = 0;
|
||||
private audio: Buffer[] = [];
|
||||
private bufferedBytes = 0;
|
||||
private totalBytes = 0;
|
||||
private processing: Promise<void> = Promise.resolve();
|
||||
private readonly completedTranscripts: string[] = [];
|
||||
private emittedTranscripts = 0;
|
||||
private processingError: Error | null = null;
|
||||
|
||||
constructor(
|
||||
private readonly req: {
|
||||
@@ -193,46 +222,85 @@ class AthenaTranscriptionSession {
|
||||
|
||||
sendAudio(audio: Buffer): void {
|
||||
if (!this.isConnected() || audio.length === 0) return;
|
||||
if (this.bytes === 0) this.req.onSpeechStart?.();
|
||||
if (this.totalBytes === 0) this.req.onSpeechStart?.();
|
||||
const maxBytes = this.config.maxSpeechSeconds * 8000;
|
||||
if (this.bytes + audio.length > maxBytes) {
|
||||
if (this.totalBytes + audio.length > maxBytes) {
|
||||
this.req.onError?.(new Error(`Athena dictation is limited to ${this.config.maxSpeechSeconds} seconds`));
|
||||
this.close();
|
||||
return;
|
||||
}
|
||||
this.audio.push(Buffer.from(audio));
|
||||
this.bytes += audio.length;
|
||||
this.bufferedBytes += audio.length;
|
||||
this.totalBytes += audio.length;
|
||||
while (this.bufferedBytes >= DICTATION_SEGMENT_BYTES) {
|
||||
this.queueFullSegment();
|
||||
}
|
||||
}
|
||||
|
||||
close(): void {
|
||||
if (this.closed) return;
|
||||
this.closed = true;
|
||||
this.connected = false;
|
||||
if (!this.bytes) return;
|
||||
const audio = Buffer.concat(this.audio);
|
||||
this.audio.length = 0;
|
||||
void this.transcribe(audio);
|
||||
if (!this.totalBytes) return;
|
||||
this.emitCompletedTranscripts();
|
||||
const tail = Buffer.concat(this.audio);
|
||||
this.audio = [];
|
||||
this.bufferedBytes = 0;
|
||||
if (tail.length > DICTATION_OVERLAP_BYTES || this.completedTranscripts.length === 0) {
|
||||
this.queueTranscription(tail);
|
||||
}
|
||||
void this.processing.finally(() => {
|
||||
this.emitCompletedTranscripts();
|
||||
if (this.processingError && this.completedTranscripts.length === 0) {
|
||||
this.req.onError?.(this.processingError);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
private async transcribe(audio: Buffer): Promise<void> {
|
||||
try {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(4500),
|
||||
});
|
||||
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
const text = String(record(await response.json()).text || "").trim();
|
||||
if (text) this.req.onTranscript?.(text);
|
||||
} catch (error) {
|
||||
this.req.onError?.(error instanceof Error ? error : new Error(String(error)));
|
||||
private queueFullSegment(): void {
|
||||
const buffered = Buffer.concat(this.audio);
|
||||
const segment = Buffer.from(buffered.subarray(0, DICTATION_SEGMENT_BYTES));
|
||||
const retained = Buffer.from(buffered.subarray(DICTATION_SEGMENT_BYTES - DICTATION_OVERLAP_BYTES));
|
||||
this.audio = retained.length ? [retained] : [];
|
||||
this.bufferedBytes = retained.length;
|
||||
this.queueTranscription(segment);
|
||||
}
|
||||
|
||||
private queueTranscription(audio: Buffer): void {
|
||||
if (!audio.length) return;
|
||||
this.processing = this.processing.then(async () => {
|
||||
const previous = this.completedTranscripts.join(" ");
|
||||
const text = await this.transcribe(audio, previous.slice(-240));
|
||||
const novel = removeTranscriptOverlap(previous, text);
|
||||
if (novel) this.completedTranscripts.push(novel);
|
||||
if (this.closed) this.emitCompletedTranscripts();
|
||||
}).catch((error: unknown) => {
|
||||
this.processingError = error instanceof Error ? error : new Error(String(error));
|
||||
});
|
||||
}
|
||||
|
||||
private emitCompletedTranscripts(): void {
|
||||
while (this.emittedTranscripts < this.completedTranscripts.length) {
|
||||
this.req.onTranscript?.(this.completedTranscripts[this.emittedTranscripts]);
|
||||
this.emittedTranscripts += 1;
|
||||
}
|
||||
}
|
||||
|
||||
private async transcribe(audio: Buffer, prompt: string): Promise<string> {
|
||||
const form = new FormData();
|
||||
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
|
||||
form.append("model", "whisper-1");
|
||||
form.append("language", this.config.language);
|
||||
if (prompt) form.append("prompt", prompt);
|
||||
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
|
||||
method: "POST",
|
||||
headers: this.config.apiKey ? { Authorization: `Bearer ${this.config.apiKey}` } : {},
|
||||
body: form,
|
||||
signal: AbortSignal.timeout(DICTATION_REQUEST_TIMEOUT_MS),
|
||||
});
|
||||
if (!response.ok) throw new Error(`Athena STT failed (HTTP ${response.status})`);
|
||||
return String(record(await response.json()).text || "").trim();
|
||||
}
|
||||
}
|
||||
|
||||
function pcmRms(pcm: Buffer): number {
|
||||
|
||||
Reference in new issue
Block a user