Stream long OpenClaw dictation in segments
This commit is contained in:
@@ -3,6 +3,14 @@ import { createServer } from "node:http";
|
||||
import { test } from "node:test";
|
||||
import plugin from "./dist/index.js";
|
||||
|
||||
async function waitFor(predicate, timeoutMs = 2000) {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (!predicate()) {
|
||||
if (Date.now() >= deadline) throw new Error("timed out waiting for condition");
|
||||
await new Promise((resolve) => setTimeout(resolve, 10));
|
||||
}
|
||||
}
|
||||
|
||||
test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => {
|
||||
let transcription;
|
||||
plugin.register({
|
||||
@@ -79,3 +87,73 @@ test("dictation reuses only a TTS key for the same Athena origin", () => {
|
||||
cfg: withTts("http://other:8081/v1"), rawConfig: {},
|
||||
}).apiKey, "");
|
||||
});
|
||||
|
||||
test("long dictation is transcribed incrementally before the recording closes", async () => {
|
||||
let transcription;
|
||||
plugin.register({
|
||||
registerRealtimeTranscriptionProvider: (value) => { transcription = value; },
|
||||
registerRealtimeVoiceProvider: () => {},
|
||||
registerHttpRoute: () => {},
|
||||
});
|
||||
|
||||
const answers = [
|
||||
"Dies ist ein langer Abschnitt",
|
||||
"langer Abschnitt mit einer Fortsetzung",
|
||||
"einer Fortsetzung und einem Ende.",
|
||||
];
|
||||
let uploads = 0;
|
||||
const server = createServer(async (req, res) => {
|
||||
const index = uploads++;
|
||||
const form = await new Request("http://localhost", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": req.headers["content-type"] },
|
||||
body: req,
|
||||
duplex: "half",
|
||||
}).formData();
|
||||
const wav = Buffer.from(await form.get("file").arrayBuffer());
|
||||
assert.equal(wav.readUInt32LE(24), 8000);
|
||||
if (index > 0) assert.ok(String(form.get("prompt") || "").length > 0);
|
||||
res.writeHead(200, { "Content-Type": "application/json" })
|
||||
.end(JSON.stringify({ text: answers[index] }));
|
||||
});
|
||||
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));
|
||||
const cfg = { talk: { realtime: { providers: { "athena-talk": {
|
||||
baseUrl: `http://127.0.0.1:${server.address().port}/v1`, language: "de",
|
||||
} } } } };
|
||||
const providerConfig = transcription.resolveConfig({ cfg, rawConfig: {} });
|
||||
const transcripts = [];
|
||||
const errors = [];
|
||||
try {
|
||||
const session = transcription.createSession({
|
||||
cfg,
|
||||
providerConfig,
|
||||
onTranscript: (text) => transcripts.push(text),
|
||||
onError: (error) => errors.push(error),
|
||||
});
|
||||
await session.connect();
|
||||
|
||||
// Six seconds start the first request while dictation is still active.
|
||||
session.sendAudio(Buffer.alloc(48_000, 0xff));
|
||||
await waitFor(() => uploads === 1);
|
||||
assert.deepEqual(transcripts, []);
|
||||
|
||||
// Another 5.5 seconds form the next overlapping segment. The remaining
|
||||
// 2 seconds are finalized only when the user stops dictation.
|
||||
session.sendAudio(Buffer.alloc(44_000, 0xff));
|
||||
await waitFor(() => uploads === 2);
|
||||
session.sendAudio(Buffer.alloc(16_000, 0xff));
|
||||
session.close();
|
||||
|
||||
await waitFor(() => transcripts.length === 3);
|
||||
assert.deepEqual(transcripts, [
|
||||
"Dies ist ein langer Abschnitt",
|
||||
"mit einer Fortsetzung",
|
||||
"und einem Ende.",
|
||||
]);
|
||||
assert.equal(uploads, 3);
|
||||
assert.deepEqual(errors, []);
|
||||
} finally {
|
||||
server.closeAllConnections();
|
||||
await new Promise((resolve) => server.close(resolve));
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user