Use qwen3-asr throughout speech integrations

This commit is contained in:
Mikei386
2026-09-25 21:46:43 +02:00
parent 8a323e5b9e
commit da10f6b48d
17 changed files with 52 additions and 58 deletions
+4 -4
View File
@@ -60,7 +60,8 @@ openclaw plugins install . --force --accept-capabilities
openclaw plugins inspect athena-talk --runtime --json
```
The plugin registers **Athena Qwen3-ASR (Diktieren)** as a separate
Version 1.3.1 sends `qwen3-asr` throughout dictation and Talk. The plugin
registers **Athena Qwen3-ASR (Diktieren)** as a separate
realtime transcription provider through OpenClaw's official plugin API. In the
browser composer, hold the microphone for dictation, then release it to send
the 8 kHz G.711 audio through the Gateway. Short recordings are converted to
@@ -95,8 +96,7 @@ An M4A voice note uploaded as a chat attachment does **not** use the realtime
dictation provider above. OpenClaw processes it through its built-in
`tools.media.audio` path. On the Unraid installation, automatic provider
selection hit `SsrFBlockedError` for the private Athena address. Configure the
existing OpenAI-compatible provider and retain the `whisper-1` API alias for
Qwen3-ASR:
existing OpenAI-compatible provider with the `qwen3-asr` model:
```json5
{
@@ -113,7 +113,7 @@ Qwen3-ASR:
models: [
{
provider: "openai",
model: "whisper-1",
model: "qwen3-asr",
baseUrl: "http://192.168.1.212:8081/v1",
capabilities: ["audio"],
},
+4 -4
View File
@@ -262,7 +262,7 @@ class AthenaTranscriptionSession {
async transcribe(audio, prompt) {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("model", "qwen3-asr");
form.append("language", this.config.language);
if (prompt)
form.append("prompt", prompt);
@@ -471,7 +471,7 @@ class AthenaTalkBridge {
const form = new FormData();
const wavBytes = Uint8Array.from(wavFromPcm16(pcm));
form.append("file", new Blob([wavBytes], { type: "audio/wav" }), "talk.wav");
form.append("model", "whisper-1");
form.append("model", "qwen3-asr");
form.append("language", this.cfg.language);
const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, {
method: "POST",
@@ -568,8 +568,8 @@ export default definePluginEntry({
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Qwen3-ASR (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
defaultModel: "qwen3-asr",
models: ["qwen3-asr"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
+4 -4
View File
@@ -289,7 +289,7 @@ class AthenaTranscriptionSession {
private async transcribe(audio: Buffer, prompt: string): Promise<string> {
const form = new FormData();
form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav");
form.append("model", "whisper-1");
form.append("model", "qwen3-asr");
form.append("language", this.config.language);
if (prompt) form.append("prompt", prompt);
const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, {
@@ -490,7 +490,7 @@ class AthenaTalkBridge {
const form = new FormData();
const wavBytes = Uint8Array.from(wavFromPcm16(pcm));
form.append("file", new Blob([wavBytes], { type: "audio/wav" }), "talk.wav");
form.append("model", "whisper-1");
form.append("model", "qwen3-asr");
form.append("language", this.cfg.language);
const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, {
method: "POST",
@@ -579,8 +579,8 @@ export default definePluginEntry({
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Qwen3-ASR (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
defaultModel: "qwen3-asr",
models: ["qwen3-asr"],
autoSelectOrder: 1,
resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig),
isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl),
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.3.0",
"version": "1.3.1",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.3.0",
"version": "1.3.1",
"devDependencies": {
"@types/node": "^24.0.0",
"openclaw": "2026.9.4",
@@ -1,6 +1,6 @@
{
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.3.0",
"version": "1.3.1",
"private": true,
"description": "Local OpenClaw Talk provider backed by Athena Qwen3-ASR and Qwen3-TTS",
"type": "module",
@@ -11,7 +11,7 @@ async function waitFor(predicate, timeoutMs = 2000) {
}
}
test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => {
test("dictation registers separately and sends G.711 audio to Athena Qwen3-ASR", async () => {
let transcription;
plugin.register({
registerRealtimeTranscriptionProvider: (value) => { transcription = value; },
@@ -37,7 +37,7 @@ test("dictation registers separately and sends G.711 audio to Athena Whisper", a
assert.equal(wav.readUInt16LE(34), 16);
assert.equal(wav.length, 48);
assert.equal(form.get("language"), "de");
assert.equal(form.get("model"), "whisper-1");
assert.equal(form.get("model"), "qwen3-asr");
res.writeHead(200, { "Content-Type": "application/json" }).end(JSON.stringify({ text: "Hallo Athena" }));
});
await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve));