Use Qwen3-ASR as production speech recognizer

This commit is contained in:
Mikei386
2026-09-25 21:31:33 +02:00
parent 6378b50086
commit 8a323e5b9e
17 changed files with 294 additions and 66 deletions
+12 -11
View File
@@ -5,7 +5,7 @@ existing Athena speech stack. An experimental browser WebRTC path is available
through the separate `services/athena-realtime-voice` service:
1. local VAD collects a spoken utterance,
2. Athena Whisper transcribes it,
2. Athena Qwen3-ASR transcribes it,
3. OpenClaw's normal agent-consult path answers with its configured model and
tools,
4. Athena Qwen3-TTS returns PCM audio to the Talk client.
@@ -60,23 +60,23 @@ openclaw plugins install . --force --accept-capabilities
openclaw plugins inspect athena-talk --runtime --json
```
Version 1.3.0 also registers **Athena Whisper (Diktieren)** as a separate
The plugin registers **Athena Qwen3-ASR (Diktieren)** as a separate
realtime transcription provider through OpenClaw's official plugin API. In the
browser composer, hold the microphone for dictation, then release it to send
the 8 kHz G.711 audio through the Gateway. Short recordings are converted to
PCM WAV and sent to Athena's existing `/audio/transcriptions` endpoint in one
request. Longer recordings are split while the user is still speaking into
six-second windows with 0.5 seconds of overlap. The plugin sends these windows
sequentially to the persistent Whisper service, carries a short text prompt
into the next request, removes duplicated overlap words, and caches finished
sequentially to the persistent Qwen3-ASR service, removes duplicated overlap
words, and caches finished
segments until recording stops. Only the short final tail then remains inside
OpenClaw's fixed five-second final-drain window. Each Whisper request is capped
OpenClaw's fixed five-second final-drain window. Each STT request is capped
at 4.5 seconds.
This is incremental pre-transcription over OpenClaw's official transcription
provider API. Whisper.cpp still receives complete short WAV segments; it is
not a native token-streaming STT protocol. No OpenClaw core file was patched
and no additional speech container was introduced. The transcribed text is
provider API. Qwen3-ASR still receives complete short WAV segments; it is
not a native token-streaming STT protocol. No OpenClaw core file was patched.
The transcribed text is
returned to the composer; this path does not invoke the agent or TTS. The
provider reuses `talk.realtime.providers.athena-talk` and the configured model
provider for its URL/key. If that model provider has no key, it reuses
@@ -85,7 +85,7 @@ origin. No second credential is needed. In `talk.catalog`, it appears under
`transcription.providers`.
The production provider allows up to 180 seconds per recording. This limit is
a local safety cap shared by dictation and Talk, not an OpenClaw or Whisper
a local safety cap shared by dictation and Talk, not an OpenClaw or Qwen3-ASR
restriction. Incremental segmentation keeps long dictation bounded while it
is being recorded.
@@ -95,7 +95,8 @@ An M4A voice note uploaded as a chat attachment does **not** use the realtime
dictation provider above. OpenClaw processes it through its built-in
`tools.media.audio` path. On the Unraid installation, automatic provider
selection hit `SsrFBlockedError` for the private Athena address. Configure the
existing OpenAI-compatible provider and select Whisper explicitly:
existing OpenAI-compatible provider and retain the `whisper-1` API alias for
Qwen3-ASR:
```json5
{
@@ -127,7 +128,7 @@ OpenClaw 2026.9.4 accepts `request.allowPrivateNetwork` under
settings hot-reload without restarting the Gateway. The existing
`ATHENA_ROUTER_API_KEY` SecretRef is reused; do not add another literal key.
On 2026-09-16, OpenClaw's official audio module transcribed the affected M4A
through Athena Whisper (127 characters returned). This confirms the endpoint
through Athena's then-active Whisper service (127 characters returned). This confirms the endpoint
and file format; a fresh attachment in the chat is still needed to verify the
full message-to-transcript flow.
+2 -2
View File
@@ -138,7 +138,7 @@ function wavFromPcm16(pcm, sampleRate = 24000) {
return Buffer.concat([header, pcm]);
}
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
// transcription endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio) {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
@@ -567,7 +567,7 @@ export default definePluginEntry({
register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
label: "Athena Qwen3-ASR (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
+2 -2
View File
@@ -158,7 +158,7 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer {
}
// The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's
// existing Whisper endpoint accepts PCM WAV uploads.
// transcription endpoint accepts PCM WAV uploads.
function wavFromMulaw8k(audio: Buffer): Buffer {
const pcm = Buffer.allocUnsafe(audio.length * 2);
for (let i = 0; i < audio.length; i += 1) {
@@ -578,7 +578,7 @@ export default definePluginEntry({
register(api) {
api.registerRealtimeTranscriptionProvider({
id: "athena-talk",
label: "Athena Whisper (Diktieren)",
label: "Athena Qwen3-ASR (Diktieren)",
defaultModel: "whisper-1",
models: ["whisper-1"],
autoSelectOrder: 1,
@@ -1,7 +1,7 @@
{
"id": "athena-talk",
"name": "Athena Local Talk",
"description": "Private OpenClaw Talk provider using Athena Whisper and Qwen3-TTS.",
"description": "Private OpenClaw Talk provider using Athena Qwen3-ASR and Qwen3-TTS.",
"activation": {
"onStartup": true
},
@@ -2,7 +2,7 @@
"name": "@casaderoll/openclaw-athena-talk",
"version": "1.3.0",
"private": true,
"description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS",
"description": "Local OpenClaw Talk provider backed by Athena Qwen3-ASR and Qwen3-TTS",
"type": "module",
"files": [
"dist",