diff --git a/ATHENA.md b/ATHENA.md index 090d333..4f861f3 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -95,8 +95,8 @@ nicht direkt. Kein automatischer Host-Neustart ist vorgesehen. - Qwen-Image-2.1 INT8: Bildgenerierung und Editing auf der RTX 5080; das Textmodell wird dafür kurz entladen und danach automatisch wiederhergestellt - Qwen3-TTS 1.7B: RTX 3060; kein Piper-Fallback -- Qwen3-ASR 0.6B Q8: CPU, hinter dem bestehenden OpenAI-kompatiblen - Transkriptionsendpunkt. `whisper-1` bleibt nur als API-Kompatibilitätsname. +- Qwen3-ASR 0.6B Q8: CPU, hinter dem OpenAI-kompatiblen + Transkriptionsendpunkt mit dem Modellnamen `qwen3-asr`. - EmbeddingGemma 300M Q8: CPU, OpenAI-kompatibel auf Port 8082; kein GPU-Zugriff Die geprüften Live-Werte stehen in [docs/LIVE_STATE.md](docs/LIVE_STATE.md). diff --git a/README.md b/README.md index e617013..6a46368 100644 --- a/README.md +++ b/README.md @@ -145,8 +145,8 @@ Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über `/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Spracherkennung nutzt Qwen3-ASR-0.6B Q8 auf der CPU; das Modell liegt unter `/data/models/qwen3-asr-0.6b-q8`. Der Adapter liefert reinen Text unter -`qwen3-asr` und dem bisherigen `whisper-1`-Kompatibilitätsnamen, sodass -OpenClaw nicht neu konfiguriert werden muss. Whisper-Container, Image und +`qwen3-asr`; OpenClaw und die Voice-Brücke verwenden denselben Modellnamen. +Whisper-Container, Image und Modellvolume sind entfernt. Audiodaten werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale Realtime-Provider unter diff --git a/dev/mock_stt_worker.py b/dev/mock_stt_worker.py index e4fef06..ebcc5bf 100644 --- a/dev/mock_stt_worker.py +++ b/dev/mock_stt_worker.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """Mock-STT-Worker für lokale Tests. -Simuliert den Whisper-STT-Worker: +Simuliert den Qwen3-ASR-Adapter: GET /status → ready: true POST /transcribe → liefert festes Transkript diff --git a/dev/test_chunked.py b/dev/test_chunked.py index a50e2fc..6d77a82 100644 --- a/dev/test_chunked.py +++ b/dev/test_chunked.py @@ -220,7 +220,7 @@ def main() -> None: # Test 1: Multipart + Content-Length (bestehender Pfad) # ------------------------------------------------------------------ print("Test 1: Multipart + Content-Length") - mp = build_multipart({"model": "whisper-1", "language": "de"}, + mp = build_multipart({"model": "qwen3-asr", "language": "de"}, file_data=fake_webm) status, body = http_request( "POST", PORTS["router"], "/v1/audio/transcriptions", @@ -235,7 +235,7 @@ def main() -> None: # Test 2: Multipart + Transfer-Encoding chunked (einfach) # ------------------------------------------------------------------ print("Test 2: Multipart + chunked (einfach)") - mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm) + mp = build_multipart({"model": "qwen3-asr"}, file_data=fake_webm) chunked = build_chunked_body([mp]) raw = ( b"POST /v1/audio/transcriptions HTTP/1.1\r\n" @@ -254,7 +254,7 @@ def main() -> None: # Test 3: Mehrere unterschiedlich große Chunks # ------------------------------------------------------------------ print("Test 3: Mehrere unterschiedlich große Chunks") - mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm) + mp = build_multipart({"model": "qwen3-asr"}, file_data=fake_webm) # In 5 Chunks aufteilen (unterschiedlich groß) chunks = [] sizes = [10, 50, 7, 100, 33] @@ -283,7 +283,7 @@ def main() -> None: # Test 4: Boundary über Chunk-Grenzen verteilt # ------------------------------------------------------------------ print("Test 4: Boundary über Chunk-Grenzen verteilt") - mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm) + mp = build_multipart({"model": "qwen3-asr"}, file_data=fake_webm) # Boundary-String finden und Chunk-Grenze genau dorthin setzen boundary_str = b"--testboundary123" idx = mp.find(boundary_str, 10) # zweite Boundary (vor file) @@ -310,7 +310,7 @@ def main() -> None: # Test 5: Chunk Extensions # ------------------------------------------------------------------ print("Test 5: Chunk Extensions") - mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm) + mp = build_multipart({"model": "qwen3-asr"}, file_data=fake_webm) chunks_ext = [ (mp[:20], "ext1=value1"), (mp[20:60], None), @@ -335,7 +335,7 @@ def main() -> None: # hier explizit mit Trailer) # ------------------------------------------------------------------ print("Test 6: 0-Chunk mit Trailer") - mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm) + mp = build_multipart({"model": "qwen3-asr"}, file_data=fake_webm) chunked = build_chunked_body([mp]) # Trailer hinzufügen chunked_with_trailer = chunked.replace( @@ -376,7 +376,7 @@ def main() -> None: print("Test 8: Uploadgrößenlimit") # MAX_UPLOAD_SIZE = 1 MB, also 2 MB senden big_data = b"A" * (2 * 1024 * 1024) - mp = build_multipart({"model": "whisper-1"}, file_data=big_data) + mp = build_multipart({"model": "qwen3-asr"}, file_data=big_data) chunked = build_chunked_body([mp[:1024 * 1024], mp[1024 * 1024:]]) raw = ( b"POST /v1/audio/transcriptions HTTP/1.1\r\n" @@ -402,7 +402,7 @@ def main() -> None: + b"WEBM_OPUS_AUDIO_DATA" * 100) boundary = "950bd961b24c4a32801e31b128c85e09" mp = build_multipart( - {"model": "whisper-1", "language": "de"}, + {"model": "qwen3-asr", "language": "de"}, file_data=webm_data, filename="recording.webm", boundary=boundary, diff --git a/dev/test_multipart.py b/dev/test_multipart.py index e346b10..fa877cd 100644 --- a/dev/test_multipart.py +++ b/dev/test_multipart.py @@ -71,13 +71,13 @@ def test_quoted_boundary(): boundary = "----WebKitFormBoundary7MA4YWxkTrZu0gW" webm = b"\x1a\x45\xdf\xa3" + b"\x00\x01\x02\x03\xff\xfe\xfd" * 50 body, ct = build( - [("model", "whisper-1", None), ("file", webm, "t.webm")], + [("model", "qwen3-asr", None), ("file", webm, "t.webm")], boundary, quoted=True, ) fd, fn, fl = parse_multipart(body, ct) assert fd == webm, "file_data mismatch" assert fn == "t.webm", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" print(" quoted boundary: OK") @@ -109,7 +109,7 @@ def test_openwebui_style(): f"\r\n--{boundary}\r\n" f'Content-Disposition: form-data; name="model"\r\n' f"\r\n" - f"whisper-1\r\n" + f"qwen3-asr\r\n" f"--{boundary}\r\n" f'Content-Disposition: form-data; name="temperature"\r\n' f"\r\n" @@ -119,7 +119,7 @@ def test_openwebui_style(): fd, fn, fl = parse_multipart(body, ct) assert fd == webm, "file_data mismatch" assert fn == "rec.webm", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" assert fl["temperature"] == "0.0", f"temperature mismatch: {fl!r}" print(" Open-WebUI-artig: OK") @@ -128,13 +128,13 @@ def test_file_before_model(): """File-Feld vor model-Feld.""" boundary = "boundary123" body, ct = build( - [("file", b"DATA", "f.wav"), ("model", "whisper-1", None)], + [("file", b"DATA", "f.wav"), ("model", "qwen3-asr", None)], boundary, quoted=False, ) fd, fn, fl = parse_multipart(body, ct) assert fd == b"DATA", "file_data mismatch" assert fn == "f.wav", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" print(" File vor model: OK") @@ -142,13 +142,13 @@ def test_file_after_model(): """model-Feld vor File-Feld.""" boundary = "boundary456" body, ct = build( - [("model", "whisper-1", None), ("file", b"DATA", "g.wav")], + [("model", "qwen3-asr", None), ("file", b"DATA", "g.wav")], boundary, quoted=False, ) fd, fn, fl = parse_multipart(body, ct) assert fd == b"DATA", "file_data mismatch" assert fn == "g.wav", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" print(" File nach model: OK") @@ -182,13 +182,13 @@ def test_extra_headers_ignored(): f"\r\n--{boundary}\r\n" f'Content-Disposition: form-data; name="model"\r\n' f"\r\n" - f"whisper-1\r\n" + f"qwen3-asr\r\n" f"--{boundary}--\r\n" ).encode() fd, fn, fl = parse_multipart(body, ct) assert fd == webm, "file_data mismatch" assert fn == "x.webm", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" print(" Extra-Header ignoriert: OK") @@ -198,7 +198,7 @@ def test_all_fields(): body, ct = build( [ ("file", b"AUDIO", "a.webm"), - ("model", "whisper-1", None), + ("model", "qwen3-asr", None), ("language", "de", None), ("prompt", "Kontext", None), ("response_format", "verbose_json", None), @@ -209,7 +209,7 @@ def test_all_fields(): fd, fn, fl = parse_multipart(body, ct) assert fd == b"AUDIO", "file_data mismatch" assert fn == "a.webm", f"filename mismatch: {fn!r}" - assert fl["model"] == "whisper-1", f"model mismatch: {fl!r}" + assert fl["model"] == "qwen3-asr", f"model mismatch: {fl!r}" assert fl["language"] == "de", f"language mismatch: {fl!r}" assert fl["prompt"] == "Kontext", f"prompt mismatch: {fl!r}" assert fl["response_format"] == "verbose_json", f"response_format mismatch: {fl!r}" diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index 116cf0e..03aeb6c 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -39,7 +39,7 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, konvertiert Ausgabeformate und stellt Qwen3-TTS sowie natives PCM-Streaming über eine stabile interne API bereit. | | `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. | | `mike-ai-qwen-asr` | Qwen3-ASR 0.6B Q8 | CPU-Inferenz für lokale deutsche Spracherkennung. | -| `mike-ai-qwen-asr-worker` | kein eigenes Modell | Audio-Adapter für `/v1/audio/transcriptions`; `whisper-1` ist nur ein API-Kompatibilitätsname. | +| `mike-ai-qwen-asr-worker` | kein eigenes Modell | Audio-Adapter für `/v1/audio/transcriptions` mit dem Modellnamen `qwen3-asr`. | | `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. | | `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. | | `mike-ai-yue2-playground` | Image `mike-ai/yue2:3b-0.1.6` | Vorhandener, gestoppter Playground; in diesem Abgleich nicht funktional getestet. | diff --git a/docs/LIVE_STATE.md b/docs/LIVE_STATE.md index 6d2d198..5ad68c1 100644 --- a/docs/LIVE_STATE.md +++ b/docs/LIVE_STATE.md @@ -2,9 +2,9 @@ Nachtrag vom 25. September 2026: Die produktive Spracherkennung läuft über Qwen3-ASR 0.6B Q8 auf der CPU (`mike-ai-qwen-asr` und -`mike-ai-qwen-asr-worker`). Der Router bietet weiterhin `whisper-1` als -Kompatibilitätsnamen und zusätzlich `qwen3-asr` an. Beide wurden mit einer -M4A-Aufnahme über `/v1/audio/transcriptions` geprüft. Whisper-Container, +`mike-ai-qwen-asr-worker`). Der Router bietet nur `qwen3-asr` als STT-Modell +an; OpenClaw und die Voice-Brücke verwenden denselben Namen. Eine M4A-Aufnahme +wurde über `/v1/audio/transcriptions` geprüft. Whisper-Container, Images und Modellvolume wurden entfernt. Das aktive Ultra-Profil wurde bei dieser Umstellung nicht gewechselt. Die Angaben zu Whisper weiter unten beschreiben den historischen Stand vom 21. September. diff --git a/integrations/openclaw-athena-talk/README.md b/integrations/openclaw-athena-talk/README.md index e719f03..7970040 100644 --- a/integrations/openclaw-athena-talk/README.md +++ b/integrations/openclaw-athena-talk/README.md @@ -60,7 +60,8 @@ openclaw plugins install . --force --accept-capabilities openclaw plugins inspect athena-talk --runtime --json ``` -The plugin registers **Athena Qwen3-ASR (Diktieren)** as a separate +Version 1.3.1 sends `qwen3-asr` throughout dictation and Talk. The plugin +registers **Athena Qwen3-ASR (Diktieren)** as a separate realtime transcription provider through OpenClaw's official plugin API. In the browser composer, hold the microphone for dictation, then release it to send the 8 kHz G.711 audio through the Gateway. Short recordings are converted to @@ -95,8 +96,7 @@ An M4A voice note uploaded as a chat attachment does **not** use the realtime dictation provider above. OpenClaw processes it through its built-in `tools.media.audio` path. On the Unraid installation, automatic provider selection hit `SsrFBlockedError` for the private Athena address. Configure the -existing OpenAI-compatible provider and retain the `whisper-1` API alias for -Qwen3-ASR: +existing OpenAI-compatible provider with the `qwen3-asr` model: ```json5 { @@ -113,7 +113,7 @@ Qwen3-ASR: models: [ { provider: "openai", - model: "whisper-1", + model: "qwen3-asr", baseUrl: "http://192.168.1.212:8081/v1", capabilities: ["audio"], }, diff --git a/integrations/openclaw-athena-talk/dist/index.js b/integrations/openclaw-athena-talk/dist/index.js index 0fcc38a..25f45e5 100644 --- a/integrations/openclaw-athena-talk/dist/index.js +++ b/integrations/openclaw-athena-talk/dist/index.js @@ -262,7 +262,7 @@ class AthenaTranscriptionSession { async transcribe(audio, prompt) { const form = new FormData(); form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav"); - form.append("model", "whisper-1"); + form.append("model", "qwen3-asr"); form.append("language", this.config.language); if (prompt) form.append("prompt", prompt); @@ -471,7 +471,7 @@ class AthenaTalkBridge { const form = new FormData(); const wavBytes = Uint8Array.from(wavFromPcm16(pcm)); form.append("file", new Blob([wavBytes], { type: "audio/wav" }), "talk.wav"); - form.append("model", "whisper-1"); + form.append("model", "qwen3-asr"); form.append("language", this.cfg.language); const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, { method: "POST", @@ -568,8 +568,8 @@ export default definePluginEntry({ api.registerRealtimeTranscriptionProvider({ id: "athena-talk", label: "Athena Qwen3-ASR (Diktieren)", - defaultModel: "whisper-1", - models: ["whisper-1"], + defaultModel: "qwen3-asr", + models: ["qwen3-asr"], autoSelectOrder: 1, resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig), isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl), diff --git a/integrations/openclaw-athena-talk/index.ts b/integrations/openclaw-athena-talk/index.ts index b2cc0b0..71e62fc 100644 --- a/integrations/openclaw-athena-talk/index.ts +++ b/integrations/openclaw-athena-talk/index.ts @@ -289,7 +289,7 @@ class AthenaTranscriptionSession { private async transcribe(audio: Buffer, prompt: string): Promise { const form = new FormData(); form.append("file", new Blob([Uint8Array.from(wavFromMulaw8k(audio))], { type: "audio/wav" }), "dictation.wav"); - form.append("model", "whisper-1"); + form.append("model", "qwen3-asr"); form.append("language", this.config.language); if (prompt) form.append("prompt", prompt); const response = await fetch(`${this.config.baseUrl}/audio/transcriptions`, { @@ -490,7 +490,7 @@ class AthenaTalkBridge { const form = new FormData(); const wavBytes = Uint8Array.from(wavFromPcm16(pcm)); form.append("file", new Blob([wavBytes], { type: "audio/wav" }), "talk.wav"); - form.append("model", "whisper-1"); + form.append("model", "qwen3-asr"); form.append("language", this.cfg.language); const response = await fetch(`${this.cfg.baseUrl}/audio/transcriptions`, { method: "POST", @@ -579,8 +579,8 @@ export default definePluginEntry({ api.registerRealtimeTranscriptionProvider({ id: "athena-talk", label: "Athena Qwen3-ASR (Diktieren)", - defaultModel: "whisper-1", - models: ["whisper-1"], + defaultModel: "qwen3-asr", + models: ["qwen3-asr"], autoSelectOrder: 1, resolveConfig: ({ cfg, rawConfig }) => resolveTranscriptionConfig(cfg, rawConfig), isConfigured: ({ providerConfig }) => Boolean(record(providerConfig).baseUrl), diff --git a/integrations/openclaw-athena-talk/package-lock.json b/integrations/openclaw-athena-talk/package-lock.json index 20025a5..6d5dccf 100644 --- a/integrations/openclaw-athena-talk/package-lock.json +++ b/integrations/openclaw-athena-talk/package-lock.json @@ -1,12 +1,12 @@ { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.3.0", + "version": "1.3.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.3.0", + "version": "1.3.1", "devDependencies": { "@types/node": "^24.0.0", "openclaw": "2026.9.4", diff --git a/integrations/openclaw-athena-talk/package.json b/integrations/openclaw-athena-talk/package.json index 43f1ccb..62958f9 100644 --- a/integrations/openclaw-athena-talk/package.json +++ b/integrations/openclaw-athena-talk/package.json @@ -1,6 +1,6 @@ { "name": "@casaderoll/openclaw-athena-talk", - "version": "1.3.0", + "version": "1.3.1", "private": true, "description": "Local OpenClaw Talk provider backed by Athena Qwen3-ASR and Qwen3-TTS", "type": "module", diff --git a/integrations/openclaw-athena-talk/test-transcription.mjs b/integrations/openclaw-athena-talk/test-transcription.mjs index 45968db..86231ab 100644 --- a/integrations/openclaw-athena-talk/test-transcription.mjs +++ b/integrations/openclaw-athena-talk/test-transcription.mjs @@ -11,7 +11,7 @@ async function waitFor(predicate, timeoutMs = 2000) { } } -test("dictation registers separately and sends G.711 audio to Athena Whisper", async () => { +test("dictation registers separately and sends G.711 audio to Athena Qwen3-ASR", async () => { let transcription; plugin.register({ registerRealtimeTranscriptionProvider: (value) => { transcription = value; }, @@ -37,7 +37,7 @@ test("dictation registers separately and sends G.711 audio to Athena Whisper", a assert.equal(wav.readUInt16LE(34), 16); assert.equal(wav.length, 48); assert.equal(form.get("language"), "de"); - assert.equal(form.get("model"), "whisper-1"); + assert.equal(form.get("model"), "qwen3-asr"); res.writeHead(200, { "Content-Type": "application/json" }).end(JSON.stringify({ text: "Hallo Athena" })); }); await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); diff --git a/platform/hermes/config.yaml b/platform/hermes/config.yaml index 8ef49a3..0a91002 100644 --- a/platform/hermes/config.yaml +++ b/platform/hermes/config.yaml @@ -133,8 +133,8 @@ display: language: "de" show_reasoning: false -# Speech input stays local and private. A fixed German hint avoids Whisper -# interpreting short utterances as English while retaining the fast base model. +# Speech input stays local and private. The fixed German hint helps with +# short utterances and product names. stt: enabled: true provider: "local" diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index b859f02..60a7d90 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -219,7 +219,7 @@ TTS_DEFAULT_FORMAT = "mp3" STT_WORKER_URL = os.environ.get("STT_WORKER_URL", "http://127.0.0.1:8084") STT_TIMEOUT = float(os.environ.get("STT_TIMEOUT", "120")) # s, pro Transkription STT_CONNECT_TIMEOUT = float(os.environ.get("STT_CONNECT_TIMEOUT", "5")) -STT_MODEL = "whisper-1" # OpenClaw/OpenAI compatibility alias; Qwen3-ASR serves it +STT_MODEL = "qwen3-asr" # Maximale Upload-Größe (Bytes) – verhindert unbegrenzten RAM-Verbrauch. # 50 MB ist für Audio-Dateien (WebM/Opus, WAV, MP3) mehr als ausreichend. @@ -2848,12 +2848,6 @@ class Handler(BaseHTTPRequestHandler): "owned_by": "qwen3-asr", "type": "transcription", }) - models.append({ - "id": "qwen3-asr", - "object": "model", - "owned_by": "qwen3-asr", - "type": "transcription", - }) if tts.get("ready"): models.append({ "id": TTS_MODEL, @@ -2974,7 +2968,7 @@ class Handler(BaseHTTPRequestHandler): # Modell-Validierung model = fields.get("model", STT_MODEL) - if model not in (STT_MODEL, "whisper", "qwen3-asr"): + if model != STT_MODEL: self._send_error(400, f"unbekanntes Modell: {model!r} " f"(erwartet: {STT_MODEL})", "invalid_request_error", "unknown_model") diff --git a/scripts/speech-roundtrip.py b/scripts/speech-roundtrip.py index 24211b0..b47cb38 100644 --- a/scripts/speech-roundtrip.py +++ b/scripts/speech-roundtrip.py @@ -47,7 +47,7 @@ def main() -> int: body = ( f"--{boundary}\r\n" 'Content-Disposition: form-data; name="model"\r\n\r\n' - "whisper-1\r\n" + "qwen3-asr\r\n" f"--{boundary}\r\n" 'Content-Disposition: form-data; name="language"\r\n\r\n' "de\r\n" diff --git a/services/athena-realtime-voice/server.py b/services/athena-realtime-voice/server.py index df73cd6..539c646 100644 --- a/services/athena-realtime-voice/server.py +++ b/services/athena-realtime-voice/server.py @@ -281,7 +281,7 @@ class RealtimeSession: try: form = FormData() form.add_field("file", make_wav(pcm), filename="talk.wav", content_type="audio/wav") - form.add_field("model", "whisper-1") + form.add_field("model", "qwen3-asr") form.add_field("language", os.environ.get("STT_LANGUAGE", "de")) async with self.http.post( os.environ["ATHENA_API_BASE_URL"].rstrip("/") + "/audio/transcriptions",