diff --git a/ATHENA.md b/ATHENA.md index deec0e2..090d333 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -24,7 +24,7 @@ Sie betreibt: - den OpenAI-kompatiblen Profile Router, - Qwen-Image-2.1 INT8 für Textbilder und Referenzbild-Bearbeitung, - Qwen3-TTS und TTS-Gateway für Sprache, -- Whisper.cpp und die WebRTC-Brücke für OpenClaw Talk, +- Qwen3-ASR auf der CPU und die WebRTC-Brücke für OpenClaw Talk, - EmbeddingGemma auf der CPU für OpenClaws semantische Memory-Suche, - die GPU-lose Mikes-Applio-UI als gesonderten Checkout, - das Athena-Dashboard, @@ -95,6 +95,8 @@ nicht direkt. Kein automatischer Host-Neustart ist vorgesehen. - Qwen-Image-2.1 INT8: Bildgenerierung und Editing auf der RTX 5080; das Textmodell wird dafür kurz entladen und danach automatisch wiederhergestellt - Qwen3-TTS 1.7B: RTX 3060; kein Piper-Fallback +- Qwen3-ASR 0.6B Q8: CPU, hinter dem bestehenden OpenAI-kompatiblen + Transkriptionsendpunkt. `whisper-1` bleibt nur als API-Kompatibilitätsname. - EmbeddingGemma 300M Q8: CPU, OpenAI-kompatibel auf Port 8082; kein GPU-Zugriff Die geprüften Live-Werte stehen in [docs/LIVE_STATE.md](docs/LIVE_STATE.md). diff --git a/README.md b/README.md index c250d4c..e617013 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@ dokumentieren den ausgerollten Stand, reduzierte Statuslatenzen und Tests. Q5-Worker auf der RTX 3060 - FLUX.2 Klein 9B FP8 Beta als gestoppter Rückfallcontainer - Qwen3-TTS 1.7B auf der RTX 3060 hinter dem TTS-Gateway; kein Piper-Fallback -- Whisper.cpp `small` auf der CPU für lokale deutsche Spracherkennung +- Qwen3-ASR 0.6B Q8 auf der CPU für lokale deutsche Spracherkennung - EmbeddingGemma 300M Q8 auf der CPU für OpenClaws hybride Memory-Suche - Live-Dashboard mit 21 Tagen Detailhistorie für GPUs und Slot-Kontextbelegung auf Port 8099 - Portainer CE als optionale Container-Ansicht auf Port 9443 @@ -142,12 +142,16 @@ Prefills konkurrieren weiterhin um Rechenleistung und den gemeinsamen KV-Pool. - Athena-Dashboard: `http://192.168.1.212:8099` Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über -`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das -Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten -werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale +`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. +Spracherkennung nutzt Qwen3-ASR-0.6B Q8 auf der CPU; das Modell liegt unter +`/data/models/qwen3-asr-0.6b-q8`. Der Adapter liefert reinen Text unter +`qwen3-asr` und dem bisherigen `whisper-1`-Kompatibilitätsnamen, sodass +OpenClaw nicht neu konfiguriert werden muss. Whisper-Container, Image und +Modellvolume sind entfernt. Audiodaten werden lokal auf Athena verarbeitet. +Für OpenClaw Talk liegt der lokale Realtime-Provider unter [`integrations/openclaw-athena-talk`](integrations/openclaw-athena-talk). Er -verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives +verbindet Mikrofon → Athena STT → normalen OpenClaw-Agenten → aktives Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis und bleibt deshalb bei normalen Container-Updates bestehen. diff --git a/compose.yaml b/compose.yaml index ef20659..6267e0d 100644 --- a/compose.yaml +++ b/compose.yaml @@ -658,7 +658,7 @@ services: MUSIC_START_TIMEOUT: "600" VOICE_CHANGE_START_TIMEOUT: "600" APPLIO_START_TIMEOUT: "900" - STT_WORKER_URL: http://whisper:8084 + STT_WORKER_URL: http://qwen-asr-worker:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] security_opt: ["no-new-privileges:true"] @@ -681,7 +681,7 @@ services: condition: service_healthy tts-gateway: condition: service_healthy - whisper: + qwen-asr-worker: condition: service_healthy image-worker: @@ -976,40 +976,77 @@ services: retries: 12 start_period: 10s - whisper: - build: - context: . - dockerfile: platform/docker/whisper/Dockerfile - args: - WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1} - image: mike-ai/whisper:local - container_name: mike-ai-whisper + qwen-asr: + image: ${LLAMA_CPU_IMAGE:-mike-ai/llama.cpp-cpu:local} + container_name: mike-ai-qwen-asr restart: unless-stopped read_only: true tmpfs: - - /tmp:size=2g,mode=1777 + - /tmp:size=256m,mode=1777 volumes: - - whisper-data:/models - environment: - WHISPER_HOST: 0.0.0.0 - WHISPER_PORT: "8084" - WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli - WHISPER_MODEL: /models/ggml-small.bin - WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin} - WHISPER_SERVER_PORT: "8085" - WHISPER_THREADS: ${WHISPER_THREADS:-8} - WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de} - networks: [frontend, inference] + - "${QWEN_ASR_MODEL_DIR:-/data/models/qwen3-asr-0.6b-q8}:/models:ro" + command: + - --model + - /models/Qwen3-ASR-0.6B-Q8_0.gguf + - --mmproj + - /models/mmproj-Qwen3-ASR-0.6B-Q8_0.gguf + - --no-mmproj-offload + - --n-gpu-layers + - "0" + - --alias + - qwen3-asr-0.6b + - --ctx-size + - "4096" + - --threads + - "6" + - --parallel + - "1" + - --host + - 0.0.0.0 + - --port + - "8080" + - --no-ui + - --fit + - "off" + cpus: 6 + mem_limit: 6g + networks: [inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] - # The entrypoint supervises whisper-server after dropping it to uid 10004. - cap_add: [CHOWN, SETUID, SETGID, KILL] healthcheck: - test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"] + test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"] interval: 10s timeout: 5s - retries: 90 - start_period: 20m + retries: 12 + start_period: 30s + + qwen-asr-worker: + build: + context: . + dockerfile: platform/docker/qwen-asr-worker/Dockerfile + image: mike-ai/qwen-asr-worker:local + container_name: mike-ai-qwen-asr-worker + restart: unless-stopped + read_only: true + tmpfs: + - /tmp:size=256m,mode=1777 + environment: + QWEN_ASR_HOST: 0.0.0.0 + QWEN_ASR_PORT: "8084" + QWEN_ASR_LANGUAGE: de + QWEN_ASR_SERVER_URL: http://qwen-asr:8080 + networks: [inference] + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + depends_on: + qwen-asr: + condition: service_healthy + healthcheck: + test: [CMD, python, -c, "import json,urllib.request; assert json.load(urllib.request.urlopen('http://127.0.0.1:8084/status', timeout=2))['ready']"] + interval: 10s + timeout: 5s + retries: 12 + start_period: 15s llama-dashboard: build: ./platform/llama-dashboard @@ -1124,7 +1161,6 @@ networks: name: mike-ai-tools-egress volumes: - whisper-data: router-state: router-images: portainer-data: diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index 6bfa9e7..116cf0e 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -32,13 +32,14 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-portainer` | kein Modell; Portainer CE | Optionale Docker-Verwaltungsoberfläche. | | `mike-ai-profile-controller` | kein Modell | Startet und stoppt ausschließlich freigegebene Modellprofile und Spezialworker in einer sicheren Reihenfolge. | | `mike-ai-qwen-cron-test` | kleines Qwen-Testmodell | Gestoppter CPU-/Cron-Worker-Versuch; nicht produktiv eingesetzt. | -| `mike-ai-realtime-voice` | kein Modell | Laufende, gesunde WebRTC-Brücke für OpenClaw Talk; verbindet über den privaten WireGuard-Pfad Athena Whisper, OpenClaw-Agent und Qwen3-TTS ohne Profilwechsel. | +| `mike-ai-realtime-voice` | kein Modell | Laufende, gesunde WebRTC-Brücke für OpenClaw Talk; verbindet über den privaten WireGuard-Pfad Athena Qwen3-ASR, OpenClaw-Agent und Qwen3-TTS ohne Profilwechsel. | | `mike-ai-qwen3-tts` | `Qwen/Qwen3-TTS-12Hz-1.7B-Base`, Stimme Serena | Hochwertige deutsche Sprachausgabe auf der RTX 3060 im LLM-Betrieb. | | `mike-ai-router` | kein eigenes Modell | Einzige OpenAI-kompatible Modelladresse; koordiniert Profile, Bildaufträge, Sprache und Betriebsarten. | | `mike-ai-stem-separator` | BS-RoFormer Viperx 1297, `htdemucs_ft`, `htdemucs_6s`, `MossFormer2_SE_48K` | Trennt Gesang, Instrumente oder Sprache/Hintergrundgeräusche im exklusiven Separationsmodus. | | `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, konvertiert Ausgabeformate und stellt Qwen3-TTS sowie natives PCM-Streaming über eine stabile interne API bereit. | | `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. | -| `mike-ai-whisper` | Whisper.cpp 1.9.4, `ggml-small` | Lokale deutsche Spracherkennung auf der CPU über `/v1/audio/transcriptions`. | +| `mike-ai-qwen-asr` | Qwen3-ASR 0.6B Q8 | CPU-Inferenz für lokale deutsche Spracherkennung. | +| `mike-ai-qwen-asr-worker` | kein eigenes Modell | Audio-Adapter für `/v1/audio/transcriptions`; `whisper-1` ist nur ein API-Kompatibilitätsname. | | `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. | | `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. | | `mike-ai-yue2-playground` | Image `mike-ai/yue2:3b-0.1.6` | Vorhandener, gestoppter Playground; in diesem Abgleich nicht funktional getestet. | diff --git a/docs/LIVE_STATE.md b/docs/LIVE_STATE.md index a5bcb96..6d2d198 100644 --- a/docs/LIVE_STATE.md +++ b/docs/LIVE_STATE.md @@ -1,5 +1,14 @@ # Geprüfter Live-Stand auf Athena +Nachtrag vom 25. September 2026: Die produktive Spracherkennung läuft über +Qwen3-ASR 0.6B Q8 auf der CPU (`mike-ai-qwen-asr` und +`mike-ai-qwen-asr-worker`). Der Router bietet weiterhin `whisper-1` als +Kompatibilitätsnamen und zusätzlich `qwen3-asr` an. Beide wurden mit einer +M4A-Aufnahme über `/v1/audio/transcriptions` geprüft. Whisper-Container, +Images und Modellvolume wurden entfernt. Das aktive Ultra-Profil wurde bei +dieser Umstellung nicht gewechselt. Die Angaben zu Whisper weiter unten +beschreiben den historischen Stand vom 21. September. + Nachtrag vom 24. September 2026: Ultra verarbeitet nun Bilder mit einem CPU-seitigen BF16-Vision-Projektor. Der Stand und der Funktionstest sind in [Ultra-Vision mit CPU-Projektor](ULTRA_CPU_VISION_20260924.md) dokumentiert. diff --git a/docs/RECOVERY.md b/docs/RECOVERY.md index 1cad98e..9dc3552 100644 --- a/docs/RECOVERY.md +++ b/docs/RECOVERY.md @@ -24,8 +24,10 @@ gesichert. Für eine Neuinstallation ist der im Git dokumentierte Build mit installieren; eine Änderung an OpenClaw-Core-Dateien ist nicht erforderlich. Piper-Daten sind kein aktueller Sicherungsbestand. Modellgewichte unter -`/data/models` und das reproduzierbare Whisper-Volume gehören nicht zu diesen -Backup-Mounts. Ein Backup ausschließlich auf `/data` schützt nicht vor deren Ausfall. +`/data/models`, einschließlich Qwen3-ASR unter +`/data/models/qwen3-asr-0.6b-q8`, gehören nicht zu diesen Backup-Mounts. +Das frühere Whisper-Volume wurde am 25. September 2026 entfernt. +Ein Backup ausschließlich auf `/data` schützt nicht vor einem Ausfall der Datenplatte. Das gilt auch für das reproduzierbare EmbeddingGemma-Gewicht unter `/data/models/embeddinggemma`; URL und SHA-256 stehen in `config/install.env.example`, sodass der Installer es erneut laden und prüfen kann. diff --git a/integrations/openclaw-athena-talk/README.md b/integrations/openclaw-athena-talk/README.md index 3c32583..e719f03 100644 --- a/integrations/openclaw-athena-talk/README.md +++ b/integrations/openclaw-athena-talk/README.md @@ -5,7 +5,7 @@ existing Athena speech stack. An experimental browser WebRTC path is available through the separate `services/athena-realtime-voice` service: 1. local VAD collects a spoken utterance, -2. Athena Whisper transcribes it, +2. Athena Qwen3-ASR transcribes it, 3. OpenClaw's normal agent-consult path answers with its configured model and tools, 4. Athena Qwen3-TTS returns PCM audio to the Talk client. @@ -60,23 +60,23 @@ openclaw plugins install . --force --accept-capabilities openclaw plugins inspect athena-talk --runtime --json ``` -Version 1.3.0 also registers **Athena Whisper (Diktieren)** as a separate +The plugin registers **Athena Qwen3-ASR (Diktieren)** as a separate realtime transcription provider through OpenClaw's official plugin API. In the browser composer, hold the microphone for dictation, then release it to send the 8 kHz G.711 audio through the Gateway. Short recordings are converted to PCM WAV and sent to Athena's existing `/audio/transcriptions` endpoint in one request. Longer recordings are split while the user is still speaking into six-second windows with 0.5 seconds of overlap. The plugin sends these windows -sequentially to the persistent Whisper service, carries a short text prompt -into the next request, removes duplicated overlap words, and caches finished +sequentially to the persistent Qwen3-ASR service, removes duplicated overlap +words, and caches finished segments until recording stops. Only the short final tail then remains inside -OpenClaw's fixed five-second final-drain window. Each Whisper request is capped +OpenClaw's fixed five-second final-drain window. Each STT request is capped at 4.5 seconds. This is incremental pre-transcription over OpenClaw's official transcription -provider API. Whisper.cpp still receives complete short WAV segments; it is -not a native token-streaming STT protocol. No OpenClaw core file was patched -and no additional speech container was introduced. The transcribed text is +provider API. Qwen3-ASR still receives complete short WAV segments; it is +not a native token-streaming STT protocol. No OpenClaw core file was patched. +The transcribed text is returned to the composer; this path does not invoke the agent or TTS. The provider reuses `talk.realtime.providers.athena-talk` and the configured model provider for its URL/key. If that model provider has no key, it reuses @@ -85,7 +85,7 @@ origin. No second credential is needed. In `talk.catalog`, it appears under `transcription.providers`. The production provider allows up to 180 seconds per recording. This limit is -a local safety cap shared by dictation and Talk, not an OpenClaw or Whisper +a local safety cap shared by dictation and Talk, not an OpenClaw or Qwen3-ASR restriction. Incremental segmentation keeps long dictation bounded while it is being recorded. @@ -95,7 +95,8 @@ An M4A voice note uploaded as a chat attachment does **not** use the realtime dictation provider above. OpenClaw processes it through its built-in `tools.media.audio` path. On the Unraid installation, automatic provider selection hit `SsrFBlockedError` for the private Athena address. Configure the -existing OpenAI-compatible provider and select Whisper explicitly: +existing OpenAI-compatible provider and retain the `whisper-1` API alias for +Qwen3-ASR: ```json5 { @@ -127,7 +128,7 @@ OpenClaw 2026.9.4 accepts `request.allowPrivateNetwork` under settings hot-reload without restarting the Gateway. The existing `ATHENA_ROUTER_API_KEY` SecretRef is reused; do not add another literal key. On 2026-09-16, OpenClaw's official audio module transcribed the affected M4A -through Athena Whisper (127 characters returned). This confirms the endpoint +through Athena's then-active Whisper service (127 characters returned). This confirms the endpoint and file format; a fresh attachment in the chat is still needed to verify the full message-to-transcript flow. diff --git a/integrations/openclaw-athena-talk/dist/index.js b/integrations/openclaw-athena-talk/dist/index.js index 7b81683..0fcc38a 100644 --- a/integrations/openclaw-athena-talk/dist/index.js +++ b/integrations/openclaw-athena-talk/dist/index.js @@ -138,7 +138,7 @@ function wavFromPcm16(pcm, sampleRate = 24000) { return Buffer.concat([header, pcm]); } // The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's -// existing Whisper endpoint accepts PCM WAV uploads. +// transcription endpoint accepts PCM WAV uploads. function wavFromMulaw8k(audio) { const pcm = Buffer.allocUnsafe(audio.length * 2); for (let i = 0; i < audio.length; i += 1) { @@ -567,7 +567,7 @@ export default definePluginEntry({ register(api) { api.registerRealtimeTranscriptionProvider({ id: "athena-talk", - label: "Athena Whisper (Diktieren)", + label: "Athena Qwen3-ASR (Diktieren)", defaultModel: "whisper-1", models: ["whisper-1"], autoSelectOrder: 1, diff --git a/integrations/openclaw-athena-talk/index.ts b/integrations/openclaw-athena-talk/index.ts index 8a37ca9..b2cc0b0 100644 --- a/integrations/openclaw-athena-talk/index.ts +++ b/integrations/openclaw-athena-talk/index.ts @@ -158,7 +158,7 @@ function wavFromPcm16(pcm: Buffer, sampleRate = 24000): Buffer { } // The browser transcription relay sends 8 kHz G.711 mu-law, while Athena's -// existing Whisper endpoint accepts PCM WAV uploads. +// transcription endpoint accepts PCM WAV uploads. function wavFromMulaw8k(audio: Buffer): Buffer { const pcm = Buffer.allocUnsafe(audio.length * 2); for (let i = 0; i < audio.length; i += 1) { @@ -578,7 +578,7 @@ export default definePluginEntry({ register(api) { api.registerRealtimeTranscriptionProvider({ id: "athena-talk", - label: "Athena Whisper (Diktieren)", + label: "Athena Qwen3-ASR (Diktieren)", defaultModel: "whisper-1", models: ["whisper-1"], autoSelectOrder: 1, diff --git a/integrations/openclaw-athena-talk/openclaw.plugin.json b/integrations/openclaw-athena-talk/openclaw.plugin.json index abf3267..ce5e5d2 100644 --- a/integrations/openclaw-athena-talk/openclaw.plugin.json +++ b/integrations/openclaw-athena-talk/openclaw.plugin.json @@ -1,7 +1,7 @@ { "id": "athena-talk", "name": "Athena Local Talk", - "description": "Private OpenClaw Talk provider using Athena Whisper and Qwen3-TTS.", + "description": "Private OpenClaw Talk provider using Athena Qwen3-ASR and Qwen3-TTS.", "activation": { "onStartup": true }, diff --git a/integrations/openclaw-athena-talk/package.json b/integrations/openclaw-athena-talk/package.json index 478d303..43f1ccb 100644 --- a/integrations/openclaw-athena-talk/package.json +++ b/integrations/openclaw-athena-talk/package.json @@ -2,7 +2,7 @@ "name": "@casaderoll/openclaw-athena-talk", "version": "1.3.0", "private": true, - "description": "Local OpenClaw Talk provider backed by Athena Whisper and Qwen3-TTS", + "description": "Local OpenClaw Talk provider backed by Athena Qwen3-ASR and Qwen3-TTS", "type": "module", "files": [ "dist", diff --git a/manage.sh b/manage.sh index 07c8aa7..c4ddaac 100755 --- a/manage.sh +++ b/manage.sh @@ -52,7 +52,8 @@ case "$command" in [[ $# -eq 0 ]] || { echo "core akzeptiert keine weiteren Services" >&2; exit 2; } run "$ROOT_DIR/platform/mcp/install-tools.sh" run "${compose[@]}" up -d --build \ - wireguard-gateway embedding qwen3-tts tts-gateway profile-controller router llama-dashboard portainer backup + wireguard-gateway embedding qwen3-tts tts-gateway profile-controller \ + qwen-asr qwen-asr-worker router llama-dashboard portainer backup else run "${compose[@]}" up -d --build --no-deps "$@" fi diff --git a/platform/docker/qwen-asr-worker/Dockerfile b/platform/docker/qwen-asr-worker/Dockerfile new file mode 100644 index 0000000..ad7e6e5 --- /dev/null +++ b/platform/docker/qwen-asr-worker/Dockerfile @@ -0,0 +1,17 @@ +FROM python:3.13.7-slim-bookworm + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && useradd --system --uid 10005 --home-dir /nonexistent --shell /usr/sbin/nologin stt + +WORKDIR /app +COPY router/qwen_asr_worker.py /app/qwen_asr_worker.py + +ENV QWEN_ASR_HOST=0.0.0.0 \ + QWEN_ASR_PORT=8084 \ + QWEN_ASR_LANGUAGE=de + +USER 10005:10005 +EXPOSE 8084 +ENTRYPOINT ["python", "/app/qwen_asr_worker.py"] diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 4ba25a3..b859f02 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -29,12 +29,11 @@ Sprachausgabe (Qwen3-TTS auf RTX 3060): POST /v1/audio/speech (OpenAI-kompatibel) GET /v1/audio/voices (verfügbare Stimmen) -Spracherkennung (whisper.cpp, deutsch, CPU-only): +Spracherkennung (Qwen3-ASR, deutsch, CPU-only): POST /v1/audio/transcriptions (OpenAI-kompatibel) GET /v1/audio/models (verfügbare Audio-Modelle) -Der TTS-Worker (mike-ai-xtts.service) und der STT-Worker -(mike-ai-whisper.service) laufen als separate, langlebige Prozesse. +TTS und Qwen3-ASR laufen als separate, langlebige Dienste. Der Router leitet /v1/audio/speech und /v1/audio/transcriptions per HTTP an die Worker weiter. @@ -216,11 +215,11 @@ TTS_DEFAULT_VOICE = os.environ.get( TTS_FORMATS = ("mp3", "wav", "pcm") TTS_DEFAULT_FORMAT = "mp3" -# --- Spracherkennung (whisper.cpp, deutsch, CPU-only) --- +# --- Spracherkennung (Qwen3-ASR, deutsch, CPU-only) --- STT_WORKER_URL = os.environ.get("STT_WORKER_URL", "http://127.0.0.1:8084") STT_TIMEOUT = float(os.environ.get("STT_TIMEOUT", "120")) # s, pro Transkription STT_CONNECT_TIMEOUT = float(os.environ.get("STT_CONNECT_TIMEOUT", "5")) -STT_MODEL = "whisper-1" # virtuelles Modell für /v1/audio/transcriptions +STT_MODEL = "whisper-1" # OpenClaw/OpenAI compatibility alias; Qwen3-ASR serves it # Maximale Upload-Größe (Bytes) – verhindert unbegrenzten RAM-Verbrauch. # 50 MB ist für Audio-Dateien (WebM/Opus, WAV, MP3) mehr als ausreichend. @@ -2846,7 +2845,13 @@ class Handler(BaseHTTPRequestHandler): models.append({ "id": STT_MODEL, "object": "model", - "owned_by": "whisper.cpp", + "owned_by": "qwen3-asr", + "type": "transcription", + }) + models.append({ + "id": "qwen3-asr", + "object": "model", + "owned_by": "qwen3-asr", "type": "transcription", }) if tts.get("ready"): @@ -2969,7 +2974,7 @@ class Handler(BaseHTTPRequestHandler): # Modell-Validierung model = fields.get("model", STT_MODEL) - if model not in (STT_MODEL, "whisper"): + if model not in (STT_MODEL, "whisper", "qwen3-asr"): self._send_error(400, f"unbekanntes Modell: {model!r} " f"(erwartet: {STT_MODEL})", "invalid_request_error", "unknown_model") diff --git a/router/qwen_asr_worker.py b/router/qwen_asr_worker.py new file mode 100644 index 0000000..5c9c20a --- /dev/null +++ b/router/qwen_asr_worker.py @@ -0,0 +1,147 @@ +#!/usr/bin/env python3 +"""OpenAI-router STT adapter for the persistent, CPU-only Qwen3-ASR server.""" + +import json +import logging +import os +import subprocess +import tempfile +import time +import urllib.request +import uuid +from email import policy +from email.parser import BytesParser +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +HOST = os.environ.get("QWEN_ASR_HOST", "0.0.0.0") +PORT = int(os.environ.get("QWEN_ASR_PORT", "8084")) +SERVER_URL = os.environ.get("QWEN_ASR_SERVER_URL", "http://qwen-asr:8080").rstrip("/") +LANGUAGE = os.environ.get("QWEN_ASR_LANGUAGE", "de") +MAX_BODY_BYTES = 25 * 1024 * 1024 + +logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s") +log = logging.getLogger("qwen-asr-worker") + + +def clean_transcript(value: str) -> str: + """llama.cpp may include a Qwen task marker before the spoken words.""" + if "" in value: + value = value.split("", 1)[1] + return value.replace("<|endoftext|>", "").strip() + + +def transcribe(audio: bytes, filename: str, language: str) -> dict: + suffix = os.path.splitext(filename)[1].lower() or ".wav" + with tempfile.TemporaryDirectory(prefix="qwen_asr_") as directory: + source = os.path.join(directory, "input" + suffix) + wav = os.path.join(directory, "audio.wav") + with open(source, "wb") as handle: + handle.write(audio) + result = subprocess.run( + ["ffmpeg", "-nostdin", "-hide_banner", "-loglevel", "error", "-y", + "-i", source, "-ac", "1", "-ar", "16000", "-c:a", "pcm_s16le", wav], + capture_output=True, text=True, timeout=30, + ) + if result.returncode: + raise ValueError("Audio konnte nicht gelesen werden: " + result.stderr[-300:]) + with open(wav, "rb") as handle: + pcm = handle.read() + + boundary = "athena-qwen-asr-" + uuid.uuid4().hex + body = b"".join([ + f"--{boundary}\r\n".encode(), + b'Content-Disposition: form-data; name="file"; filename="audio.wav"\r\n', + b"Content-Type: audio/wav\r\n\r\n", pcm, b"\r\n", + f"--{boundary}\r\n".encode(), + b'Content-Disposition: form-data; name="model"\r\n\r\n', + b"qwen3-asr-0.6b\r\n", + f"--{boundary}\r\n".encode(), + b'Content-Disposition: form-data; name="language"\r\n\r\n', + language.encode(), b"\r\n", + f"--{boundary}--\r\n".encode(), + ]) + request = urllib.request.Request( + SERVER_URL + "/v1/audio/transcriptions", data=body, + headers={"Content-Type": f"multipart/form-data; boundary={boundary}"}, + method="POST", + ) + started = time.monotonic() + with urllib.request.urlopen(request, timeout=60) as response: + payload = json.loads(response.read()) + if not isinstance(payload, dict) or not isinstance(payload.get("text"), str): + raise RuntimeError("Qwen3-ASR returned no transcription") + text = clean_transcript(payload["text"]) + elapsed = int((time.monotonic() - started) * 1000) + log.info("Qwen3-ASR transcribed %d characters in %d ms", len(text), elapsed) + return {"text": text, "language": language, "duration_ms": elapsed, + "engine": "qwen3-asr-0.6b"} + + +def parse_audio(body: bytes, content_type: str) -> tuple[bytes, str, str]: + if "multipart/form-data" not in content_type.lower(): + return body, "audio.wav", LANGUAGE + message = BytesParser(policy=policy.default).parsebytes( + b"MIME-Version: 1.0\r\nContent-Type: " + content_type.encode() + + b"\r\n\r\n" + body + ) + if not message.is_multipart(): + raise ValueError("Invalid multipart upload") + audio = b"" + filename = "audio.wav" + language = LANGUAGE + for part in message.iter_parts(): + name = part.get_param("name", header="content-disposition") + if name == "file": + audio = part.get_payload(decode=True) or b"" + filename = os.path.basename(part.get_filename() or filename) + elif name == "language": + language = (part.get_payload(decode=True) or b"").decode("utf-8").strip() + return audio, filename, language if language and language != "auto" else LANGUAGE + + +class Handler(BaseHTTPRequestHandler): + def send_json(self, status: int, data: dict) -> None: + body = json.dumps(data, ensure_ascii=False).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json; charset=utf-8") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_GET(self) -> None: + if self.path != "/status": + self.send_json(404, {"error": "not found"}) + return + try: + with urllib.request.urlopen(SERVER_URL + "/health", timeout=2) as response: + ready = response.status == 200 + except Exception: + ready = False + self.send_json(200, {"ready": ready, "model": "qwen3-asr-0.6b", + "engine": "qwen3-asr", "language": LANGUAGE}) + + def do_POST(self) -> None: + if self.path != "/transcribe": + self.send_json(404, {"error": "not found"}) + return + try: + size = int(self.headers.get("Content-Length", "0")) + if not 0 < size <= MAX_BODY_BYTES: + self.send_json(413, {"error": "Invalid audio size"}) + return + audio, filename, language = parse_audio( + self.rfile.read(size), self.headers.get("Content-Type", "") + ) + if not audio: + raise ValueError("Missing audio file") + self.send_json(200, transcribe(audio, filename, language)) + except ValueError as exc: + self.send_json(400, {"error": str(exc)}) + except Exception: + log.exception("Transcription failed") + self.send_json(503, {"error": "Qwen3-ASR unavailable"}) + + +if __name__ == "__main__": + ThreadingHTTPServer((HOST, PORT), Handler).serve_forever() diff --git a/services/athena-realtime-voice/README.md b/services/athena-realtime-voice/README.md index ddae30d..ce40286 100644 --- a/services/athena-realtime-voice/README.md +++ b/services/athena-realtime-voice/README.md @@ -1,7 +1,7 @@ # Athena realtime voice bridge This independent service lets OpenClaw's existing browser Talk UI use Athena -Whisper, OpenClaw's agent, and Athena Qwen3-TTS through the browser's supported +Qwen3-ASR, OpenClaw's agent, and Athena Qwen3-TTS through the browser's supported OpenAI-style WebRTC transport. It does not modify OpenClaw or switch an Athena profile. The existing `gateway-relay` path remains available. @@ -90,7 +90,7 @@ python3.11 -m venv .venv ``` The synthetic microphone test passed over the actual OpenClaw HTTPS offer -route and WireGuard media path with production Whisper and Qwen3-TTS on +route and WireGuard media path with production Qwen3-ASR and Qwen3-TTS on 2026-09-16. Subsequent real browser Talk sessions successfully transcribed and answered multiple user turns. Browser dictation uses a separate plugin path. Since version 1.3.0, recordings longer than six seconds are diff --git a/smoke-test.sh b/smoke-test.sh index 7946ec8..cb7b259 100755 --- a/smoke-test.sh +++ b/smoke-test.sh @@ -27,6 +27,8 @@ for name in \ mike-ai-wireguard-gateway \ mike-ai-profile-controller \ mike-ai-embedding \ + mike-ai-qwen-asr \ + mike-ai-qwen-asr-worker \ mike-ai-router \ mike-ai-qwen3-tts \ mike-ai-tts-gateway \ @@ -70,7 +72,8 @@ for legacy in \ mike-ai-mcp-deemix \ mike-ai-mcp-github \ mike-ai-mcp-homeassistant \ - mike-ai-mcp-navidrome; do + mike-ai-mcp-navidrome \ + mike-ai-whisper; do [[ -z $(docker inspect -f '{{.Name}}' "$legacy" 2>/dev/null || true) ]] || \ fail "Altlast existiert noch: $legacy" done