diff --git a/README.md b/README.md index 637f129..61a25ee 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - Profile Router auf Port 8081 - FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080 - XTTS auf der RTX 3060 mit Piper als CPU-Fallback +- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 - Portainer CE als optionale Container-Ansicht auf Port 9443 - WireGuard-Gateway, Datenbackup und Athena-Operator @@ -98,6 +99,11 @@ Standard, bis Hermes' Sitzungsfehler behoben ist. - Router: `http://192.168.1.212:8081/v1` - Athena-Dashboard: `http://192.168.1.212:8099` + +Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über +`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das +Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten +werden lokal auf Athena verarbeitet. - Portainer: `https://192.168.1.212:9443` - Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` diff --git a/compose.yaml b/compose.yaml index 0bd9a0f..ebd1d00 100644 --- a/compose.yaml +++ b/compose.yaml @@ -545,7 +545,9 @@ services: TTS_MODEL: piper TTS_VOICES: alloy TTS_DEFAULT_VOICE: alloy - ENABLE_STT: "false" + ENABLE_STT: "true" + STT_WORKER_URL: http://whisper:8084 + STT_TIMEOUT: "300" networks: [frontend, control, inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] @@ -569,6 +571,8 @@ services: condition: service_healthy tts-gateway: condition: service_healthy + whisper: + condition: service_healthy image-worker: build: @@ -710,6 +714,39 @@ services: retries: 12 start_period: 10s + whisper: + build: + context: . + dockerfile: platform/docker/whisper/Dockerfile + args: + WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1} + image: mike-ai/whisper:local + container_name: mike-ai-whisper + restart: unless-stopped + read_only: true + tmpfs: + - /tmp:size=2g,mode=1777 + volumes: + - whisper-data:/models + environment: + WHISPER_HOST: 0.0.0.0 + WHISPER_PORT: "8084" + WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli + WHISPER_MODEL: /models/ggml-large-v3-turbo.bin + WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin} + WHISPER_THREADS: ${WHISPER_THREADS:-8} + WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de} + networks: [frontend, inference] + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + cap_add: [CHOWN, SETUID, SETGID] + healthcheck: + test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"] + interval: 10s + timeout: 5s + retries: 90 + start_period: 20m + llama-dashboard: build: ./platform/llama-dashboard image: mike-ai/llama-dashboard:local @@ -809,6 +846,7 @@ networks: volumes: piper-data: + whisper-data: router-state: router-images: portainer-data: diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 77f5e22..6fdba27 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -8,6 +8,7 @@ flowchart LR P --> Q[genau ein llama.cpp-Profil
Qwen Fast / Medium / Large / Ultra / Uncensored] R --> I[FLUX.2-klein-4B
RTX 5080, Text + Editing] R --> T[XTTS RTX 3060
Piper CPU-Fallback] + R --> STT[Whisper.cpp large-v3-turbo
CPU, lokale Spracherkennung] H --> U[MUA / Unraid MCP] H --> A[ARR-MCP] diff --git a/docs/CURRENT_RUNTIME_NOTES.md b/docs/CURRENT_RUNTIME_NOTES.md index 62bbdbe..2bd1b2c 100644 --- a/docs/CURRENT_RUNTIME_NOTES.md +++ b/docs/CURRENT_RUNTIME_NOTES.md @@ -2,6 +2,19 @@ Stand: 3. September 2026 +## Lokale Spracherkennung + +Athena betreibt Whisper.cpp v1.9.1 mit `large-v3-turbo` als CPU-Dienst. Der +Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt +`/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download +bleiben im persistenten Docker-Volume `whisper-data` erhalten. + +Ein lokaler Rundlauftest (Athena-TTS → WAV → Athena-STT) wurde erfolgreich +durchgeführt. OpenClaw ist ebenfalls auf diesen lokalen Endpunkt eingestellt +und wurde mit `openclaw infer audio transcribe` erfolgreich geprüft. Für den +lokalen Provider ist der Zugriff auf Athenas private IP ausdrücklich erlaubt; +andere private Ziele werden dadurch nicht freigeschaltet. + ## Produktive llama.cpp-Runtime Alle Textprofile verwenden llama.cpp Build 10781, diff --git a/platform/docker/whisper/Dockerfile b/platform/docker/whisper/Dockerfile new file mode 100644 index 0000000..8c2925f --- /dev/null +++ b/platform/docker/whisper/Dockerfile @@ -0,0 +1,35 @@ +FROM python:3.13.7-slim-bookworm + +ARG WHISPER_CPP_VERSION=v1.9.1 + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + build-essential ca-certificates cmake curl ffmpeg git gosu \ + && git clone --branch "${WHISPER_CPP_VERSION}" --depth 1 \ + https://github.com/ggml-org/whisper.cpp.git /opt/whisper.cpp \ + && cmake -S /opt/whisper.cpp -B /opt/whisper.cpp/build \ + -DCMAKE_BUILD_TYPE=Release \ + -DGGML_NATIVE=ON \ + -DWHISPER_BUILD_TESTS=OFF \ + -DWHISPER_BUILD_SERVER=OFF \ + && cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \ + && apt-get purge -y --auto-remove build-essential cmake git \ + && rm -rf /var/lib/apt/lists/* /root/.cache \ + && useradd --system --uid 10004 --home-dir /nonexistent --shell /usr/sbin/nologin whisper + +WORKDIR /app +COPY router/stt_worker.py /app/stt_worker.py +COPY platform/docker/whisper/entrypoint.sh /usr/local/bin/mike-ai-whisper-entrypoint +RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint + +ENV WHISPER_HOST=0.0.0.0 \ + WHISPER_PORT=8084 \ + WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \ + WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \ + WHISPER_THREADS=8 \ + WHISPER_LANGUAGE=de \ + FFMPEG_BIN=/usr/bin/ffmpeg + +VOLUME ["/models"] +EXPOSE 8084 +ENTRYPOINT ["/usr/local/bin/mike-ai-whisper-entrypoint"] diff --git a/platform/docker/whisper/entrypoint.sh b/platform/docker/whisper/entrypoint.sh new file mode 100644 index 0000000..2bf53eb --- /dev/null +++ b/platform/docker/whisper/entrypoint.sh @@ -0,0 +1,25 @@ +#!/bin/sh +set -eu + +model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin} +model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin} +model_dir=$(dirname "$model") + +mkdir -p "$model_dir" +chown 10004:10004 "$model_dir" + +if [ ! -s "$model" ]; then + partial="${model}.part" + echo "Downloading Whisper model to persistent storage" + # All writes below must use the volume owner. The container deliberately + # drops CAP_DAC_OVERRIDE, so even uid 0 cannot rename a file in the + # whisper-owned directory after capabilities have been removed. + if [ ! -s "$partial" ]; then + gosu whisper rm -f "$partial" + gosu whisper curl --fail --location --retry 5 --retry-delay 5 \ + --output "$partial" "$model_url" + fi + gosu whisper mv "$partial" "$model" +fi + +exec gosu whisper python /app/stt_worker.py diff --git a/scripts/speech-roundtrip.py b/scripts/speech-roundtrip.py new file mode 100644 index 0000000..4ac63c8 --- /dev/null +++ b/scripts/speech-roundtrip.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""End-to-end smoke test for Athena's OpenAI-compatible TTS and STT APIs.""" + +import json +import os +import sys +import urllib.request +import uuid + + +BASE_URL = os.environ.get("ROUTER_URL", "http://127.0.0.1:8081/v1").rstrip("/") +API_KEY = os.environ.get("ROUTER_API_KEY", "") +TEST_TEXT = "Dies ist ein lokaler Test der Spracherkennung auf Athena." + + +def request(path: str, data: bytes, content_type: str) -> bytes: + req = urllib.request.Request( + f"{BASE_URL}{path}", + data=data, + headers={ + "Authorization": f"Bearer {API_KEY}", + "Content-Type": content_type, + }, + method="POST", + ) + with urllib.request.urlopen(req, timeout=360) as response: + return response.read() + + +def main() -> int: + if not API_KEY: + print("ROUTER_API_KEY is required", file=sys.stderr) + return 2 + + speech = request( + "/audio/speech", + json.dumps({ + "model": "piper", + "voice": "alloy", + "response_format": "wav", + "input": TEST_TEXT, + }).encode(), + "application/json", + ) + + boundary = f"speech-{uuid.uuid4().hex}" + body = ( + f"--{boundary}\r\n" + 'Content-Disposition: form-data; name="model"\r\n\r\n' + "whisper-1\r\n" + f"--{boundary}\r\n" + 'Content-Disposition: form-data; name="language"\r\n\r\n' + "de\r\n" + f"--{boundary}\r\n" + 'Content-Disposition: form-data; name="file"; filename="test.wav"\r\n' + "Content-Type: audio/wav\r\n\r\n" + ).encode() + speech + f"\r\n--{boundary}--\r\n".encode() + + result = json.loads(request( + "/audio/transcriptions", + body, + f"multipart/form-data; boundary={boundary}", + )) + transcript = result.get("text", "").strip() + print(transcript) + return 0 if transcript else 1 + + +if __name__ == "__main__": + raise SystemExit(main())