diff --git a/README.md b/README.md
index 637f129..61a25ee 100644
--- a/README.md
+++ b/README.md
@@ -12,6 +12,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- Profile Router auf Port 8081
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
+- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
- Portainer CE als optionale Container-Ansicht auf Port 9443
- WireGuard-Gateway, Datenbackup und Athena-Operator
@@ -98,6 +99,11 @@ Standard, bis Hermes' Sitzungsfehler behoben ist.
- Router: `http://192.168.1.212:8081/v1`
- Athena-Dashboard: `http://192.168.1.212:8099`
+
+Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
+`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das
+Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
+werden lokal auf Athena verarbeitet.
- Portainer: `https://192.168.1.212:9443`
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
diff --git a/compose.yaml b/compose.yaml
index 0bd9a0f..ebd1d00 100644
--- a/compose.yaml
+++ b/compose.yaml
@@ -545,7 +545,9 @@ services:
TTS_MODEL: piper
TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy
- ENABLE_STT: "false"
+ ENABLE_STT: "true"
+ STT_WORKER_URL: http://whisper:8084
+ STT_TIMEOUT: "300"
networks: [frontend, control, inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
@@ -569,6 +571,8 @@ services:
condition: service_healthy
tts-gateway:
condition: service_healthy
+ whisper:
+ condition: service_healthy
image-worker:
build:
@@ -710,6 +714,39 @@ services:
retries: 12
start_period: 10s
+ whisper:
+ build:
+ context: .
+ dockerfile: platform/docker/whisper/Dockerfile
+ args:
+ WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1}
+ image: mike-ai/whisper:local
+ container_name: mike-ai-whisper
+ restart: unless-stopped
+ read_only: true
+ tmpfs:
+ - /tmp:size=2g,mode=1777
+ volumes:
+ - whisper-data:/models
+ environment:
+ WHISPER_HOST: 0.0.0.0
+ WHISPER_PORT: "8084"
+ WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
+ WHISPER_MODEL: /models/ggml-large-v3-turbo.bin
+ WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
+ WHISPER_THREADS: ${WHISPER_THREADS:-8}
+ WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
+ networks: [frontend, inference]
+ security_opt: ["no-new-privileges:true"]
+ cap_drop: [ALL]
+ cap_add: [CHOWN, SETUID, SETGID]
+ healthcheck:
+ test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
+ interval: 10s
+ timeout: 5s
+ retries: 90
+ start_period: 20m
+
llama-dashboard:
build: ./platform/llama-dashboard
image: mike-ai/llama-dashboard:local
@@ -809,6 +846,7 @@ networks:
volumes:
piper-data:
+ whisper-data:
router-state:
router-images:
portainer-data:
diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md
index 77f5e22..6fdba27 100644
--- a/docs/ARCHITECTURE.md
+++ b/docs/ARCHITECTURE.md
@@ -8,6 +8,7 @@ flowchart LR
P --> Q[genau ein llama.cpp-Profil
Qwen Fast / Medium / Large / Ultra / Uncensored]
R --> I[FLUX.2-klein-4B
RTX 5080, Text + Editing]
R --> T[XTTS RTX 3060
Piper CPU-Fallback]
+ R --> STT[Whisper.cpp large-v3-turbo
CPU, lokale Spracherkennung]
H --> U[MUA / Unraid MCP]
H --> A[ARR-MCP]
diff --git a/docs/CURRENT_RUNTIME_NOTES.md b/docs/CURRENT_RUNTIME_NOTES.md
index 62bbdbe..2bd1b2c 100644
--- a/docs/CURRENT_RUNTIME_NOTES.md
+++ b/docs/CURRENT_RUNTIME_NOTES.md
@@ -2,6 +2,19 @@
Stand: 3. September 2026
+## Lokale Spracherkennung
+
+Athena betreibt Whisper.cpp v1.9.1 mit `large-v3-turbo` als CPU-Dienst. Der
+Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt
+`/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download
+bleiben im persistenten Docker-Volume `whisper-data` erhalten.
+
+Ein lokaler Rundlauftest (Athena-TTS → WAV → Athena-STT) wurde erfolgreich
+durchgeführt. OpenClaw ist ebenfalls auf diesen lokalen Endpunkt eingestellt
+und wurde mit `openclaw infer audio transcribe` erfolgreich geprüft. Für den
+lokalen Provider ist der Zugriff auf Athenas private IP ausdrücklich erlaubt;
+andere private Ziele werden dadurch nicht freigeschaltet.
+
## Produktive llama.cpp-Runtime
Alle Textprofile verwenden llama.cpp Build 10781,
diff --git a/platform/docker/whisper/Dockerfile b/platform/docker/whisper/Dockerfile
new file mode 100644
index 0000000..8c2925f
--- /dev/null
+++ b/platform/docker/whisper/Dockerfile
@@ -0,0 +1,35 @@
+FROM python:3.13.7-slim-bookworm
+
+ARG WHISPER_CPP_VERSION=v1.9.1
+
+RUN apt-get update \
+ && apt-get install -y --no-install-recommends \
+ build-essential ca-certificates cmake curl ffmpeg git gosu \
+ && git clone --branch "${WHISPER_CPP_VERSION}" --depth 1 \
+ https://github.com/ggml-org/whisper.cpp.git /opt/whisper.cpp \
+ && cmake -S /opt/whisper.cpp -B /opt/whisper.cpp/build \
+ -DCMAKE_BUILD_TYPE=Release \
+ -DGGML_NATIVE=ON \
+ -DWHISPER_BUILD_TESTS=OFF \
+ -DWHISPER_BUILD_SERVER=OFF \
+ && cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \
+ && apt-get purge -y --auto-remove build-essential cmake git \
+ && rm -rf /var/lib/apt/lists/* /root/.cache \
+ && useradd --system --uid 10004 --home-dir /nonexistent --shell /usr/sbin/nologin whisper
+
+WORKDIR /app
+COPY router/stt_worker.py /app/stt_worker.py
+COPY platform/docker/whisper/entrypoint.sh /usr/local/bin/mike-ai-whisper-entrypoint
+RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint
+
+ENV WHISPER_HOST=0.0.0.0 \
+ WHISPER_PORT=8084 \
+ WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \
+ WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \
+ WHISPER_THREADS=8 \
+ WHISPER_LANGUAGE=de \
+ FFMPEG_BIN=/usr/bin/ffmpeg
+
+VOLUME ["/models"]
+EXPOSE 8084
+ENTRYPOINT ["/usr/local/bin/mike-ai-whisper-entrypoint"]
diff --git a/platform/docker/whisper/entrypoint.sh b/platform/docker/whisper/entrypoint.sh
new file mode 100644
index 0000000..2bf53eb
--- /dev/null
+++ b/platform/docker/whisper/entrypoint.sh
@@ -0,0 +1,25 @@
+#!/bin/sh
+set -eu
+
+model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin}
+model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
+model_dir=$(dirname "$model")
+
+mkdir -p "$model_dir"
+chown 10004:10004 "$model_dir"
+
+if [ ! -s "$model" ]; then
+ partial="${model}.part"
+ echo "Downloading Whisper model to persistent storage"
+ # All writes below must use the volume owner. The container deliberately
+ # drops CAP_DAC_OVERRIDE, so even uid 0 cannot rename a file in the
+ # whisper-owned directory after capabilities have been removed.
+ if [ ! -s "$partial" ]; then
+ gosu whisper rm -f "$partial"
+ gosu whisper curl --fail --location --retry 5 --retry-delay 5 \
+ --output "$partial" "$model_url"
+ fi
+ gosu whisper mv "$partial" "$model"
+fi
+
+exec gosu whisper python /app/stt_worker.py
diff --git a/scripts/speech-roundtrip.py b/scripts/speech-roundtrip.py
new file mode 100644
index 0000000..4ac63c8
--- /dev/null
+++ b/scripts/speech-roundtrip.py
@@ -0,0 +1,70 @@
+#!/usr/bin/env python3
+"""End-to-end smoke test for Athena's OpenAI-compatible TTS and STT APIs."""
+
+import json
+import os
+import sys
+import urllib.request
+import uuid
+
+
+BASE_URL = os.environ.get("ROUTER_URL", "http://127.0.0.1:8081/v1").rstrip("/")
+API_KEY = os.environ.get("ROUTER_API_KEY", "")
+TEST_TEXT = "Dies ist ein lokaler Test der Spracherkennung auf Athena."
+
+
+def request(path: str, data: bytes, content_type: str) -> bytes:
+ req = urllib.request.Request(
+ f"{BASE_URL}{path}",
+ data=data,
+ headers={
+ "Authorization": f"Bearer {API_KEY}",
+ "Content-Type": content_type,
+ },
+ method="POST",
+ )
+ with urllib.request.urlopen(req, timeout=360) as response:
+ return response.read()
+
+
+def main() -> int:
+ if not API_KEY:
+ print("ROUTER_API_KEY is required", file=sys.stderr)
+ return 2
+
+ speech = request(
+ "/audio/speech",
+ json.dumps({
+ "model": "piper",
+ "voice": "alloy",
+ "response_format": "wav",
+ "input": TEST_TEXT,
+ }).encode(),
+ "application/json",
+ )
+
+ boundary = f"speech-{uuid.uuid4().hex}"
+ body = (
+ f"--{boundary}\r\n"
+ 'Content-Disposition: form-data; name="model"\r\n\r\n'
+ "whisper-1\r\n"
+ f"--{boundary}\r\n"
+ 'Content-Disposition: form-data; name="language"\r\n\r\n'
+ "de\r\n"
+ f"--{boundary}\r\n"
+ 'Content-Disposition: form-data; name="file"; filename="test.wav"\r\n'
+ "Content-Type: audio/wav\r\n\r\n"
+ ).encode() + speech + f"\r\n--{boundary}--\r\n".encode()
+
+ result = json.loads(request(
+ "/audio/transcriptions",
+ body,
+ f"multipart/form-data; boundary={boundary}",
+ ))
+ transcript = result.get("text", "").strip()
+ print(transcript)
+ return 0 if transcript else 1
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())