Add local Whisper speech recognition
This commit is contained in:
@@ -12,6 +12,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
||||
- Profile Router auf Port 8081
|
||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
||||
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
|
||||
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
||||
- WireGuard-Gateway, Datenbackup und Athena-Operator
|
||||
@@ -98,6 +99,11 @@ Standard, bis Hermes' Sitzungsfehler behoben ist.
|
||||
|
||||
- Router: `http://192.168.1.212:8081/v1`
|
||||
- Athena-Dashboard: `http://192.168.1.212:8099`
|
||||
|
||||
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
||||
`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das
|
||||
Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
|
||||
werden lokal auf Athena verarbeitet.
|
||||
- Portainer: `https://192.168.1.212:9443`
|
||||
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
||||
|
||||
|
||||
+39
-1
@@ -545,7 +545,9 @@ services:
|
||||
TTS_MODEL: piper
|
||||
TTS_VOICES: alloy
|
||||
TTS_DEFAULT_VOICE: alloy
|
||||
ENABLE_STT: "false"
|
||||
ENABLE_STT: "true"
|
||||
STT_WORKER_URL: http://whisper:8084
|
||||
STT_TIMEOUT: "300"
|
||||
networks: [frontend, control, inference]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
cap_drop: [ALL]
|
||||
@@ -569,6 +571,8 @@ services:
|
||||
condition: service_healthy
|
||||
tts-gateway:
|
||||
condition: service_healthy
|
||||
whisper:
|
||||
condition: service_healthy
|
||||
|
||||
image-worker:
|
||||
build:
|
||||
@@ -710,6 +714,39 @@ services:
|
||||
retries: 12
|
||||
start_period: 10s
|
||||
|
||||
whisper:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: platform/docker/whisper/Dockerfile
|
||||
args:
|
||||
WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1}
|
||||
image: mike-ai/whisper:local
|
||||
container_name: mike-ai-whisper
|
||||
restart: unless-stopped
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp:size=2g,mode=1777
|
||||
volumes:
|
||||
- whisper-data:/models
|
||||
environment:
|
||||
WHISPER_HOST: 0.0.0.0
|
||||
WHISPER_PORT: "8084"
|
||||
WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
|
||||
WHISPER_MODEL: /models/ggml-large-v3-turbo.bin
|
||||
WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
||||
WHISPER_THREADS: ${WHISPER_THREADS:-8}
|
||||
WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
|
||||
networks: [frontend, inference]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
cap_drop: [ALL]
|
||||
cap_add: [CHOWN, SETUID, SETGID]
|
||||
healthcheck:
|
||||
test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 90
|
||||
start_period: 20m
|
||||
|
||||
llama-dashboard:
|
||||
build: ./platform/llama-dashboard
|
||||
image: mike-ai/llama-dashboard:local
|
||||
@@ -809,6 +846,7 @@ networks:
|
||||
|
||||
volumes:
|
||||
piper-data:
|
||||
whisper-data:
|
||||
router-state:
|
||||
router-images:
|
||||
portainer-data:
|
||||
|
||||
@@ -8,6 +8,7 @@ flowchart LR
|
||||
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
|
||||
R --> I[FLUX.2-klein-4B<br/>RTX 5080, Text + Editing]
|
||||
R --> T[XTTS RTX 3060<br/>Piper CPU-Fallback]
|
||||
R --> STT[Whisper.cpp large-v3-turbo<br/>CPU, lokale Spracherkennung]
|
||||
|
||||
H --> U[MUA / Unraid MCP]
|
||||
H --> A[ARR-MCP]
|
||||
|
||||
@@ -2,6 +2,19 @@
|
||||
|
||||
Stand: 3. September 2026
|
||||
|
||||
## Lokale Spracherkennung
|
||||
|
||||
Athena betreibt Whisper.cpp v1.9.1 mit `large-v3-turbo` als CPU-Dienst. Der
|
||||
Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt
|
||||
`/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download
|
||||
bleiben im persistenten Docker-Volume `whisper-data` erhalten.
|
||||
|
||||
Ein lokaler Rundlauftest (Athena-TTS → WAV → Athena-STT) wurde erfolgreich
|
||||
durchgeführt. OpenClaw ist ebenfalls auf diesen lokalen Endpunkt eingestellt
|
||||
und wurde mit `openclaw infer audio transcribe` erfolgreich geprüft. Für den
|
||||
lokalen Provider ist der Zugriff auf Athenas private IP ausdrücklich erlaubt;
|
||||
andere private Ziele werden dadurch nicht freigeschaltet.
|
||||
|
||||
## Produktive llama.cpp-Runtime
|
||||
|
||||
Alle Textprofile verwenden llama.cpp Build 10781,
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
FROM python:3.13.7-slim-bookworm
|
||||
|
||||
ARG WHISPER_CPP_VERSION=v1.9.1
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
build-essential ca-certificates cmake curl ffmpeg git gosu \
|
||||
&& git clone --branch "${WHISPER_CPP_VERSION}" --depth 1 \
|
||||
https://github.com/ggml-org/whisper.cpp.git /opt/whisper.cpp \
|
||||
&& cmake -S /opt/whisper.cpp -B /opt/whisper.cpp/build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DGGML_NATIVE=ON \
|
||||
-DWHISPER_BUILD_TESTS=OFF \
|
||||
-DWHISPER_BUILD_SERVER=OFF \
|
||||
&& cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \
|
||||
&& apt-get purge -y --auto-remove build-essential cmake git \
|
||||
&& rm -rf /var/lib/apt/lists/* /root/.cache \
|
||||
&& useradd --system --uid 10004 --home-dir /nonexistent --shell /usr/sbin/nologin whisper
|
||||
|
||||
WORKDIR /app
|
||||
COPY router/stt_worker.py /app/stt_worker.py
|
||||
COPY platform/docker/whisper/entrypoint.sh /usr/local/bin/mike-ai-whisper-entrypoint
|
||||
RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint
|
||||
|
||||
ENV WHISPER_HOST=0.0.0.0 \
|
||||
WHISPER_PORT=8084 \
|
||||
WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \
|
||||
WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \
|
||||
WHISPER_THREADS=8 \
|
||||
WHISPER_LANGUAGE=de \
|
||||
FFMPEG_BIN=/usr/bin/ffmpeg
|
||||
|
||||
VOLUME ["/models"]
|
||||
EXPOSE 8084
|
||||
ENTRYPOINT ["/usr/local/bin/mike-ai-whisper-entrypoint"]
|
||||
@@ -0,0 +1,25 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin}
|
||||
model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
||||
model_dir=$(dirname "$model")
|
||||
|
||||
mkdir -p "$model_dir"
|
||||
chown 10004:10004 "$model_dir"
|
||||
|
||||
if [ ! -s "$model" ]; then
|
||||
partial="${model}.part"
|
||||
echo "Downloading Whisper model to persistent storage"
|
||||
# All writes below must use the volume owner. The container deliberately
|
||||
# drops CAP_DAC_OVERRIDE, so even uid 0 cannot rename a file in the
|
||||
# whisper-owned directory after capabilities have been removed.
|
||||
if [ ! -s "$partial" ]; then
|
||||
gosu whisper rm -f "$partial"
|
||||
gosu whisper curl --fail --location --retry 5 --retry-delay 5 \
|
||||
--output "$partial" "$model_url"
|
||||
fi
|
||||
gosu whisper mv "$partial" "$model"
|
||||
fi
|
||||
|
||||
exec gosu whisper python /app/stt_worker.py
|
||||
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env python3
|
||||
"""End-to-end smoke test for Athena's OpenAI-compatible TTS and STT APIs."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
import uuid
|
||||
|
||||
|
||||
BASE_URL = os.environ.get("ROUTER_URL", "http://127.0.0.1:8081/v1").rstrip("/")
|
||||
API_KEY = os.environ.get("ROUTER_API_KEY", "")
|
||||
TEST_TEXT = "Dies ist ein lokaler Test der Spracherkennung auf Athena."
|
||||
|
||||
|
||||
def request(path: str, data: bytes, content_type: str) -> bytes:
|
||||
req = urllib.request.Request(
|
||||
f"{BASE_URL}{path}",
|
||||
data=data,
|
||||
headers={
|
||||
"Authorization": f"Bearer {API_KEY}",
|
||||
"Content-Type": content_type,
|
||||
},
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=360) as response:
|
||||
return response.read()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
if not API_KEY:
|
||||
print("ROUTER_API_KEY is required", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
speech = request(
|
||||
"/audio/speech",
|
||||
json.dumps({
|
||||
"model": "piper",
|
||||
"voice": "alloy",
|
||||
"response_format": "wav",
|
||||
"input": TEST_TEXT,
|
||||
}).encode(),
|
||||
"application/json",
|
||||
)
|
||||
|
||||
boundary = f"speech-{uuid.uuid4().hex}"
|
||||
body = (
|
||||
f"--{boundary}\r\n"
|
||||
'Content-Disposition: form-data; name="model"\r\n\r\n'
|
||||
"whisper-1\r\n"
|
||||
f"--{boundary}\r\n"
|
||||
'Content-Disposition: form-data; name="language"\r\n\r\n'
|
||||
"de\r\n"
|
||||
f"--{boundary}\r\n"
|
||||
'Content-Disposition: form-data; name="file"; filename="test.wav"\r\n'
|
||||
"Content-Type: audio/wav\r\n\r\n"
|
||||
).encode() + speech + f"\r\n--{boundary}--\r\n".encode()
|
||||
|
||||
result = json.loads(request(
|
||||
"/audio/transcriptions",
|
||||
body,
|
||||
f"multipart/form-data; boundary={boundary}",
|
||||
))
|
||||
transcript = result.get("text", "").strip()
|
||||
print(transcript)
|
||||
return 0 if transcript else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user