Add local Whisper speech recognition
This commit is contained in:
@@ -12,6 +12,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
|||||||
- Profile Router auf Port 8081
|
- Profile Router auf Port 8081
|
||||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
||||||
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
|
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
|
||||||
|
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||||
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
||||||
- WireGuard-Gateway, Datenbackup und Athena-Operator
|
- WireGuard-Gateway, Datenbackup und Athena-Operator
|
||||||
@@ -98,6 +99,11 @@ Standard, bis Hermes' Sitzungsfehler behoben ist.
|
|||||||
|
|
||||||
- Router: `http://192.168.1.212:8081/v1`
|
- Router: `http://192.168.1.212:8081/v1`
|
||||||
- Athena-Dashboard: `http://192.168.1.212:8099`
|
- Athena-Dashboard: `http://192.168.1.212:8099`
|
||||||
|
|
||||||
|
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
||||||
|
`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das
|
||||||
|
Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
|
||||||
|
werden lokal auf Athena verarbeitet.
|
||||||
- Portainer: `https://192.168.1.212:9443`
|
- Portainer: `https://192.168.1.212:9443`
|
||||||
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
||||||
|
|
||||||
|
|||||||
+39
-1
@@ -545,7 +545,9 @@ services:
|
|||||||
TTS_MODEL: piper
|
TTS_MODEL: piper
|
||||||
TTS_VOICES: alloy
|
TTS_VOICES: alloy
|
||||||
TTS_DEFAULT_VOICE: alloy
|
TTS_DEFAULT_VOICE: alloy
|
||||||
ENABLE_STT: "false"
|
ENABLE_STT: "true"
|
||||||
|
STT_WORKER_URL: http://whisper:8084
|
||||||
|
STT_TIMEOUT: "300"
|
||||||
networks: [frontend, control, inference]
|
networks: [frontend, control, inference]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
cap_drop: [ALL]
|
cap_drop: [ALL]
|
||||||
@@ -569,6 +571,8 @@ services:
|
|||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
tts-gateway:
|
tts-gateway:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
whisper:
|
||||||
|
condition: service_healthy
|
||||||
|
|
||||||
image-worker:
|
image-worker:
|
||||||
build:
|
build:
|
||||||
@@ -710,6 +714,39 @@ services:
|
|||||||
retries: 12
|
retries: 12
|
||||||
start_period: 10s
|
start_period: 10s
|
||||||
|
|
||||||
|
whisper:
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: platform/docker/whisper/Dockerfile
|
||||||
|
args:
|
||||||
|
WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1}
|
||||||
|
image: mike-ai/whisper:local
|
||||||
|
container_name: mike-ai-whisper
|
||||||
|
restart: unless-stopped
|
||||||
|
read_only: true
|
||||||
|
tmpfs:
|
||||||
|
- /tmp:size=2g,mode=1777
|
||||||
|
volumes:
|
||||||
|
- whisper-data:/models
|
||||||
|
environment:
|
||||||
|
WHISPER_HOST: 0.0.0.0
|
||||||
|
WHISPER_PORT: "8084"
|
||||||
|
WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
|
||||||
|
WHISPER_MODEL: /models/ggml-large-v3-turbo.bin
|
||||||
|
WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
||||||
|
WHISPER_THREADS: ${WHISPER_THREADS:-8}
|
||||||
|
WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
|
||||||
|
networks: [frontend, inference]
|
||||||
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
cap_drop: [ALL]
|
||||||
|
cap_add: [CHOWN, SETUID, SETGID]
|
||||||
|
healthcheck:
|
||||||
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 90
|
||||||
|
start_period: 20m
|
||||||
|
|
||||||
llama-dashboard:
|
llama-dashboard:
|
||||||
build: ./platform/llama-dashboard
|
build: ./platform/llama-dashboard
|
||||||
image: mike-ai/llama-dashboard:local
|
image: mike-ai/llama-dashboard:local
|
||||||
@@ -809,6 +846,7 @@ networks:
|
|||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
piper-data:
|
piper-data:
|
||||||
|
whisper-data:
|
||||||
router-state:
|
router-state:
|
||||||
router-images:
|
router-images:
|
||||||
portainer-data:
|
portainer-data:
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ flowchart LR
|
|||||||
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
|
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
|
||||||
R --> I[FLUX.2-klein-4B<br/>RTX 5080, Text + Editing]
|
R --> I[FLUX.2-klein-4B<br/>RTX 5080, Text + Editing]
|
||||||
R --> T[XTTS RTX 3060<br/>Piper CPU-Fallback]
|
R --> T[XTTS RTX 3060<br/>Piper CPU-Fallback]
|
||||||
|
R --> STT[Whisper.cpp large-v3-turbo<br/>CPU, lokale Spracherkennung]
|
||||||
|
|
||||||
H --> U[MUA / Unraid MCP]
|
H --> U[MUA / Unraid MCP]
|
||||||
H --> A[ARR-MCP]
|
H --> A[ARR-MCP]
|
||||||
|
|||||||
@@ -2,6 +2,19 @@
|
|||||||
|
|
||||||
Stand: 3. September 2026
|
Stand: 3. September 2026
|
||||||
|
|
||||||
|
## Lokale Spracherkennung
|
||||||
|
|
||||||
|
Athena betreibt Whisper.cpp v1.9.1 mit `large-v3-turbo` als CPU-Dienst. Der
|
||||||
|
Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt
|
||||||
|
`/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download
|
||||||
|
bleiben im persistenten Docker-Volume `whisper-data` erhalten.
|
||||||
|
|
||||||
|
Ein lokaler Rundlauftest (Athena-TTS → WAV → Athena-STT) wurde erfolgreich
|
||||||
|
durchgeführt. OpenClaw ist ebenfalls auf diesen lokalen Endpunkt eingestellt
|
||||||
|
und wurde mit `openclaw infer audio transcribe` erfolgreich geprüft. Für den
|
||||||
|
lokalen Provider ist der Zugriff auf Athenas private IP ausdrücklich erlaubt;
|
||||||
|
andere private Ziele werden dadurch nicht freigeschaltet.
|
||||||
|
|
||||||
## Produktive llama.cpp-Runtime
|
## Produktive llama.cpp-Runtime
|
||||||
|
|
||||||
Alle Textprofile verwenden llama.cpp Build 10781,
|
Alle Textprofile verwenden llama.cpp Build 10781,
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
FROM python:3.13.7-slim-bookworm
|
||||||
|
|
||||||
|
ARG WHISPER_CPP_VERSION=v1.9.1
|
||||||
|
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
build-essential ca-certificates cmake curl ffmpeg git gosu \
|
||||||
|
&& git clone --branch "${WHISPER_CPP_VERSION}" --depth 1 \
|
||||||
|
https://github.com/ggml-org/whisper.cpp.git /opt/whisper.cpp \
|
||||||
|
&& cmake -S /opt/whisper.cpp -B /opt/whisper.cpp/build \
|
||||||
|
-DCMAKE_BUILD_TYPE=Release \
|
||||||
|
-DGGML_NATIVE=ON \
|
||||||
|
-DWHISPER_BUILD_TESTS=OFF \
|
||||||
|
-DWHISPER_BUILD_SERVER=OFF \
|
||||||
|
&& cmake --build /opt/whisper.cpp/build --config Release -j"$(nproc)" \
|
||||||
|
&& apt-get purge -y --auto-remove build-essential cmake git \
|
||||||
|
&& rm -rf /var/lib/apt/lists/* /root/.cache \
|
||||||
|
&& useradd --system --uid 10004 --home-dir /nonexistent --shell /usr/sbin/nologin whisper
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
COPY router/stt_worker.py /app/stt_worker.py
|
||||||
|
COPY platform/docker/whisper/entrypoint.sh /usr/local/bin/mike-ai-whisper-entrypoint
|
||||||
|
RUN chmod 0755 /usr/local/bin/mike-ai-whisper-entrypoint
|
||||||
|
|
||||||
|
ENV WHISPER_HOST=0.0.0.0 \
|
||||||
|
WHISPER_PORT=8084 \
|
||||||
|
WHISPER_CLI=/opt/whisper.cpp/build/bin/whisper-cli \
|
||||||
|
WHISPER_MODEL=/models/ggml-large-v3-turbo.bin \
|
||||||
|
WHISPER_THREADS=8 \
|
||||||
|
WHISPER_LANGUAGE=de \
|
||||||
|
FFMPEG_BIN=/usr/bin/ffmpeg
|
||||||
|
|
||||||
|
VOLUME ["/models"]
|
||||||
|
EXPOSE 8084
|
||||||
|
ENTRYPOINT ["/usr/local/bin/mike-ai-whisper-entrypoint"]
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
model=${WHISPER_MODEL:-/models/ggml-large-v3-turbo.bin}
|
||||||
|
model_url=${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin}
|
||||||
|
model_dir=$(dirname "$model")
|
||||||
|
|
||||||
|
mkdir -p "$model_dir"
|
||||||
|
chown 10004:10004 "$model_dir"
|
||||||
|
|
||||||
|
if [ ! -s "$model" ]; then
|
||||||
|
partial="${model}.part"
|
||||||
|
echo "Downloading Whisper model to persistent storage"
|
||||||
|
# All writes below must use the volume owner. The container deliberately
|
||||||
|
# drops CAP_DAC_OVERRIDE, so even uid 0 cannot rename a file in the
|
||||||
|
# whisper-owned directory after capabilities have been removed.
|
||||||
|
if [ ! -s "$partial" ]; then
|
||||||
|
gosu whisper rm -f "$partial"
|
||||||
|
gosu whisper curl --fail --location --retry 5 --retry-delay 5 \
|
||||||
|
--output "$partial" "$model_url"
|
||||||
|
fi
|
||||||
|
gosu whisper mv "$partial" "$model"
|
||||||
|
fi
|
||||||
|
|
||||||
|
exec gosu whisper python /app/stt_worker.py
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""End-to-end smoke test for Athena's OpenAI-compatible TTS and STT APIs."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import urllib.request
|
||||||
|
import uuid
|
||||||
|
|
||||||
|
|
||||||
|
BASE_URL = os.environ.get("ROUTER_URL", "http://127.0.0.1:8081/v1").rstrip("/")
|
||||||
|
API_KEY = os.environ.get("ROUTER_API_KEY", "")
|
||||||
|
TEST_TEXT = "Dies ist ein lokaler Test der Spracherkennung auf Athena."
|
||||||
|
|
||||||
|
|
||||||
|
def request(path: str, data: bytes, content_type: str) -> bytes:
|
||||||
|
req = urllib.request.Request(
|
||||||
|
f"{BASE_URL}{path}",
|
||||||
|
data=data,
|
||||||
|
headers={
|
||||||
|
"Authorization": f"Bearer {API_KEY}",
|
||||||
|
"Content-Type": content_type,
|
||||||
|
},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
with urllib.request.urlopen(req, timeout=360) as response:
|
||||||
|
return response.read()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
if not API_KEY:
|
||||||
|
print("ROUTER_API_KEY is required", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
speech = request(
|
||||||
|
"/audio/speech",
|
||||||
|
json.dumps({
|
||||||
|
"model": "piper",
|
||||||
|
"voice": "alloy",
|
||||||
|
"response_format": "wav",
|
||||||
|
"input": TEST_TEXT,
|
||||||
|
}).encode(),
|
||||||
|
"application/json",
|
||||||
|
)
|
||||||
|
|
||||||
|
boundary = f"speech-{uuid.uuid4().hex}"
|
||||||
|
body = (
|
||||||
|
f"--{boundary}\r\n"
|
||||||
|
'Content-Disposition: form-data; name="model"\r\n\r\n'
|
||||||
|
"whisper-1\r\n"
|
||||||
|
f"--{boundary}\r\n"
|
||||||
|
'Content-Disposition: form-data; name="language"\r\n\r\n'
|
||||||
|
"de\r\n"
|
||||||
|
f"--{boundary}\r\n"
|
||||||
|
'Content-Disposition: form-data; name="file"; filename="test.wav"\r\n'
|
||||||
|
"Content-Type: audio/wav\r\n\r\n"
|
||||||
|
).encode() + speech + f"\r\n--{boundary}--\r\n".encode()
|
||||||
|
|
||||||
|
result = json.loads(request(
|
||||||
|
"/audio/transcriptions",
|
||||||
|
body,
|
||||||
|
f"multipart/form-data; boundary={boundary}",
|
||||||
|
))
|
||||||
|
transcript = result.get("text", "").strip()
|
||||||
|
print(transcript)
|
||||||
|
return 0 if transcript else 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
Reference in New Issue
Block a user