Add reproducible Piper TTS service
This commit is contained in:
@@ -25,33 +25,42 @@ else
|
||||
fail "nvidia-smi fehlt"
|
||||
fi
|
||||
|
||||
for service in mike-ai-llama-ui mike-ai-profile-router; do
|
||||
if systemctl is-active --quiet "$service"; then
|
||||
pass "$service aktiv"
|
||||
container_healthy() {
|
||||
local name=$1 state health
|
||||
state="$(docker inspect --format '{{.State.Status}}' "$name" 2>/dev/null || true)"
|
||||
health="$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$name" 2>/dev/null || true)"
|
||||
[[ $state == running && ( -z $health || $health == healthy ) ]]
|
||||
}
|
||||
|
||||
for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai-open-webui; do
|
||||
if container_healthy "$container"; then
|
||||
pass "$container gesund"
|
||||
else
|
||||
fail "$service nicht aktiv"
|
||||
fail "$container fehlt oder ist nicht gesund"
|
||||
fi
|
||||
done
|
||||
|
||||
for optional in mike-ai-whisper mike-ai-xtts mike-ai-web-search; do
|
||||
if systemctl is-active --quiet "$optional"; then
|
||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|long|experimental)$' || true)"
|
||||
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
|
||||
pass "exakt ein llama.cpp-Profil aktiv"
|
||||
else
|
||||
fail "$ACTIVE_LLAMA llama.cpp-Profile aktiv (erwartet: 1)"
|
||||
fi
|
||||
|
||||
for optional in mike-ai-mcp-web mike-ai-mcp-homeassistant mike-ai-mcp-arr mike-ai-mcp-unraid-official; do
|
||||
if container_healthy "$optional"; then
|
||||
pass "$optional aktiv"
|
||||
else
|
||||
warn "$optional nicht aktiv oder nicht installiert"
|
||||
fi
|
||||
done
|
||||
|
||||
if curl -fsS --max-time 3 http://127.0.0.1:8080/health >/dev/null; then
|
||||
pass "llama.cpp Health-Check"
|
||||
if docker exec mike-ai-router python -c \
|
||||
"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=3)" \
|
||||
>/dev/null 2>&1; then
|
||||
pass "Router-Liveness intern erreichbar"
|
||||
else
|
||||
fail "llama.cpp auf Port 8080 nicht gesund"
|
||||
fi
|
||||
|
||||
if STATUS="$(curl -fsS --max-time 3 http://127.0.0.1:8081/status 2>/dev/null)"; then
|
||||
PROFILE="$(python3 -c 'import json,sys; print(json.load(sys.stdin).get("current_profile"))' <<<"$STATUS" 2>/dev/null)"
|
||||
pass "Router erreichbar, Profil ${PROFILE:-unbekannt}"
|
||||
else
|
||||
fail "Router auf Port 8081 nicht erreichbar"
|
||||
fail "Router-Liveness intern nicht erreichbar"
|
||||
fi
|
||||
|
||||
if systemctl list-unit-files --no-legend 2>/dev/null | grep -Eq '(vision-rx|whisper-rx|granite-rx).*enabled'; then
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
FROM python:3.12-slim-bookworm
|
||||
|
||||
ARG PIPER_TTS_VERSION=1.6.0
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl ffmpeg gosu \
|
||||
&& python -m pip install --no-cache-dir "piper-tts==${PIPER_TTS_VERSION}" \
|
||||
&& useradd --system --uid 10003 --home-dir /nonexistent --shell /usr/sbin/nologin piper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
COPY piper_worker.py /app/piper_worker.py
|
||||
COPY entrypoint.sh /usr/local/bin/mike-ai-piper-entrypoint
|
||||
RUN chmod 0755 /usr/local/bin/mike-ai-piper-entrypoint
|
||||
|
||||
ENV PIPER_DATA_DIR=/data \
|
||||
PIPER_VOICE=de_DE-thorsten-high \
|
||||
PIPER_VOICE_ALIAS=alloy \
|
||||
PIPER_HOST=0.0.0.0 \
|
||||
PIPER_PORT=8085
|
||||
|
||||
VOLUME ["/data"]
|
||||
EXPOSE 8085
|
||||
ENTRYPOINT ["/usr/local/bin/mike-ai-piper-entrypoint"]
|
||||
@@ -0,0 +1,15 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
data_dir=${PIPER_DATA_DIR:-/data}
|
||||
voice=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||
|
||||
mkdir -p "$data_dir"
|
||||
chown 10003:10003 "$data_dir"
|
||||
|
||||
if [ ! -s "$data_dir/$voice.onnx" ] || [ ! -s "$data_dir/$voice.onnx.json" ]; then
|
||||
echo "Downloading Piper voice: $voice"
|
||||
gosu piper python -m piper.download_voices --data-dir "$data_dir" "$voice"
|
||||
fi
|
||||
|
||||
exec gosu piper python /app/piper_worker.py
|
||||
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small, private Piper worker for the Mike AI profile router.
|
||||
|
||||
The public OpenAI-compatible endpoint remains in the router. This worker only
|
||||
accepts the narrow internal /status and /tts protocol and never logs input text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import threading
|
||||
import wave
|
||||
from http import HTTPStatus
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
|
||||
from piper import PiperVoice, SynthesisConfig
|
||||
|
||||
|
||||
DATA_DIR = Path(os.getenv("PIPER_DATA_DIR", "/data"))
|
||||
VOICE_NAME = os.getenv("PIPER_VOICE", "de_DE-thorsten-high")
|
||||
VOICE_ALIAS = os.getenv("PIPER_VOICE_ALIAS", "alloy")
|
||||
HOST = os.getenv("PIPER_HOST", "0.0.0.0")
|
||||
PORT = int(os.getenv("PIPER_PORT", "8085"))
|
||||
MAX_TEXT_CHARS = int(os.getenv("PIPER_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("PIPER_MAX_REQUEST_BYTES", "65536"))
|
||||
|
||||
VOICE_PATH = DATA_DIR / f"{VOICE_NAME}.onnx"
|
||||
VOICE = PiperVoice.load(str(VOICE_PATH))
|
||||
SYNTHESIS_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def synthesize_wav(text: str, speed: float) -> bytes:
|
||||
"""Synthesize a complete WAV in memory without retaining the text."""
|
||||
output = io.BytesIO()
|
||||
config = SynthesisConfig(length_scale=1.0 / speed)
|
||||
with SYNTHESIS_LOCK, wave.open(output, "wb") as wav_file:
|
||||
VOICE.synthesize_wav(text, wav_file, syn_config=config)
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def wav_to_mp3(wav_bytes: bytes) -> bytes:
|
||||
"""Convert Piper's WAV to the MP3 format Open WebUI requests by default."""
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-loglevel", "error",
|
||||
"-f", "wav", "-i", "pipe:0",
|
||||
"-codec:a", "libmp3lame", "-b:a", "96k",
|
||||
"-f", "mp3", "pipe:1",
|
||||
],
|
||||
input=wav_bytes,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
check=False,
|
||||
timeout=120,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError("ffmpeg conversion failed")
|
||||
return result.stdout
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
protocol_version = "HTTP/1.1"
|
||||
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
# Deliberately omit URLs and request bodies from the log.
|
||||
print(f"piper-worker: {self.command} -> {args[1] if len(args) > 1 else '-'}")
|
||||
|
||||
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", content_type)
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def send_json(self, status: int, payload: dict) -> None:
|
||||
self.send_bytes(
|
||||
status,
|
||||
json.dumps(payload, separators=(",", ":")).encode(),
|
||||
"application/json",
|
||||
)
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
if self.path != "/status":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
self.send_json(
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": True,
|
||||
"engine": "piper",
|
||||
"model": VOICE_NAME,
|
||||
"voices": [VOICE_ALIAS],
|
||||
},
|
||||
)
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.path != "/tts":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
|
||||
try:
|
||||
content_length = int(self.headers.get("Content-Length", "0"))
|
||||
except ValueError:
|
||||
content_length = 0
|
||||
if content_length <= 0 or content_length > MAX_REQUEST_BYTES:
|
||||
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, {"error": "invalid request size"})
|
||||
return
|
||||
|
||||
try:
|
||||
request = json.loads(self.rfile.read(content_length))
|
||||
text = request.get("text", "")
|
||||
voice = request.get("voice", VOICE_ALIAS)
|
||||
output_format = request.get("format", "mp3")
|
||||
speed = float(request.get("speed", 1.0))
|
||||
except (json.JSONDecodeError, TypeError, ValueError):
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
|
||||
return
|
||||
|
||||
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
|
||||
return
|
||||
if voice != VOICE_ALIAS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||
return
|
||||
if output_format not in {"wav", "mp3"}:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||
return
|
||||
if not 0.5 <= speed <= 2.0:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
|
||||
return
|
||||
|
||||
try:
|
||||
audio = synthesize_wav(text.strip(), speed)
|
||||
if output_format == "mp3":
|
||||
audio = wav_to_mp3(audio)
|
||||
content_type = "audio/mpeg"
|
||||
else:
|
||||
content_type = "audio/wav"
|
||||
except (OSError, RuntimeError, subprocess.SubprocessError):
|
||||
self.send_json(HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "synthesis failed"})
|
||||
return
|
||||
|
||||
self.send_bytes(HTTPStatus.OK, audio, content_type)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"Piper worker ready: {VOICE_NAME} as {VOICE_ALIAS} on {HOST}:{PORT}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
Reference in New Issue
Block a user