router: deutsche Sprachausgabe mit Kokoro-82M (CPU-only)
Fügt einen OpenAI-kompatiblen TTS-Endpunkt POST /v1/audio/speech hinzu. Die Synthese läuft in einem separaten, langlebigen Worker (mike-ai-kokoro.service) mit eigenem Venv (CPU-only torch) und hält die Modelle dauerhaft im RAM (niedrige Warm-Start-Latenz). - Zwei deutsche Stimmen: kikiri-german-martin, kikiri-german-victoria (Apache 2.0, je ~327 MB) unter /opt/mike-ai/models/kokoro/ - Deutsche G2P über espeak-ng (phonemizer), kein spacy/thinc 9.x nötig (kokoro mit --no-deps + misaki ohne [en], Python 3.13-kompatibel) - Formate: mp3 (Default), wav, flac, pcm; speed 0.5-2.0 - /status um tts.*-Felder erweitert (reachable, ready, voices, ...) - TTS ohne GPU-Lock: blockiert weder Qwen/llama.cpp noch FLUX - systemd-Unit mike-ai-kokoro.service (Start beim Boot) - install.sh/deploy.sh um Kokoro-Venv + Modell-Download erweitert - Mock-TTS-Worker + 11 TTS-Tests (insgesamt 43, alle bestanden) - Hörproben (je ~40 s) + Benchmark (RTF ~0.23, ~4.3x Echtzeit) Kein Push – erst nach User-Freigabe.
This commit is contained in:
+99
-1
@@ -6,13 +6,14 @@ cd "$(dirname "$0")/.."
|
||||
|
||||
UP_PORT=18080
|
||||
RT_PORT=18081
|
||||
TTS_PORT=18082
|
||||
BASE="http://127.0.0.1:$RT_PORT"
|
||||
FAKE_DIR="$PWD/dev/fake-profile-dir"
|
||||
PASS=0
|
||||
FAIL=0
|
||||
|
||||
cleanup() {
|
||||
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" 2>/dev/null || true
|
||||
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" "${TTS_PID:-}" 2>/dev/null || true
|
||||
rm -f /tmp/mock_pid2 /tmp/mock_upstream_pid
|
||||
wait 2>/dev/null || true
|
||||
}
|
||||
@@ -52,11 +53,21 @@ IMAGE_DIR=/tmp/test-images \
|
||||
IMAGE_WORKER_LOG=/tmp/test_worker.log \
|
||||
IMAGE_GEN_TIMEOUT=30 \
|
||||
MOCK_WORKER_LOG=/tmp/test_worker_requests.jsonl \
|
||||
TTS_WORKER_URL="http://127.0.0.1:$TTS_PORT" \
|
||||
python3 router/ai_profile_router.py >/tmp/router_test.log 2>&1 &
|
||||
ROUTER_PID=$!
|
||||
sleep 0.5
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
|
||||
# --- Mock-TTS-Worker starten ----------------------------------------------------
|
||||
echo "== Starte Mock-TTS-Worker (Port $TTS_PORT)"
|
||||
MOCK_TTS_PORT="$TTS_PORT" MOCK_TTS_DELAY=0.1 \
|
||||
MOCK_TTS_LOG=/tmp/test_tts_requests.jsonl \
|
||||
python3 dev/mock_tts_worker.py >/tmp/mock_tts.log 2>&1 &
|
||||
TTS_PID=$!
|
||||
sleep 0.5
|
||||
rm -f /tmp/test_tts_requests.jsonl
|
||||
|
||||
# --- 1. /v1/models -------------------------------------------------------------
|
||||
echo "== Test 1: /v1/models"
|
||||
RESP=$(curl -sf "$BASE/v1/models")
|
||||
@@ -395,6 +406,93 @@ wait $IMG_PID
|
||||
[ "$CODE" = "200" ] && [ "$ELAPSED" -ge 2 ] \
|
||||
&& ok "Chat wartete ${ELAPSED}s (kein 502), dann 200" || bad "Chat: Code $CODE, ${ELAPSED}s"
|
||||
|
||||
# --- 27. TTS: /status zeigt tts-Section ---------------------------------------------------------------
|
||||
echo "== Test 27: /status mit tts-Section"
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
tts=d["tts"]
|
||||
assert tts["reachable"] is True, tts
|
||||
assert tts["ready"] is True, tts
|
||||
assert set(tts["voices"])=={"martin","victoria"}, tts
|
||||
' && ok "Status: TTS erreichbar, bereit, 2 Stimmen" || bad "Status tts-Section"
|
||||
|
||||
# --- 28. TTS: POST /v1/audio/speech (wav) ---------------------------------------------------------------
|
||||
echo "== Test 28: POST /v1/audio/speech (wav)"
|
||||
CODE=$(curl -s -o /tmp/tts28.wav -w "%{http_code}" -D /tmp/hdr28.txt \
|
||||
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \
|
||||
-d '{"model":"kokoro-german","input":"Hallo Welt","voice":"martin","response_format":"wav"}')
|
||||
CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r")
|
||||
[ "$CODE" = "200" ] && [ -s /tmp/tts28.wav ] && echo "$CTYPE" | grep -qi "audio/wav" \
|
||||
&& ok "TTS wav (200, $CTYPE, $(stat -f%z /tmp/tts28.wav 2>/dev/null || stat -c%s /tmp/tts28.wav) Bytes)" \
|
||||
|| bad "TTS wav (Code $CODE, $CTYPE)"
|
||||
|
||||
# --- 29. TTS: POST /v1/audio/speech (mp3, Default) -------------------------------------------------------
|
||||
echo "== Test 29: POST /v1/audio/speech (mp3, Default)"
|
||||
CODE=$(curl -s -o /tmp/tts29.mp3 -w "%{http_code}" -D /tmp/hdr29.txt \
|
||||
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \
|
||||
-d '{"input":"Guten Tag","voice":"victoria"}')
|
||||
CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r")
|
||||
[ "$CODE" = "200" ] && [ -s /tmp/tts29.mp3 ] && echo "$CTYPE" | grep -qi "audio/mpeg" \
|
||||
&& ok "TTS mp3 (200, $CTYPE)" || bad "TTS mp3 (Code $CODE, $CTYPE)"
|
||||
|
||||
# --- 30. TTS: Validierung --------------------------------------------------------------------------------
|
||||
echo "== Test 30: TTS-Validierung"
|
||||
CODE=$(curl -s -o /tmp/err30a.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"voice":"martin"}')
|
||||
cat /tmp/err30a.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei fehlendem input" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err30b.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"x","voice":"bogus"}')
|
||||
cat /tmp/err30b.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültiger Stimme" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err30c.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"x","response_format":"ogg"}')
|
||||
cat /tmp/err30c.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültigem Format" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err30d.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"x","model":"gpt-4"}')
|
||||
cat /tmp/err30d.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei unbekanntem Modell" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
# --- 31. TTS: Worker-Fehler → 503 ------------------------------------------------------------------------
|
||||
echo "== Test 31: TTS-Worker-Fehler → 503"
|
||||
CODE=$(curl -s -o /tmp/err31.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"FAIL","voice":"martin"}')
|
||||
cat /tmp/err31.json; echo
|
||||
[ "$CODE" = "503" ] && ok "503 bei TTS-Worker-Fehler" || bad "erwartet 503, bekam $CODE"
|
||||
|
||||
# --- 32. TTS: Worker down → 503 ---------------------------------------------------------------------------
|
||||
echo "== Test 32: TTS-Worker down → 503"
|
||||
kill "$TTS_PID" 2>/dev/null || true
|
||||
sleep 0.5
|
||||
CODE=$(curl -s -o /tmp/err32.json -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"Hallo","voice":"martin"}')
|
||||
cat /tmp/err32.json; echo
|
||||
[ "$CODE" = "503" ] && ok "503 bei downem TTS-Worker" || bad "erwartet 503, bekam $CODE"
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["tts"]["reachable"] is False, d["tts"]
|
||||
' && ok "Status: TTS nicht erreichbar" || bad "Status nach TTS-Down"
|
||||
|
||||
# --- 33. TTS: Worker-Neustart → Recovery -------------------------------------------------------------------
|
||||
echo "== Test 33: TTS-Worker-Neustart → Recovery"
|
||||
MOCK_TTS_PORT="$TTS_PORT" MOCK_TTS_DELAY=0.1 \
|
||||
python3 dev/mock_tts_worker.py >/tmp/mock_tts2.log 2>&1 &
|
||||
TTS_PID=$!
|
||||
sleep 0.5
|
||||
CODE=$(curl -s -o /tmp/tts33.wav -w "%{http_code}" "$BASE/v1/audio/speech" \
|
||||
-H "Content-Type: application/json" -d '{"input":"Wieder da","voice":"martin","response_format":"wav"}')
|
||||
[ "$CODE" = "200" ] && [ -s /tmp/tts33.wav ] \
|
||||
&& ok "TTS nach Neustart wieder verfügbar" || bad "TTS-Recovery (Code $CODE)"
|
||||
|
||||
# --- Ergebnis --------------------------------------------------------------------------------------------
|
||||
echo
|
||||
echo "== Ergebnis: $PASS bestanden, $FAIL fehlgeschlagen =="
|
||||
|
||||
Reference in New Issue
Block a user