TTS: Kokoro → XTTS-v2 (CPU-only, Claribel Dervla)

- Neues xtts_worker.py: Coqui XTTS-v2, HTTP-API auf Port 8085
- Router: TTS_WORKER_URL → 8085, TTS_MODEL → xtts-v2, TTS_VOICES → claribel
- deploy: mike-ai-xtts.service, install.sh + deploy.sh aktualisiert
- Tests: 54/54 bestanden (mock_tts_worker + test_local.sh auf XTTS umgestellt)
- README: TTS-Section auf XTTS-v2 aktualisiert
- Kokoro-Service gestoppt und deaktiviert (Dateien bleiben als Backup)
This commit is contained in:
Mikei386
2026-08-19 22:01:58 +02:00
parent ed471196d4
commit d399d2b4f7
9 changed files with 451 additions and 171 deletions
+12 -13
View File
@@ -427,14 +427,14 @@ d=json.load(sys.stdin)
tts=d["tts"]
assert tts["reachable"] is True, tts
assert tts["ready"] is True, tts
assert set(tts["voices"])=={"martin","victoria"}, tts
assert set(tts["voices"])=={"claribel"}, tts
' && ok "Status: TTS erreichbar, bereit, 2 Stimmen" || bad "Status tts-Section"
# --- 28. TTS: POST /v1/audio/speech (wav) ---------------------------------------------------------------
echo "== Test 28: POST /v1/audio/speech (wav)"
CODE=$(curl -s -o /tmp/tts28.wav -w "%{http_code}" -D /tmp/hdr28.txt \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"model":"kokoro-german","input":"Hallo Welt","voice":"martin","response_format":"wav"}')
-d '{"model":"xtts-v2","input":"Hallo Welt","voice":"claribel","response_format":"wav"}')
CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r")
[ "$CODE" = "200" ] && [ -s /tmp/tts28.wav ] && echo "$CTYPE" | grep -qi "audio/wav" \
&& ok "TTS wav (200, $CTYPE, $(stat -f%z /tmp/tts28.wav 2>/dev/null || stat -c%s /tmp/tts28.wav) Bytes)" \
@@ -444,7 +444,7 @@ CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r")
echo "== Test 29: POST /v1/audio/speech (mp3, Default)"
CODE=$(curl -s -o /tmp/tts29.mp3 -w "%{http_code}" -D /tmp/hdr29.txt \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"input":"Guten Tag","voice":"victoria"}')
-d '{"input":"Guten Tag","voice":"claribel"}')
CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r")
[ "$CODE" = "200" ] && [ -s /tmp/tts29.mp3 ] && echo "$CTYPE" | grep -qi "audio/mpeg" \
&& ok "TTS mp3 (200, $CTYPE)" || bad "TTS mp3 (Code $CODE, $CTYPE)"
@@ -452,7 +452,7 @@ CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r")
# --- 30. TTS: Validierung --------------------------------------------------------------------------------
echo "== Test 30: TTS-Validierung"
CODE=$(curl -s -o /tmp/err30a.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"voice":"martin"}')
-H "Content-Type: application/json" -d '{"voice":"claribel"}')
cat /tmp/err30a.json; echo
[ "$CODE" = "400" ] && ok "400 bei fehlendem input" || bad "erwartet 400, bekam $CODE"
@@ -474,7 +474,7 @@ cat /tmp/err30d.json; echo
# --- 31. TTS: Worker-Fehler → 503 ------------------------------------------------------------------------
echo "== Test 31: TTS-Worker-Fehler → 503"
CODE=$(curl -s -o /tmp/err31.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"FAIL","voice":"martin"}')
-H "Content-Type: application/json" -d '{"input":"FAIL","voice":"claribel"}')
cat /tmp/err31.json; echo
[ "$CODE" = "503" ] && ok "503 bei TTS-Worker-Fehler" || bad "erwartet 503, bekam $CODE"
@@ -483,7 +483,7 @@ echo "== Test 32: TTS-Worker down → 503"
kill "$TTS_PID" 2>/dev/null || true
sleep 0.5
CODE=$(curl -s -o /tmp/err32.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"Hallo","voice":"martin"}')
-H "Content-Type: application/json" -d '{"input":"Hallo","voice":"claribel"}')
cat /tmp/err32.json; echo
[ "$CODE" = "503" ] && ok "503 bei downem TTS-Worker" || bad "erwartet 503, bekam $CODE"
RESP=$(curl -sf "$BASE/status")
@@ -500,7 +500,7 @@ MOCK_TTS_PORT="$TTS_PORT" MOCK_TTS_DELAY=0.1 \
TTS_PID=$!
sleep 0.5
CODE=$(curl -s -o /tmp/tts33.wav -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"Wieder da","voice":"martin","response_format":"wav"}')
-H "Content-Type: application/json" -d '{"input":"Wieder da","voice":"claribel","response_format":"wav"}')
[ "$CODE" = "200" ] && [ -s /tmp/tts33.wav ] \
&& ok "TTS nach Neustart wieder verfügbar" || bad "TTS-Recovery (Code $CODE)"
@@ -593,8 +593,8 @@ import json,sys
d=json.load(sys.stdin)
ids={m["id"] for m in d["data"]}
assert "whisper-1" in ids, ids
assert "kokoro-german" in ids, ids
' && ok "Audio-Modelle: whisper-1 + kokoro-german" || bad "Audio-Modelle"
assert "xtts-v2" in ids, ids
' && ok "Audio-Modelle: whisper-1 + xtts-v2" || bad "Audio-Modelle"
# --- 41. /v1/audio/voices ------------------------------------------------------------------------------------------
echo "== Test 41: GET /v1/audio/voices"
@@ -604,9 +604,8 @@ echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
ids={v["id"] for v in d["data"]}
assert "martin" in ids, ids
assert "victoria" in ids, ids
' && ok "Audio-Voices: martin + victoria" || bad "Audio-Voices"
assert "claribel" in ids, ids
' && ok "Audio-Voices: claribel" || bad "Audio-Voices"
# --- 42. STT + Qwen parallel ----------------------------------------------------------------------------------------
echo "== Test 42: STT + Qwen parallel"
@@ -637,7 +636,7 @@ sleep 0.2
# TTS-Request
CODE=$(curl -s -o /tmp/tts43.mp3 -w "%{http_code}" \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"input":"Hallo","voice":"martin"}')
-d '{"input":"Hallo","voice":"claribel"}')
wait $STT_PID43
[ "$CODE" = "200" ] && [ -s /tmp/tts43.mp3 ] \
&& ok "STT + TTS parallel (beide 200)" || bad "STT + TTS parallel (TTS Code $CODE)"