Add XTTS primary voice with Piper fallback

This commit is contained in:
Mikei386
2026-08-23 12:22:59 +02:00
parent 16a6288a4d
commit f90fc93f9c
18 changed files with 715 additions and 28 deletions
+3
View File
@@ -7,6 +7,9 @@ WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER
OPENWEBUI_IMAGE=ghcr.io/open-webui/open-webui:v0.9.5 OPENWEBUI_IMAGE=ghcr.io/open-webui/open-webui:v0.9.5
PIPER_TTS_VERSION=1.6.0 PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
AI_DNS=192.168.1.1 AI_DNS=192.168.1.1
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
+5 -2
View File
@@ -52,10 +52,13 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt.
| llama.cpp | nur Docker-intern | Inferenz und integrierte Vision | | llama.cpp | nur Docker-intern | Inferenz und integrierte Vision |
| Profile Controller | nur Docker-intern | eng begrenzter Profil-/FLUX-Hot-Swap | | Profile Controller | nur Docker-intern | eng begrenzter Profil-/FLUX-Hot-Swap |
| FLUX Worker | nur Docker-intern, normalerweise gestoppt | Bildgenerierung auf RTX 5080 | | FLUX Worker | nur Docker-intern, normalerweise gestoppt | Bildgenerierung auf RTX 5080 |
| Piper | nur Docker-intern | lokale deutsche Sprachausgabe | | XTTS-v2 | nur Docker-intern, RTX 3060 | primäre mehrsprachige Sprachausgabe |
| TTS Gateway | nur Docker-intern | Annmarie Nele, Queue und Piper-Fallback |
| Piper | nur Docker-intern, CPU | ausfallsichere deutsche Ersatzstimme |
| MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid | | MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid |
Piper-TTS und der FLUX.2-Klein-Hot-Swap sind reproduzierbare Kerndienste; STT XTTS-v2, TTS-Gateway, Piper-Fallback und der FLUX.2-Klein-Hot-Swap sind
reproduzierbare Kerndienste; STT
bleibt optional. Web-, Home-Assistant-, bleibt optional. Web-, Home-Assistant-,
ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter
`platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne `platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne
+75 -1
View File
@@ -529,7 +529,11 @@ services:
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true" ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true" ENABLE_TTS: "true"
TTS_WORKER_URL: http://piper:8085 # Stable OpenAI compatibility names remain piper/alloy because an
# existing Open WebUI database persists those values. The gateway maps
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
# Piper if XTTS is unavailable, busy or returns an error.
TTS_WORKER_URL: http://tts-gateway:8085
TTS_MODEL: piper TTS_MODEL: piper
TTS_VOICES: alloy TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy TTS_DEFAULT_VOICE: alloy
@@ -555,6 +559,8 @@ services:
condition: service_healthy condition: service_healthy
piper: piper:
condition: service_healthy condition: service_healthy
tts-gateway:
condition: service_healthy
flux-worker: flux-worker:
build: build:
@@ -622,6 +628,74 @@ services:
retries: 30 retries: 30
start_period: 120s start_period: 120s
xtts:
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
container_name: mike-ai-xtts
restart: unless-stopped
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids:
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
capabilities: [gpu]
read_only: true
shm_size: 1g
tmpfs:
- /tmp:size=1g,mode=1777
- /root/.cache:size=2g,mode=0700
volumes:
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
environment:
COQUI_TOS_AGREED: "1"
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
CUDA_VISIBLE_DEVICES: "0"
NUM_THREADS: "4"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
interval: 10s
timeout: 5s
retries: 36
start_period: 240s
tts-gateway:
build:
context: platform/docker/tts-gateway
image: mike-ai/tts-gateway:local
container_name: mike-ai-tts-gateway
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
environment:
TTS_GATEWAY_HOST: 0.0.0.0
TTS_GATEWAY_PORT: "8085"
XTTS_URL: http://xtts:80
PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy
XTTS_SPEAKER: Annmarie Nele
TTS_DEFAULT_LANGUAGE: de
XTTS_QUEUE_TIMEOUT: "15"
XTTS_TIMEOUT: "120"
PIPER_TIMEOUT: "120"
networks: [frontend]
depends_on:
piper:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 12
start_period: 10s
open-webui: open-webui:
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5} image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
container_name: mike-ai-open-webui container_name: mike-ai-open-webui
+4
View File
@@ -91,3 +91,7 @@ OPENWEBUI_ENABLE_SIGNUP=false
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
PIPER_TTS_VERSION=1.6.0 PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
+16 -6
View File
@@ -24,7 +24,9 @@ Heimnetz / VPN-Clients
+-- llama-uncensored (80K, Abliterated, dual GPU) +-- llama-uncensored (80K, Abliterated, dual GPU)
+-- llama-experimental +-- llama-experimental
+-- llama-ultra (256K, text-only, dual GPU) +-- llama-ultra (256K, text-only, dual GPU)
+-- Piper-TTS (CPU, nur intern) +-- TTS-Gateway
| +-- XTTS-v2 / Annmarie Nele (RTX 3060, primär)
| +-- Piper-TTS (CPU, automatischer Fallback)
+-- internes MCP-Netz +-- internes MCP-Netz
+-- Web-MCP + TinySearch + SearXNG +-- Web-MCP + TinySearch + SearXNG
+-- Home-Assistant-MCP-Relay +-- Home-Assistant-MCP-Relay
@@ -41,7 +43,9 @@ Heimnetz / VPN-Clients
| Profile Router | nur Docker-intern | OpenAI-API und Profilwahl | | Profile Router | nur Docker-intern | OpenAI-API und Profilwahl |
| Profile Controller | nein | startet ausschließlich fest erlaubte Profile | | Profile Controller | nein | startet ausschließlich fest erlaubte Profile |
| llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision | | llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision |
| Piper | nein | lokale deutsche Text-to-Speech-Ausgabe | | XTTS-v2 | nein | primäre deutsche/englische Text-to-Speech-Ausgabe auf RTX 3060 |
| TTS-Gateway | nein | serialisiert XTTS, segmentiert Sprachwechsel und fällt auf Piper zurück |
| Piper | nein | CPU-basierte Text-to-Speech-Rückfallebene |
| MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche | | MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche |
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP | | SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
@@ -99,10 +103,16 @@ Ultra bleibt für maximalen Kontext bewusst text-only.
## Optionale Erweiterungen ## Optionale Erweiterungen
Piper läuft als eigener CPU-Container und wird von Open WebUI über den Router Open WebUI spricht ausschließlich den Router an. Dieser reicht TTS intern an
angesprochen. Sein Port wird nicht veröffentlicht. Das Stimmenmodell liegt im das TTS-Gateway weiter. Das Gateway nutzt primär XTTS-v2 mit der Stimme
persistenten Volume `piper-data` und wird beim ersten Start reproduzierbar `Annmarie Nele` auf der RTX 3060. Deutsche Texte werden an bekannten
nachgeladen. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert. englischen IT-Begriffen segmentiert; reine englische Texte laufen vollständig
mit `language=en`. Da der offizielle XTTS-Streamingserver nur einen Auftrag
gleichzeitig unterstützt, serialisiert das Gateway die Aufträge. Bei Fehler,
Timeout oder belegter Queue übernimmt automatisch Piper auf der CPU. Kein
TTS-Port wird veröffentlicht. Der äußere Kompatibilitätsname bleibt bewusst
`piper/alloy`, damit persistente Open-WebUI-Einstellungen nach Updates und
Restores gültig bleiben. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert.
Home Assistant, ARR und Unraid sind vorbereitete Home Assistant, ARR und Unraid sind vorbereitete
MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only
Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
+3 -1
View File
@@ -12,7 +12,9 @@
| ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional | | ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional |
| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional | | Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional |
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional | | Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | eigener interner CPU-Container, Stimme `de_DE-thorsten-high` | Kern | | XTTS-v2 | Coqui, offizielles CUDA-12.1-Image per Digest | RTX-3060-Container, Stimme `Annmarie Nele`, CPML | Kern |
| TTS-Gateway | `platform/docker/tts-gateway/` | interne Queue, Deutsch/Englisch-Segmentierung und Piper-Fallback | Kern |
| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | interner CPU-Fallback, Stimme `de_DE-thorsten-high` | Kern |
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional | | FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional | | LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional | | Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
+13 -8
View File
@@ -126,15 +126,20 @@ leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus.
### XTTS ### XTTS
Dieser Abschnitt beschreibt ausschließlich den alten Referenzhost. Im neuen Der aktuelle Docker-Stack nutzt Coqui XTTS-v2 als primäre Sprachausgabe.
Docker-Zielsystem ersetzt Piper (`de_DE-thorsten-high`) diesen Dienst. Der isolierte Eignungs- und Ausfalltest ist in
[`XTTS_EVALUATION_2026-08-23.md`](XTTS_EVALUATION_2026-08-23.md) dokumentiert.
- Modell: Coqui XTTS-v2 - Modell: Coqui XTTS-v2, offizielles CUDA-12.1-Image per Digest gepinnt
- CPU-only - GPU: ausschließlich RTX 3060 über ihre stabile GPU-UUID
- Stimme: `claribel` - Stimme: `Annmarie Nele`
- Deutsch und Englisch - Deutsch und Englisch; bekannte englische IT-Begriffe werden segmentiert
- Port 8085, auf dem alten Host noch im LAN gebunden - kein veröffentlichter Port, nur Docker-intern erreichbar
- eigenes Python-3.11-Venv - serielles TTS-Gateway vor XTTS, weil der Server nur einen Auftrag zugleich
zuverlässig verarbeitet
- Piper mit `de_DE-thorsten-high` bleibt als automatischer CPU-Fallback aktiv
- OpenWebUI behält aus Kompatibilitätsgründen `model=piper` und `voice=alloy`;
der Router leitet diese Werte an das Gateway weiter
## Websuche ## Websuche
+5 -2
View File
@@ -76,8 +76,11 @@ laufen, sondern alle fachlichen Funktionen geprüft wurden.
- [ ] FLUX erzeugt Standard- und High-Bild - [ ] FLUX erzeugt Standard- und High-Bild
- [ ] Qwen-Profil wird nach FLUX wiederhergestellt - [ ] Qwen-Profil wird nach FLUX wiederhergestellt
- [ ] Whisper transkribiert deutsche und englische Testdatei - [ ] Whisper transkribiert deutsche und englische Testdatei
- [ ] Piper ist gesund und erzeugt über den Router deutsche WAV- und MP3-Ausgabe - [ ] XTTS-v2 läuft ausschließlich auf der RTX 3060 und meldet `Annmarie Nele`
- [ ] Stimme und `piper-tts`-Version entsprechen der Installationskonfiguration - [ ] TTS-Gateway erzeugt über den Router deutsche und englische WAV-/MP3-Ausgabe
- [ ] englische IT-Begriffe im deutschen Satz werden sprachlich segmentiert
- [ ] gestopptes XTTS fällt ohne Router-/OpenWebUI-Neustart auf Piper zurück
- [ ] Piper-Fallback und `piper-tts`-Version entsprechen der Installationskonfiguration
- [ ] STT/TTS blockieren das Textmodell nicht unzulässig - [ ] STT/TTS blockieren das Textmodell nicht unzulässig
## Phase F – Sicherheitsprüfung ## Phase F – Sicherheitsprüfung
+12 -2
View File
@@ -84,7 +84,9 @@ OpenSSH-Dienst des Hosts.
Die TTS-Verbindung wird für eine frische Open-WebUI-Datenbank automatisch als Die TTS-Verbindung wird für eine frische Open-WebUI-Datenbank automatisch als
OpenAI-kompatibler Audio-Endpunkt des Routers vorbelegt. Der Router reicht sie OpenAI-kompatibler Audio-Endpunkt des Routers vorbelegt. Der Router reicht sie
intern an Piper weiter; Port 8085 wird nicht am Host veröffentlicht. Ein intern an das TTS-Gateway weiter. Primär spricht XTTS-v2 mit `Annmarie Nele`
auf der RTX 3060; bei Fehlern oder Queue-Timeout übernimmt Piper auf der CPU.
Der Port 8085 wird nicht am Host veröffentlicht. Ein
Restore setzt zusätzlich die vier persistenten Audiofelder gezielt neu, damit Restore setzt zusätzlich die vier persistenten Audiofelder gezielt neu, damit
alte Werte wie `tts-1` oder `coral` die Compose-Vorgaben nicht überstimmen. Ein alte Werte wie `tts-1` oder `coral` die Compose-Vorgaben nicht überstimmen. Ein
Ende-zu-Ende-Test ohne Ausgabe des API-Schlüssels: Ende-zu-Ende-Test ohne Ausgabe des API-Schlüssels:
@@ -95,9 +97,17 @@ curl -fsS http://127.0.0.1:8081/v1/audio/speech \
-H "Authorization: Bearer $ROUTER_API_KEY" \ -H "Authorization: Bearer $ROUTER_API_KEY" \
-H 'Content-Type: application/json' \ -H 'Content-Type: application/json' \
-d '{"model":"piper","voice":"alloy","input":"Hallo von Athena.","response_format":"mp3"}' \ -d '{"model":"piper","voice":"alloy","input":"Hallo von Athena.","response_format":"mp3"}' \
-o /tmp/athena-piper-test.mp3 -o /tmp/athena-tts-test.mp3
``` ```
Der beibehaltene API-Name `piper/alloy` ist eine Kompatibilitätsschnittstelle;
bei gesundem XTTS stammt die Ausgabe von `Annmarie Nele`. Der interne Status
des TTS-Gateways nennt `last_backend`, `primary_ready`, `fallback_ready` und
die Zahl der Piper-Rückfälle. Ein Fallback-Test stoppt ausschließlich XTTS,
erzeugt einen synthetischen Satz über denselben Router-Endpunkt und startet
XTTS anschließend wieder. OpenWebUI und Router müssen dafür nicht geändert
oder neu gestartet werden.
Zusätzlich prüfen: Standort-LAN sieht keine KI-Ports; Heimnetz erreicht beide; Zusätzlich prüfen: Standort-LAN sieht keine KI-Ports; Heimnetz erreicht beide;
gestopptes VPN-Gateway lässt KI-Container nicht ins Internet; jeder Profilwechsel gestopptes VPN-Gateway lässt KI-Container nicht ins Internet; jeder Profilwechsel
startet exakt einen llama-Container; Text, Tool Call, Bild und Sprachausgabe funktionieren. startet exakt einen llama-Container; Text, Tool Call, Bild und Sprachausgabe funktionieren.
+4 -1
View File
@@ -35,7 +35,10 @@ Pflichtrollen:
- BF16 Vision-Projektor - BF16 Vision-Projektor
- Whisper large-v3-turbo - Whisper large-v3-turbo
- FLUX.2 klein - FLUX.2 klein
- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high` - XTTS-v2, per Digest gepinntes CUDA-12.1-Image und CPML-Akzeptanz
- XTTS-Stimme `Annmarie Nele`, RTX-3060-UUID und persistenter Modellcache
- internes TTS-Gateway mit Queue, Sprachsegmentierung und Piper-Fallback
- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high` als CPU-Fallback
## 2. Externe Komponenten und Commits – teilweise gesichert ## 2. Externe Komponenten und Commits – teilweise gesichert
+113
View File
@@ -0,0 +1,113 @@
# XTTS-v2 GPU evaluation on Athena (2026-08-23)
## Purpose and safety boundary
This was an isolated, reversible evaluation of Coqui XTTS-v2 as a possible
replacement for Piper. The user accepted the Coqui Public Model License for
this private test.
- Official image: `ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121`
- Pulled digest: `sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90`
- Test container: `mike-ai-xtts-test`
- GPU visibility: RTX 3060 only
- Host binding: `127.0.0.1:18105` only
- Restart policy: `no`
- Model cache: `/data/xtts-test/cache`
- Piper, Open WebUI and the router were not reconfigured.
The official server describes itself as a demo server. In particular, it does
not support concurrent streaming requests and is not an OpenAI-compatible
production endpoint. A queueing/OpenAI compatibility proxy is therefore
required before integration with Open WebUI.
## XTTS resource use
With the Medium profile already running, XTTS increased RTX 3060 use from
about 4,471 MiB to about 6,419 MiB. XTTS therefore occupied approximately
1,948 MiB and left about 5,492 MiB free. It did not use the RTX 5080.
With Ultra (256K) and XTTS loaded together:
| GPU | Used | Free |
|---|---:|---:|
| RTX 3060 12 GB | 8,669 MiB | 3,242 MiB |
| RTX 5080 16 GB | 15,770 MiB | 89 MiB |
The combination loaded successfully without OOM. This confirms that XTTS fits
even beside the largest standard text profile. The RTX 5080 must remain
unavailable to XTTS because Ultra already fills it almost completely.
## Streaming measurements
The initial measurements used built-in female speaker `Ana Florence`. A
subsequent five-voice German comparison selected **`Annmarie Nele`** as the
production voice. Tests used harmless synthetic text.
| Test | First audio | Generation time | Produced audio | RTF |
|---|---:|---:|---:|---:|
| German | 0.701 s | 2.889 s | 6.965 s | 0.415 |
| English | 0.305 s | 1.330 s | 3.989 s | 0.333 |
| German sentence with English IT terms | 0.309 s | 2.496 s | 7.339 s | 0.340 |
After warm-up, audio starts after roughly 0.3 seconds and synthesis is around
2.4 to 3 times faster than real time. Perceived Open WebUI latency also
includes Qwen's time to finish the first sentence and proxy buffering.
## Effect on Qwen throughput
| Profile | XTTS state | Generation speed |
|---|---|---:|
| Medium 160K | loaded but idle | 71.92 token/s |
| Medium 160K | actively speaking | 59.90 token/s |
| Ultra 256K | loaded but idle | 66.89 token/s |
| Ultra 256K | actively speaking | 55.27 token/s |
Active synthesis costs roughly 17% of Qwen generation speed because Qwen also
uses the RTX 3060. The slowdown ends with the speech request. Merely keeping
XTTS resident did not cause instability.
## Result and recommendation
XTTS-v2 is technically viable on the RTX 3060 and fits alongside every current
profile, including Ultra 256K. It provides early streaming and substantially
more natural multilingual speech than the current German-only Piper voice.
The production design keeps Piper and adds a small internal proxy that provides:
1. OpenAI-compatible `/v1/audio/speech` input and output.
2. A one-request queue because the official XTTS server has no concurrency.
3. German/English text segmentation so English product names are synthesized
with `language=en` while surrounding German remains `language=de`.
4. Cached speaker conditioning and a fixed allowlist of voices.
5. Health checks, bounded timeouts and automatic fallback to Piper.
This gateway now lives under `platform/docker/tts-gateway/`. The externally
visible compatibility values remain `model=piper` and `voice=alloy`; internally
that alias selects `Annmarie Nele` whenever XTTS is healthy.
## Production result and rollback
After the isolated evaluation, the compatibility gateway was tested in three
stages and then deployed to production:
1. Healthy XTTS produced valid WAV through the router-compatible endpoint.
2. XTTS was deliberately stopped; the same endpoint returned valid Piper WAV.
3. XTTS was restarted and automatically became the active backend again.
The production services are `mike-ai-xtts` and `mike-ai-tts-gateway`, both
Docker-internal. Piper remained healthy throughout. OpenWebUI required no
configuration or database change. The router's public compatibility values
remain `model=piper` and `voice=alloy`.
The initial Compose GPU declaration exposed both NVIDIA cards and caused XTTS
to select the nearly full RTX 5080. This was caught before the router switch.
The final declaration uses a Docker device reservation with the stable RTX
3060 UUID; inspecting the container must show exactly that UUID in
`DeviceRequests`.
The verified pre-deployment state is backed up below
`/data/backups/mike-ai/20260823-xtts-production`. The reusable rollback helper
is `platform/scripts/rollback-tts-production.sh`; it restores the saved Compose
and environment files, recreates the old Piper-connected router and removes
only XTTS and its gateway. Both VPN and university-network SSH paths were
verified after deployment.
+6 -2
View File
@@ -283,7 +283,8 @@ setup_wireguard() {
install_stack_files() { install_stack_files() {
log "Stackdateien installieren" log "Stackdateien installieren"
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$STATE_DIR/backups" XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache}
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups"
rsync -a --delete --exclude .git --exclude '*.local.*' \ rsync -a --delete --exclude .git --exclude '*.local.*' \
--exclude config/install.env "$ROOT_DIR/" "$STACK_DIR/" --exclude config/install.env "$ROOT_DIR/" "$STACK_DIR/"
install -d -m 0700 "$SECRETS_DIR" install -d -m 0700 "$SECRETS_DIR"
@@ -313,6 +314,9 @@ OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false}
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0} PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high} PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
XTTS_CACHE_DIR=$XTTS_CACHE_DIR
XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
AI_DNS=${WG_DNS:-1.1.1.1} AI_DNS=${WG_DNS:-1.1.1.1}
FAST_MODEL_FILE=$FAST_MODEL_FILE FAST_MODEL_FILE=$FAST_MODEL_FILE
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
@@ -475,7 +479,7 @@ build_and_start() {
llama-fast llama-medium llama-large llama-ultra llama-experimental llama-fast llama-medium llama-large llama-ultra llama-experimental
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
profile-controller router open-webui xtts piper tts-gateway profile-controller router open-webui
if [[ ${WIREGUARD_MODE:-container} == container ]]; then if [[ ${WIREGUARD_MODE:-container} == container ]]; then
systemctl restart mike-ai-container-vpn-guard.service systemctl restart mike-ai-container-vpn-guard.service
+2 -1
View File
@@ -47,7 +47,8 @@ container_healthy() {
[[ $state == running && ( -z $health || $health == healthy ) ]] [[ $state == running && ( -z $health || $health == healthy ) ]]
} }
for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai-open-webui; do for container in mike-ai-profile-controller mike-ai-router mike-ai-xtts \
mike-ai-piper mike-ai-tts-gateway mike-ai-open-webui; do
if container_healthy "$container"; then if container_healthy "$container"; then
pass "$container gesund" pass "$container gesund"
else else
+16
View File
@@ -0,0 +1,16 @@
FROM python:3.12-slim
RUN apt-get update \
&& apt-get install --no-install-recommends -y curl ffmpeg \
&& rm -rf /var/lib/apt/lists/* \
&& useradd --system --uid 10005 --home-dir /nonexistent --shell /usr/sbin/nologin tts
COPY tts_gateway.py /app/tts_gateway.py
USER 10005:10005
EXPOSE 8085
HEALTHCHECK --interval=10s --timeout=5s --retries=12 \
CMD curl -fsS http://127.0.0.1:8085/status || exit 1
ENTRYPOINT ["python", "/app/tts_gateway.py"]
@@ -0,0 +1,59 @@
import importlib.util
import pathlib
import unittest
MODULE_PATH = pathlib.Path(__file__).with_name("tts_gateway.py")
SPEC = importlib.util.spec_from_file_location("tts_gateway", MODULE_PATH)
gateway = importlib.util.module_from_spec(SPEC)
SPEC.loader.exec_module(gateway)
class LanguageSegmentationTests(unittest.TestCase):
def test_german_only(self):
self.assertEqual(
gateway.segment_languages("Guten Abend, wie warm ist es heute?"),
[("de", "Guten Abend, wie warm ist es heute?")],
)
def test_english_only(self):
text = "This is a short test and it is running on the local server."
self.assertEqual(gateway.segment_languages(text), [("en", text)])
def test_mixed_compounds(self):
text = "Ich öffne das Unraid-Dashboard und prüfe die Docker-Container."
self.assertEqual(
gateway.segment_languages(text),
[
("de", "Ich öffne das "),
("en", "Unraid-Dashboard"),
("de", " und prüfe die "),
("en", "Docker-Container"),
("de", "."),
],
)
class FallbackTests(unittest.TestCase):
def setUp(self):
self.original_xtts = gateway.synthesize_xtts
self.original_piper = gateway.synthesize_piper
def tearDown(self):
gateway.synthesize_xtts = self.original_xtts
gateway.synthesize_piper = self.original_piper
def test_piper_is_used_when_xtts_fails(self):
def fail(*_args):
raise RuntimeError("synthetic XTTS failure")
gateway.synthesize_xtts = fail
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
self.assertEqual(
gateway.synthesize("synthetic test", "wav", 1.0),
(b"piper", "audio/wav"),
)
if __name__ == "__main__":
unittest.main()
+356
View File
@@ -0,0 +1,356 @@
#!/usr/bin/env python3
"""Private XTTS-first TTS gateway with a Piper fallback.
The gateway implements the narrow /status and /tts protocol already consumed
by the profile router. Request text is never logged or persisted.
"""
from __future__ import annotations
import io
import json
import os
import re
import subprocess
import threading
import time
import urllib.error
import urllib.request
import wave
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
SILENCE_MS = int(os.getenv("XTTS_SEGMENT_SILENCE_MS", "20"))
SYNTHESIS_LOCK = threading.Lock()
STATE_LOCK = threading.Lock()
SPEAKER_LOCK = threading.Lock()
SPEAKER_CONDITIONING: dict | None = None
STATE = {
"last_backend": None,
"xtts_failures": 0,
"piper_fallbacks": 0,
"last_error": None,
}
# Prefer full compounds to isolated terms. This keeps switches infrequent and
# avoids making mixed-language speech sound like a sequence of separate clips.
ENGLISH_TERMS = (
"Home Assistant", "Open WebUI", "OpenWebUI", "Unraid Dashboard",
"Unraid-Dashboard", "Docker Container", "Docker-Container",
"Server Log", "Server-Log", "GitHub Repository", "GitHub Repo",
"WireGuard Tunnel", "Cron Job", "Cronjob", "Home Server",
"API Key", "Tool Calling", "Context Window", "Prompt Injection",
"Unraid", "Docker", "Container", "Dashboard", "Server", "Log",
"OpenAI", "GitHub", "WireGuard", "Linux", "Debian", "Frontend",
"Backend", "Router", "Browser", "Web", "Token", "Prompt", "Context",
"Model", "Image", "Tool", "Workflow", "Benchmark", "Streaming",
"SSH", "MCP", "API", "CPU", "GPU", "VRAM", "RAM", "HTTP", "HTTPS",
)
TERM_PATTERN = re.compile(
r"(?<![\w])(" + "|".join(
re.escape(term) for term in sorted(ENGLISH_TERMS, key=len, reverse=True)
) + r")(?![\w])",
re.IGNORECASE,
)
GERMAN_MARKERS = {
"aber", "auch", "auf", "das", "der", "die", "ein", "eine", "für",
"ich", "ist", "kann", "mit", "nicht", "noch", "oder", "soll", "und",
"wenn", "wir", "wird", "zu",
}
ENGLISH_MARKERS = {
"a", "and", "are", "can", "for", "from", "if", "in", "is", "it",
"of", "on", "or", "please", "the", "this", "to", "with", "you",
}
def _request(url: str, *, payload: dict | None = None,
timeout: float = 10) -> tuple[bytes, str]:
data = None
headers = {}
method = "GET"
if payload is not None:
data = json.dumps(payload, separators=(",", ":")).encode()
headers["Content-Type"] = "application/json"
method = "POST"
request = urllib.request.Request(
url, data=data, headers=headers, method=method)
with urllib.request.urlopen(request, timeout=timeout) as response:
body = response.read(MAX_AUDIO_BYTES + 1)
if len(body) > MAX_AUDIO_BYTES:
raise RuntimeError("upstream audio response is too large")
return body, response.headers.get_content_type()
def _json(url: str, timeout: float = 10) -> dict | list:
body, _ = _request(url, timeout=timeout)
return json.loads(body)
def _reachable(url: str, path: str, timeout: float = 2) -> bool:
try:
_request(f"{url}{path}", timeout=timeout)
return True
except (OSError, ValueError, RuntimeError, urllib.error.URLError):
return False
def _looks_english(text: str) -> bool:
words = re.findall(r"[A-Za-zÀ-ÿ]+", text.lower())
if not words:
return False
german = sum(word in GERMAN_MARKERS for word in words)
english = sum(word in ENGLISH_MARKERS for word in words)
return english >= 2 and english > german * 1.5 and not re.search(r"[äöüß]", text.lower())
def segment_languages(text: str) -> list[tuple[str, str]]:
"""Return a compact German/English segment sequence."""
if _looks_english(text):
return [("en", text)]
if DEFAULT_LANGUAGE != "de":
return [(DEFAULT_LANGUAGE, text)]
segments: list[tuple[str, str]] = []
cursor = 0
for match in TERM_PATTERN.finditer(text):
if match.start() > cursor:
segments.append(("de", text[cursor:match.start()]))
segments.append(("en", match.group(0)))
cursor = match.end()
if cursor < len(text):
segments.append(("de", text[cursor:]))
if not segments:
return [("de", text)]
merged: list[tuple[str, str]] = []
for language, part in segments:
if not part:
continue
if merged and merged[-1][0] == language:
previous_language, previous_text = merged[-1]
merged[-1] = (previous_language, previous_text + part)
else:
merged.append((language, part))
return merged
def _speaker_conditioning() -> dict:
global SPEAKER_CONDITIONING
with SPEAKER_LOCK:
if SPEAKER_CONDITIONING is not None:
return SPEAKER_CONDITIONING
speakers = _json(f"{XTTS_URL}/studio_speakers", XTTS_TIMEOUT)
if not isinstance(speakers, dict) or XTTS_SPEAKER not in speakers:
raise RuntimeError("configured XTTS speaker is unavailable")
selected = speakers[XTTS_SPEAKER]
SPEAKER_CONDITIONING = {
"speaker_embedding": selected["speaker_embedding"],
"gpt_cond_latent": selected["gpt_cond_latent"],
}
return SPEAKER_CONDITIONING
def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes:
payload = {
**conditioning,
"text": text,
"language": language,
"add_wav_header": True,
"stream_chunk_size": "20",
}
audio, _ = _request(f"{XTTS_URL}/tts_stream", payload=payload,
timeout=XTTS_TIMEOUT)
if len(audio) < 44 or audio[:4] != b"RIFF" or audio[8:12] != b"WAVE":
raise RuntimeError("XTTS returned invalid WAV data")
return audio[44:]
def _wav(pcm: bytes) -> bytes:
output = io.BytesIO()
with wave.open(output, "wb") as wav_file:
wav_file.setnchannels(1)
wav_file.setsampwidth(2)
wav_file.setframerate(24000)
wav_file.writeframes(pcm)
return output.getvalue()
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
if output_format == "wav" and speed == 1.0:
return wav_bytes, "audio/wav"
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \
if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"]
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
"-i", "pipe:0"]
if speed != 1.0:
command.extend(["-filter:a", f"atempo={speed:.4f}"])
command.extend([*codec, "pipe:1"])
result = subprocess.run(
command, input=wav_bytes, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
check=False, timeout=120)
if result.returncode != 0 or not result.stdout:
raise RuntimeError("audio conversion failed")
return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav"
def synthesize_xtts(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
conditioning = _speaker_conditioning()
pcm_parts: list[bytes] = []
silence = b"\x00\x00" * int(24000 * max(0, SILENCE_MS) / 1000)
for language, segment in segment_languages(text):
if not segment.strip():
continue
pcm_parts.append(_xtts_pcm(segment, language, conditioning))
if silence:
pcm_parts.append(silence)
if pcm_parts and silence:
pcm_parts.pop()
if not pcm_parts:
raise RuntimeError("no speech segments generated")
return _convert(_wav(b"".join(pcm_parts)), output_format, speed)
def synthesize_piper(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
return _request(
f"{PIPER_URL}/tts",
payload={"text": text, "voice": "alloy", "speed": speed,
"format": output_format},
timeout=PIPER_TIMEOUT,
)
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if acquired:
try:
audio = synthesize_xtts(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "xtts-v2"
STATE["last_error"] = None
return audio
except Exception as exc: # fallback must cover all XTTS failures
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["last_error"] = type(exc).__name__
finally:
SYNTHESIS_LOCK.release()
else:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["last_error"] = "queue-timeout"
audio = synthesize_piper(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "piper"
STATE["piper_fallbacks"] += 1
return audio
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def log_message(self, fmt: str, *args: object) -> None:
# Never log request URLs, bodies, synthesized text or speaker vectors.
print(f"tts-gateway: {self.command} -> {args[1] if len(args) > 1 else '-'}")
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
self.send_response(status)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(body)))
self.send_header("Cache-Control", "no-store")
self.end_headers()
self.wfile.write(body)
def send_json(self, status: int, payload: dict) -> None:
self.send_bytes(status, json.dumps(payload, separators=(",", ":")).encode(),
"application/json")
def do_GET(self) -> None: # noqa: N802
if self.path != "/status":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
primary_ready = _reachable(XTTS_URL, "/languages")
fallback_ready = _reachable(PIPER_URL, "/status")
with STATE_LOCK:
state = dict(STATE)
self.send_json(
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
{
"ready": fallback_ready,
"engine": "xtts-v2-with-piper-fallback",
"model": "xtts-v2",
"voices": [VOICE_ALIAS],
"speaker": XTTS_SPEAKER,
"primary_ready": primary_ready,
"fallback_ready": fallback_ready,
**state,
},
)
def do_POST(self) -> None: # noqa: N802
if self.path != "/tts":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
try:
length = int(self.headers.get("Content-Length", "0"))
except ValueError:
length = 0
if length <= 0 or length > MAX_REQUEST_BYTES:
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE,
{"error": "invalid request size"})
return
try:
request = json.loads(self.rfile.read(length))
text = request.get("text", "")
voice = request.get("voice", VOICE_ALIAS)
output_format = request.get("format", "mp3")
speed = float(request.get("speed", 1.0))
except (json.JSONDecodeError, TypeError, ValueError):
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
return
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
return
if voice != VOICE_ALIAS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
return
if output_format not in {"wav", "mp3"}:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
return
if not 0.5 <= speed <= 2.0:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
return
started = time.monotonic()
try:
audio, content_type = synthesize(text.strip(), output_format, speed)
except Exception as exc:
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "all local speech backends failed"})
return
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
f"{time.monotonic() - started:.2f}s")
self.send_bytes(HTTPStatus.OK, audio, content_type)
if __name__ == "__main__":
print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
@@ -172,10 +172,11 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen" log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
cd "$STACK_DIR" cd "$STACK_DIR"
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \ docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
router open-webui xtts piper tts-gateway router open-webui
deadline=$((SECONDS + 180)) deadline=$((SECONDS + 180))
for container in mike-ai-router mike-ai-open-webui; do for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \
mike-ai-router mike-ai-open-webui; do
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \ until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
"$container" 2>/dev/null || true) == healthy ]]; do "$container" 2>/dev/null || true) == healthy ]]; do
(( SECONDS < deadline )) || die "$container wurde nicht rechtzeitig gesund." (( SECONDS < deadline )) || die "$container wurde nicht rechtzeitig gesund."
+20
View File
@@ -0,0 +1,20 @@
#!/bin/sh
set -eu
backup_dir=${1:?usage: rollback-tts-production.sh BACKUP_DIR}
case "$backup_dir" in
/data/backups/mike-ai/*) ;;
*) echo "refusing unexpected backup path" >&2; exit 2 ;;
esac
test -f "$backup_dir/compose.yaml"
test -f "$backup_dir/stack.env"
cp -a "$backup_dir/compose.yaml" /opt/mike-ai/stack/compose.yaml
cp -a "$backup_dir/stack.env" /etc/mike-ai/stack.env
chmod 0600 /etc/mike-ai/stack.env
cd /opt/mike-ai/stack
docker compose --env-file /etc/mike-ai/stack.env up -d --no-deps --force-recreate router
docker rm -f mike-ai-tts-gateway mike-ai-xtts >/dev/null 2>&1 || true