Add XTTS primary voice with Piper fallback
This commit is contained in:
@@ -7,6 +7,9 @@ WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER
|
|||||||
OPENWEBUI_IMAGE=ghcr.io/open-webui/open-webui:v0.9.5
|
OPENWEBUI_IMAGE=ghcr.io/open-webui/open-webui:v0.9.5
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
|
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||||
|
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
||||||
|
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
AI_DNS=192.168.1.1
|
AI_DNS=192.168.1.1
|
||||||
|
|
||||||
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
|
|||||||
@@ -52,10 +52,13 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt.
|
|||||||
| llama.cpp | nur Docker-intern | Inferenz und integrierte Vision |
|
| llama.cpp | nur Docker-intern | Inferenz und integrierte Vision |
|
||||||
| Profile Controller | nur Docker-intern | eng begrenzter Profil-/FLUX-Hot-Swap |
|
| Profile Controller | nur Docker-intern | eng begrenzter Profil-/FLUX-Hot-Swap |
|
||||||
| FLUX Worker | nur Docker-intern, normalerweise gestoppt | Bildgenerierung auf RTX 5080 |
|
| FLUX Worker | nur Docker-intern, normalerweise gestoppt | Bildgenerierung auf RTX 5080 |
|
||||||
| Piper | nur Docker-intern | lokale deutsche Sprachausgabe |
|
| XTTS-v2 | nur Docker-intern, RTX 3060 | primäre mehrsprachige Sprachausgabe |
|
||||||
|
| TTS Gateway | nur Docker-intern | Annmarie Nele, Queue und Piper-Fallback |
|
||||||
|
| Piper | nur Docker-intern, CPU | ausfallsichere deutsche Ersatzstimme |
|
||||||
| MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid |
|
| MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid |
|
||||||
|
|
||||||
Piper-TTS und der FLUX.2-Klein-Hot-Swap sind reproduzierbare Kerndienste; STT
|
XTTS-v2, TTS-Gateway, Piper-Fallback und der FLUX.2-Klein-Hot-Swap sind
|
||||||
|
reproduzierbare Kerndienste; STT
|
||||||
bleibt optional. Web-, Home-Assistant-,
|
bleibt optional. Web-, Home-Assistant-,
|
||||||
ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter
|
ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter
|
||||||
`platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne
|
`platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne
|
||||||
|
|||||||
+75
-1
@@ -529,7 +529,11 @@ services:
|
|||||||
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
||||||
ENABLE_IMAGE_GENERATION: "true"
|
ENABLE_IMAGE_GENERATION: "true"
|
||||||
ENABLE_TTS: "true"
|
ENABLE_TTS: "true"
|
||||||
TTS_WORKER_URL: http://piper:8085
|
# Stable OpenAI compatibility names remain piper/alloy because an
|
||||||
|
# existing Open WebUI database persists those values. The gateway maps
|
||||||
|
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
|
||||||
|
# Piper if XTTS is unavailable, busy or returns an error.
|
||||||
|
TTS_WORKER_URL: http://tts-gateway:8085
|
||||||
TTS_MODEL: piper
|
TTS_MODEL: piper
|
||||||
TTS_VOICES: alloy
|
TTS_VOICES: alloy
|
||||||
TTS_DEFAULT_VOICE: alloy
|
TTS_DEFAULT_VOICE: alloy
|
||||||
@@ -555,6 +559,8 @@ services:
|
|||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
piper:
|
piper:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
tts-gateway:
|
||||||
|
condition: service_healthy
|
||||||
|
|
||||||
flux-worker:
|
flux-worker:
|
||||||
build:
|
build:
|
||||||
@@ -622,6 +628,74 @@ services:
|
|||||||
retries: 30
|
retries: 30
|
||||||
start_period: 120s
|
start_period: 120s
|
||||||
|
|
||||||
|
xtts:
|
||||||
|
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
||||||
|
container_name: mike-ai-xtts
|
||||||
|
restart: unless-stopped
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
device_ids:
|
||||||
|
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
|
capabilities: [gpu]
|
||||||
|
read_only: true
|
||||||
|
shm_size: 1g
|
||||||
|
tmpfs:
|
||||||
|
- /tmp:size=1g,mode=1777
|
||||||
|
- /root/.cache:size=2g,mode=0700
|
||||||
|
volumes:
|
||||||
|
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
|
||||||
|
environment:
|
||||||
|
COQUI_TOS_AGREED: "1"
|
||||||
|
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
CUDA_VISIBLE_DEVICES: "0"
|
||||||
|
NUM_THREADS: "4"
|
||||||
|
networks: [frontend]
|
||||||
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
cap_drop: [ALL]
|
||||||
|
healthcheck:
|
||||||
|
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 36
|
||||||
|
start_period: 240s
|
||||||
|
|
||||||
|
tts-gateway:
|
||||||
|
build:
|
||||||
|
context: platform/docker/tts-gateway
|
||||||
|
image: mike-ai/tts-gateway:local
|
||||||
|
container_name: mike-ai-tts-gateway
|
||||||
|
restart: unless-stopped
|
||||||
|
read_only: true
|
||||||
|
tmpfs:
|
||||||
|
- /tmp:size=256m,mode=1777
|
||||||
|
environment:
|
||||||
|
TTS_GATEWAY_HOST: 0.0.0.0
|
||||||
|
TTS_GATEWAY_PORT: "8085"
|
||||||
|
XTTS_URL: http://xtts:80
|
||||||
|
PIPER_URL: http://piper:8085
|
||||||
|
TTS_VOICE_ALIAS: alloy
|
||||||
|
XTTS_SPEAKER: Annmarie Nele
|
||||||
|
TTS_DEFAULT_LANGUAGE: de
|
||||||
|
XTTS_QUEUE_TIMEOUT: "15"
|
||||||
|
XTTS_TIMEOUT: "120"
|
||||||
|
PIPER_TIMEOUT: "120"
|
||||||
|
networks: [frontend]
|
||||||
|
depends_on:
|
||||||
|
piper:
|
||||||
|
condition: service_healthy
|
||||||
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
cap_drop: [ALL]
|
||||||
|
healthcheck:
|
||||||
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 12
|
||||||
|
start_period: 10s
|
||||||
|
|
||||||
open-webui:
|
open-webui:
|
||||||
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
|
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
|
||||||
container_name: mike-ai-open-webui
|
container_name: mike-ai-open-webui
|
||||||
|
|||||||
@@ -91,3 +91,7 @@ OPENWEBUI_ENABLE_SIGNUP=false
|
|||||||
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
|
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
|
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||||
|
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
||||||
|
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
|
||||||
|
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
|
|||||||
+16
-6
@@ -24,7 +24,9 @@ Heimnetz / VPN-Clients
|
|||||||
+-- llama-uncensored (80K, Abliterated, dual GPU)
|
+-- llama-uncensored (80K, Abliterated, dual GPU)
|
||||||
+-- llama-experimental
|
+-- llama-experimental
|
||||||
+-- llama-ultra (256K, text-only, dual GPU)
|
+-- llama-ultra (256K, text-only, dual GPU)
|
||||||
+-- Piper-TTS (CPU, nur intern)
|
+-- TTS-Gateway
|
||||||
|
| +-- XTTS-v2 / Annmarie Nele (RTX 3060, primär)
|
||||||
|
| +-- Piper-TTS (CPU, automatischer Fallback)
|
||||||
+-- internes MCP-Netz
|
+-- internes MCP-Netz
|
||||||
+-- Web-MCP + TinySearch + SearXNG
|
+-- Web-MCP + TinySearch + SearXNG
|
||||||
+-- Home-Assistant-MCP-Relay
|
+-- Home-Assistant-MCP-Relay
|
||||||
@@ -41,7 +43,9 @@ Heimnetz / VPN-Clients
|
|||||||
| Profile Router | nur Docker-intern | OpenAI-API und Profilwahl |
|
| Profile Router | nur Docker-intern | OpenAI-API und Profilwahl |
|
||||||
| Profile Controller | nein | startet ausschließlich fest erlaubte Profile |
|
| Profile Controller | nein | startet ausschließlich fest erlaubte Profile |
|
||||||
| llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision |
|
| llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision |
|
||||||
| Piper | nein | lokale deutsche Text-to-Speech-Ausgabe |
|
| XTTS-v2 | nein | primäre deutsche/englische Text-to-Speech-Ausgabe auf RTX 3060 |
|
||||||
|
| TTS-Gateway | nein | serialisiert XTTS, segmentiert Sprachwechsel und fällt auf Piper zurück |
|
||||||
|
| Piper | nein | CPU-basierte Text-to-Speech-Rückfallebene |
|
||||||
| MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche |
|
| MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche |
|
||||||
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
|
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
|
||||||
|
|
||||||
@@ -99,10 +103,16 @@ Ultra bleibt für maximalen Kontext bewusst text-only.
|
|||||||
|
|
||||||
## Optionale Erweiterungen
|
## Optionale Erweiterungen
|
||||||
|
|
||||||
Piper läuft als eigener CPU-Container und wird von Open WebUI über den Router
|
Open WebUI spricht ausschließlich den Router an. Dieser reicht TTS intern an
|
||||||
angesprochen. Sein Port wird nicht veröffentlicht. Das Stimmenmodell liegt im
|
das TTS-Gateway weiter. Das Gateway nutzt primär XTTS-v2 mit der Stimme
|
||||||
persistenten Volume `piper-data` und wird beim ersten Start reproduzierbar
|
`Annmarie Nele` auf der RTX 3060. Deutsche Texte werden an bekannten
|
||||||
nachgeladen. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert.
|
englischen IT-Begriffen segmentiert; reine englische Texte laufen vollständig
|
||||||
|
mit `language=en`. Da der offizielle XTTS-Streamingserver nur einen Auftrag
|
||||||
|
gleichzeitig unterstützt, serialisiert das Gateway die Aufträge. Bei Fehler,
|
||||||
|
Timeout oder belegter Queue übernimmt automatisch Piper auf der CPU. Kein
|
||||||
|
TTS-Port wird veröffentlicht. Der äußere Kompatibilitätsname bleibt bewusst
|
||||||
|
`piper/alloy`, damit persistente Open-WebUI-Einstellungen nach Updates und
|
||||||
|
Restores gültig bleiben. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert.
|
||||||
Home Assistant, ARR und Unraid sind vorbereitete
|
Home Assistant, ARR und Unraid sind vorbereitete
|
||||||
MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only
|
MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only
|
||||||
Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
|
Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
|
||||||
|
|||||||
+3
-1
@@ -12,7 +12,9 @@
|
|||||||
| ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional |
|
| ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional |
|
||||||
| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional |
|
| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional |
|
||||||
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
|
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
|
||||||
| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | eigener interner CPU-Container, Stimme `de_DE-thorsten-high` | Kern |
|
| XTTS-v2 | Coqui, offizielles CUDA-12.1-Image per Digest | RTX-3060-Container, Stimme `Annmarie Nele`, CPML | Kern |
|
||||||
|
| TTS-Gateway | `platform/docker/tts-gateway/` | interne Queue, Deutsch/Englisch-Segmentierung und Piper-Fallback | Kern |
|
||||||
|
| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | interner CPU-Fallback, Stimme `de_DE-thorsten-high` | Kern |
|
||||||
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
|
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
|
||||||
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
|
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
|
||||||
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
|
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
|
||||||
|
|||||||
@@ -126,15 +126,20 @@ leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus.
|
|||||||
|
|
||||||
### XTTS
|
### XTTS
|
||||||
|
|
||||||
Dieser Abschnitt beschreibt ausschließlich den alten Referenzhost. Im neuen
|
Der aktuelle Docker-Stack nutzt Coqui XTTS-v2 als primäre Sprachausgabe.
|
||||||
Docker-Zielsystem ersetzt Piper (`de_DE-thorsten-high`) diesen Dienst.
|
Der isolierte Eignungs- und Ausfalltest ist in
|
||||||
|
[`XTTS_EVALUATION_2026-08-23.md`](XTTS_EVALUATION_2026-08-23.md) dokumentiert.
|
||||||
|
|
||||||
- Modell: Coqui XTTS-v2
|
- Modell: Coqui XTTS-v2, offizielles CUDA-12.1-Image per Digest gepinnt
|
||||||
- CPU-only
|
- GPU: ausschließlich RTX 3060 über ihre stabile GPU-UUID
|
||||||
- Stimme: `claribel`
|
- Stimme: `Annmarie Nele`
|
||||||
- Deutsch und Englisch
|
- Deutsch und Englisch; bekannte englische IT-Begriffe werden segmentiert
|
||||||
- Port 8085, auf dem alten Host noch im LAN gebunden
|
- kein veröffentlichter Port, nur Docker-intern erreichbar
|
||||||
- eigenes Python-3.11-Venv
|
- serielles TTS-Gateway vor XTTS, weil der Server nur einen Auftrag zugleich
|
||||||
|
zuverlässig verarbeitet
|
||||||
|
- Piper mit `de_DE-thorsten-high` bleibt als automatischer CPU-Fallback aktiv
|
||||||
|
- OpenWebUI behält aus Kompatibilitätsgründen `model=piper` und `voice=alloy`;
|
||||||
|
der Router leitet diese Werte an das Gateway weiter
|
||||||
|
|
||||||
## Websuche
|
## Websuche
|
||||||
|
|
||||||
|
|||||||
@@ -76,8 +76,11 @@ laufen, sondern alle fachlichen Funktionen geprüft wurden.
|
|||||||
- [ ] FLUX erzeugt Standard- und High-Bild
|
- [ ] FLUX erzeugt Standard- und High-Bild
|
||||||
- [ ] Qwen-Profil wird nach FLUX wiederhergestellt
|
- [ ] Qwen-Profil wird nach FLUX wiederhergestellt
|
||||||
- [ ] Whisper transkribiert deutsche und englische Testdatei
|
- [ ] Whisper transkribiert deutsche und englische Testdatei
|
||||||
- [ ] Piper ist gesund und erzeugt über den Router deutsche WAV- und MP3-Ausgabe
|
- [ ] XTTS-v2 läuft ausschließlich auf der RTX 3060 und meldet `Annmarie Nele`
|
||||||
- [ ] Stimme und `piper-tts`-Version entsprechen der Installationskonfiguration
|
- [ ] TTS-Gateway erzeugt über den Router deutsche und englische WAV-/MP3-Ausgabe
|
||||||
|
- [ ] englische IT-Begriffe im deutschen Satz werden sprachlich segmentiert
|
||||||
|
- [ ] gestopptes XTTS fällt ohne Router-/OpenWebUI-Neustart auf Piper zurück
|
||||||
|
- [ ] Piper-Fallback und `piper-tts`-Version entsprechen der Installationskonfiguration
|
||||||
- [ ] STT/TTS blockieren das Textmodell nicht unzulässig
|
- [ ] STT/TTS blockieren das Textmodell nicht unzulässig
|
||||||
|
|
||||||
## Phase F – Sicherheitsprüfung
|
## Phase F – Sicherheitsprüfung
|
||||||
|
|||||||
+12
-2
@@ -84,7 +84,9 @@ OpenSSH-Dienst des Hosts.
|
|||||||
|
|
||||||
Die TTS-Verbindung wird für eine frische Open-WebUI-Datenbank automatisch als
|
Die TTS-Verbindung wird für eine frische Open-WebUI-Datenbank automatisch als
|
||||||
OpenAI-kompatibler Audio-Endpunkt des Routers vorbelegt. Der Router reicht sie
|
OpenAI-kompatibler Audio-Endpunkt des Routers vorbelegt. Der Router reicht sie
|
||||||
intern an Piper weiter; Port 8085 wird nicht am Host veröffentlicht. Ein
|
intern an das TTS-Gateway weiter. Primär spricht XTTS-v2 mit `Annmarie Nele`
|
||||||
|
auf der RTX 3060; bei Fehlern oder Queue-Timeout übernimmt Piper auf der CPU.
|
||||||
|
Der Port 8085 wird nicht am Host veröffentlicht. Ein
|
||||||
Restore setzt zusätzlich die vier persistenten Audiofelder gezielt neu, damit
|
Restore setzt zusätzlich die vier persistenten Audiofelder gezielt neu, damit
|
||||||
alte Werte wie `tts-1` oder `coral` die Compose-Vorgaben nicht überstimmen. Ein
|
alte Werte wie `tts-1` oder `coral` die Compose-Vorgaben nicht überstimmen. Ein
|
||||||
Ende-zu-Ende-Test ohne Ausgabe des API-Schlüssels:
|
Ende-zu-Ende-Test ohne Ausgabe des API-Schlüssels:
|
||||||
@@ -95,9 +97,17 @@ curl -fsS http://127.0.0.1:8081/v1/audio/speech \
|
|||||||
-H "Authorization: Bearer $ROUTER_API_KEY" \
|
-H "Authorization: Bearer $ROUTER_API_KEY" \
|
||||||
-H 'Content-Type: application/json' \
|
-H 'Content-Type: application/json' \
|
||||||
-d '{"model":"piper","voice":"alloy","input":"Hallo von Athena.","response_format":"mp3"}' \
|
-d '{"model":"piper","voice":"alloy","input":"Hallo von Athena.","response_format":"mp3"}' \
|
||||||
-o /tmp/athena-piper-test.mp3
|
-o /tmp/athena-tts-test.mp3
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Der beibehaltene API-Name `piper/alloy` ist eine Kompatibilitätsschnittstelle;
|
||||||
|
bei gesundem XTTS stammt die Ausgabe von `Annmarie Nele`. Der interne Status
|
||||||
|
des TTS-Gateways nennt `last_backend`, `primary_ready`, `fallback_ready` und
|
||||||
|
die Zahl der Piper-Rückfälle. Ein Fallback-Test stoppt ausschließlich XTTS,
|
||||||
|
erzeugt einen synthetischen Satz über denselben Router-Endpunkt und startet
|
||||||
|
XTTS anschließend wieder. OpenWebUI und Router müssen dafür nicht geändert
|
||||||
|
oder neu gestartet werden.
|
||||||
|
|
||||||
Zusätzlich prüfen: Standort-LAN sieht keine KI-Ports; Heimnetz erreicht beide;
|
Zusätzlich prüfen: Standort-LAN sieht keine KI-Ports; Heimnetz erreicht beide;
|
||||||
gestopptes VPN-Gateway lässt KI-Container nicht ins Internet; jeder Profilwechsel
|
gestopptes VPN-Gateway lässt KI-Container nicht ins Internet; jeder Profilwechsel
|
||||||
startet exakt einen llama-Container; Text, Tool Call, Bild und Sprachausgabe funktionieren.
|
startet exakt einen llama-Container; Text, Tool Call, Bild und Sprachausgabe funktionieren.
|
||||||
|
|||||||
@@ -35,7 +35,10 @@ Pflichtrollen:
|
|||||||
- BF16 Vision-Projektor
|
- BF16 Vision-Projektor
|
||||||
- Whisper large-v3-turbo
|
- Whisper large-v3-turbo
|
||||||
- FLUX.2 klein
|
- FLUX.2 klein
|
||||||
- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high`
|
- XTTS-v2, per Digest gepinntes CUDA-12.1-Image und CPML-Akzeptanz
|
||||||
|
- XTTS-Stimme `Annmarie Nele`, RTX-3060-UUID und persistenter Modellcache
|
||||||
|
- internes TTS-Gateway mit Queue, Sprachsegmentierung und Piper-Fallback
|
||||||
|
- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high` als CPU-Fallback
|
||||||
|
|
||||||
## 2. Externe Komponenten und Commits – teilweise gesichert
|
## 2. Externe Komponenten und Commits – teilweise gesichert
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,113 @@
|
|||||||
|
# XTTS-v2 GPU evaluation on Athena (2026-08-23)
|
||||||
|
|
||||||
|
## Purpose and safety boundary
|
||||||
|
|
||||||
|
This was an isolated, reversible evaluation of Coqui XTTS-v2 as a possible
|
||||||
|
replacement for Piper. The user accepted the Coqui Public Model License for
|
||||||
|
this private test.
|
||||||
|
|
||||||
|
- Official image: `ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121`
|
||||||
|
- Pulled digest: `sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90`
|
||||||
|
- Test container: `mike-ai-xtts-test`
|
||||||
|
- GPU visibility: RTX 3060 only
|
||||||
|
- Host binding: `127.0.0.1:18105` only
|
||||||
|
- Restart policy: `no`
|
||||||
|
- Model cache: `/data/xtts-test/cache`
|
||||||
|
- Piper, Open WebUI and the router were not reconfigured.
|
||||||
|
|
||||||
|
The official server describes itself as a demo server. In particular, it does
|
||||||
|
not support concurrent streaming requests and is not an OpenAI-compatible
|
||||||
|
production endpoint. A queueing/OpenAI compatibility proxy is therefore
|
||||||
|
required before integration with Open WebUI.
|
||||||
|
|
||||||
|
## XTTS resource use
|
||||||
|
|
||||||
|
With the Medium profile already running, XTTS increased RTX 3060 use from
|
||||||
|
about 4,471 MiB to about 6,419 MiB. XTTS therefore occupied approximately
|
||||||
|
1,948 MiB and left about 5,492 MiB free. It did not use the RTX 5080.
|
||||||
|
|
||||||
|
With Ultra (256K) and XTTS loaded together:
|
||||||
|
|
||||||
|
| GPU | Used | Free |
|
||||||
|
|---|---:|---:|
|
||||||
|
| RTX 3060 12 GB | 8,669 MiB | 3,242 MiB |
|
||||||
|
| RTX 5080 16 GB | 15,770 MiB | 89 MiB |
|
||||||
|
|
||||||
|
The combination loaded successfully without OOM. This confirms that XTTS fits
|
||||||
|
even beside the largest standard text profile. The RTX 5080 must remain
|
||||||
|
unavailable to XTTS because Ultra already fills it almost completely.
|
||||||
|
|
||||||
|
## Streaming measurements
|
||||||
|
|
||||||
|
The initial measurements used built-in female speaker `Ana Florence`. A
|
||||||
|
subsequent five-voice German comparison selected **`Annmarie Nele`** as the
|
||||||
|
production voice. Tests used harmless synthetic text.
|
||||||
|
|
||||||
|
| Test | First audio | Generation time | Produced audio | RTF |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| German | 0.701 s | 2.889 s | 6.965 s | 0.415 |
|
||||||
|
| English | 0.305 s | 1.330 s | 3.989 s | 0.333 |
|
||||||
|
| German sentence with English IT terms | 0.309 s | 2.496 s | 7.339 s | 0.340 |
|
||||||
|
|
||||||
|
After warm-up, audio starts after roughly 0.3 seconds and synthesis is around
|
||||||
|
2.4 to 3 times faster than real time. Perceived Open WebUI latency also
|
||||||
|
includes Qwen's time to finish the first sentence and proxy buffering.
|
||||||
|
|
||||||
|
## Effect on Qwen throughput
|
||||||
|
|
||||||
|
| Profile | XTTS state | Generation speed |
|
||||||
|
|---|---|---:|
|
||||||
|
| Medium 160K | loaded but idle | 71.92 token/s |
|
||||||
|
| Medium 160K | actively speaking | 59.90 token/s |
|
||||||
|
| Ultra 256K | loaded but idle | 66.89 token/s |
|
||||||
|
| Ultra 256K | actively speaking | 55.27 token/s |
|
||||||
|
|
||||||
|
Active synthesis costs roughly 17% of Qwen generation speed because Qwen also
|
||||||
|
uses the RTX 3060. The slowdown ends with the speech request. Merely keeping
|
||||||
|
XTTS resident did not cause instability.
|
||||||
|
|
||||||
|
## Result and recommendation
|
||||||
|
|
||||||
|
XTTS-v2 is technically viable on the RTX 3060 and fits alongside every current
|
||||||
|
profile, including Ultra 256K. It provides early streaming and substantially
|
||||||
|
more natural multilingual speech than the current German-only Piper voice.
|
||||||
|
|
||||||
|
The production design keeps Piper and adds a small internal proxy that provides:
|
||||||
|
|
||||||
|
1. OpenAI-compatible `/v1/audio/speech` input and output.
|
||||||
|
2. A one-request queue because the official XTTS server has no concurrency.
|
||||||
|
3. German/English text segmentation so English product names are synthesized
|
||||||
|
with `language=en` while surrounding German remains `language=de`.
|
||||||
|
4. Cached speaker conditioning and a fixed allowlist of voices.
|
||||||
|
5. Health checks, bounded timeouts and automatic fallback to Piper.
|
||||||
|
|
||||||
|
This gateway now lives under `platform/docker/tts-gateway/`. The externally
|
||||||
|
visible compatibility values remain `model=piper` and `voice=alloy`; internally
|
||||||
|
that alias selects `Annmarie Nele` whenever XTTS is healthy.
|
||||||
|
|
||||||
|
## Production result and rollback
|
||||||
|
|
||||||
|
After the isolated evaluation, the compatibility gateway was tested in three
|
||||||
|
stages and then deployed to production:
|
||||||
|
|
||||||
|
1. Healthy XTTS produced valid WAV through the router-compatible endpoint.
|
||||||
|
2. XTTS was deliberately stopped; the same endpoint returned valid Piper WAV.
|
||||||
|
3. XTTS was restarted and automatically became the active backend again.
|
||||||
|
|
||||||
|
The production services are `mike-ai-xtts` and `mike-ai-tts-gateway`, both
|
||||||
|
Docker-internal. Piper remained healthy throughout. OpenWebUI required no
|
||||||
|
configuration or database change. The router's public compatibility values
|
||||||
|
remain `model=piper` and `voice=alloy`.
|
||||||
|
|
||||||
|
The initial Compose GPU declaration exposed both NVIDIA cards and caused XTTS
|
||||||
|
to select the nearly full RTX 5080. This was caught before the router switch.
|
||||||
|
The final declaration uses a Docker device reservation with the stable RTX
|
||||||
|
3060 UUID; inspecting the container must show exactly that UUID in
|
||||||
|
`DeviceRequests`.
|
||||||
|
|
||||||
|
The verified pre-deployment state is backed up below
|
||||||
|
`/data/backups/mike-ai/20260823-xtts-production`. The reusable rollback helper
|
||||||
|
is `platform/scripts/rollback-tts-production.sh`; it restores the saved Compose
|
||||||
|
and environment files, recreates the old Piper-connected router and removes
|
||||||
|
only XTTS and its gateway. Both VPN and university-network SSH paths were
|
||||||
|
verified after deployment.
|
||||||
+6
-2
@@ -283,7 +283,8 @@ setup_wireguard() {
|
|||||||
|
|
||||||
install_stack_files() {
|
install_stack_files() {
|
||||||
log "Stackdateien installieren"
|
log "Stackdateien installieren"
|
||||||
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$STATE_DIR/backups"
|
XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache}
|
||||||
|
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups"
|
||||||
rsync -a --delete --exclude .git --exclude '*.local.*' \
|
rsync -a --delete --exclude .git --exclude '*.local.*' \
|
||||||
--exclude config/install.env "$ROOT_DIR/" "$STACK_DIR/"
|
--exclude config/install.env "$ROOT_DIR/" "$STACK_DIR/"
|
||||||
install -d -m 0700 "$SECRETS_DIR"
|
install -d -m 0700 "$SECRETS_DIR"
|
||||||
@@ -313,6 +314,9 @@ OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false}
|
|||||||
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
||||||
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
||||||
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||||
|
XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
||||||
|
XTTS_CACHE_DIR=$XTTS_CACHE_DIR
|
||||||
|
XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
AI_DNS=${WG_DNS:-1.1.1.1}
|
AI_DNS=${WG_DNS:-1.1.1.1}
|
||||||
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
||||||
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
||||||
@@ -475,7 +479,7 @@ build_and_start() {
|
|||||||
llama-fast llama-medium llama-large llama-ultra llama-experimental
|
llama-fast llama-medium llama-large llama-ultra llama-experimental
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker
|
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||||
profile-controller router open-webui
|
xtts piper tts-gateway profile-controller router open-webui
|
||||||
|
|
||||||
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
||||||
systemctl restart mike-ai-container-vpn-guard.service
|
systemctl restart mike-ai-container-vpn-guard.service
|
||||||
|
|||||||
@@ -47,7 +47,8 @@ container_healthy() {
|
|||||||
[[ $state == running && ( -z $health || $health == healthy ) ]]
|
[[ $state == running && ( -z $health || $health == healthy ) ]]
|
||||||
}
|
}
|
||||||
|
|
||||||
for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai-open-webui; do
|
for container in mike-ai-profile-controller mike-ai-router mike-ai-xtts \
|
||||||
|
mike-ai-piper mike-ai-tts-gateway mike-ai-open-webui; do
|
||||||
if container_healthy "$container"; then
|
if container_healthy "$container"; then
|
||||||
pass "$container gesund"
|
pass "$container gesund"
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
FROM python:3.12-slim
|
||||||
|
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install --no-install-recommends -y curl ffmpeg \
|
||||||
|
&& rm -rf /var/lib/apt/lists/* \
|
||||||
|
&& useradd --system --uid 10005 --home-dir /nonexistent --shell /usr/sbin/nologin tts
|
||||||
|
|
||||||
|
COPY tts_gateway.py /app/tts_gateway.py
|
||||||
|
|
||||||
|
USER 10005:10005
|
||||||
|
EXPOSE 8085
|
||||||
|
|
||||||
|
HEALTHCHECK --interval=10s --timeout=5s --retries=12 \
|
||||||
|
CMD curl -fsS http://127.0.0.1:8085/status || exit 1
|
||||||
|
|
||||||
|
ENTRYPOINT ["python", "/app/tts_gateway.py"]
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
import importlib.util
|
||||||
|
import pathlib
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
|
||||||
|
MODULE_PATH = pathlib.Path(__file__).with_name("tts_gateway.py")
|
||||||
|
SPEC = importlib.util.spec_from_file_location("tts_gateway", MODULE_PATH)
|
||||||
|
gateway = importlib.util.module_from_spec(SPEC)
|
||||||
|
SPEC.loader.exec_module(gateway)
|
||||||
|
|
||||||
|
|
||||||
|
class LanguageSegmentationTests(unittest.TestCase):
|
||||||
|
def test_german_only(self):
|
||||||
|
self.assertEqual(
|
||||||
|
gateway.segment_languages("Guten Abend, wie warm ist es heute?"),
|
||||||
|
[("de", "Guten Abend, wie warm ist es heute?")],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_english_only(self):
|
||||||
|
text = "This is a short test and it is running on the local server."
|
||||||
|
self.assertEqual(gateway.segment_languages(text), [("en", text)])
|
||||||
|
|
||||||
|
def test_mixed_compounds(self):
|
||||||
|
text = "Ich öffne das Unraid-Dashboard und prüfe die Docker-Container."
|
||||||
|
self.assertEqual(
|
||||||
|
gateway.segment_languages(text),
|
||||||
|
[
|
||||||
|
("de", "Ich öffne das "),
|
||||||
|
("en", "Unraid-Dashboard"),
|
||||||
|
("de", " und prüfe die "),
|
||||||
|
("en", "Docker-Container"),
|
||||||
|
("de", "."),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class FallbackTests(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.original_xtts = gateway.synthesize_xtts
|
||||||
|
self.original_piper = gateway.synthesize_piper
|
||||||
|
|
||||||
|
def tearDown(self):
|
||||||
|
gateway.synthesize_xtts = self.original_xtts
|
||||||
|
gateway.synthesize_piper = self.original_piper
|
||||||
|
|
||||||
|
def test_piper_is_used_when_xtts_fails(self):
|
||||||
|
def fail(*_args):
|
||||||
|
raise RuntimeError("synthetic XTTS failure")
|
||||||
|
|
||||||
|
gateway.synthesize_xtts = fail
|
||||||
|
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
|
||||||
|
self.assertEqual(
|
||||||
|
gateway.synthesize("synthetic test", "wav", 1.0),
|
||||||
|
(b"piper", "audio/wav"),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,356 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Private XTTS-first TTS gateway with a Piper fallback.
|
||||||
|
|
||||||
|
The gateway implements the narrow /status and /tts protocol already consumed
|
||||||
|
by the profile router. Request text is never logged or persisted.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
import wave
|
||||||
|
from http import HTTPStatus
|
||||||
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
|
|
||||||
|
|
||||||
|
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
|
||||||
|
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
|
||||||
|
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||||
|
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||||
|
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||||
|
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
|
||||||
|
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
|
||||||
|
MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
|
||||||
|
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
|
||||||
|
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
||||||
|
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||||
|
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||||
|
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||||
|
SILENCE_MS = int(os.getenv("XTTS_SEGMENT_SILENCE_MS", "20"))
|
||||||
|
|
||||||
|
SYNTHESIS_LOCK = threading.Lock()
|
||||||
|
STATE_LOCK = threading.Lock()
|
||||||
|
SPEAKER_LOCK = threading.Lock()
|
||||||
|
SPEAKER_CONDITIONING: dict | None = None
|
||||||
|
STATE = {
|
||||||
|
"last_backend": None,
|
||||||
|
"xtts_failures": 0,
|
||||||
|
"piper_fallbacks": 0,
|
||||||
|
"last_error": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Prefer full compounds to isolated terms. This keeps switches infrequent and
|
||||||
|
# avoids making mixed-language speech sound like a sequence of separate clips.
|
||||||
|
ENGLISH_TERMS = (
|
||||||
|
"Home Assistant", "Open WebUI", "OpenWebUI", "Unraid Dashboard",
|
||||||
|
"Unraid-Dashboard", "Docker Container", "Docker-Container",
|
||||||
|
"Server Log", "Server-Log", "GitHub Repository", "GitHub Repo",
|
||||||
|
"WireGuard Tunnel", "Cron Job", "Cronjob", "Home Server",
|
||||||
|
"API Key", "Tool Calling", "Context Window", "Prompt Injection",
|
||||||
|
"Unraid", "Docker", "Container", "Dashboard", "Server", "Log",
|
||||||
|
"OpenAI", "GitHub", "WireGuard", "Linux", "Debian", "Frontend",
|
||||||
|
"Backend", "Router", "Browser", "Web", "Token", "Prompt", "Context",
|
||||||
|
"Model", "Image", "Tool", "Workflow", "Benchmark", "Streaming",
|
||||||
|
"SSH", "MCP", "API", "CPU", "GPU", "VRAM", "RAM", "HTTP", "HTTPS",
|
||||||
|
)
|
||||||
|
TERM_PATTERN = re.compile(
|
||||||
|
r"(?<![\w])(" + "|".join(
|
||||||
|
re.escape(term) for term in sorted(ENGLISH_TERMS, key=len, reverse=True)
|
||||||
|
) + r")(?![\w])",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
GERMAN_MARKERS = {
|
||||||
|
"aber", "auch", "auf", "das", "der", "die", "ein", "eine", "für",
|
||||||
|
"ich", "ist", "kann", "mit", "nicht", "noch", "oder", "soll", "und",
|
||||||
|
"wenn", "wir", "wird", "zu",
|
||||||
|
}
|
||||||
|
ENGLISH_MARKERS = {
|
||||||
|
"a", "and", "are", "can", "for", "from", "if", "in", "is", "it",
|
||||||
|
"of", "on", "or", "please", "the", "this", "to", "with", "you",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _request(url: str, *, payload: dict | None = None,
|
||||||
|
timeout: float = 10) -> tuple[bytes, str]:
|
||||||
|
data = None
|
||||||
|
headers = {}
|
||||||
|
method = "GET"
|
||||||
|
if payload is not None:
|
||||||
|
data = json.dumps(payload, separators=(",", ":")).encode()
|
||||||
|
headers["Content-Type"] = "application/json"
|
||||||
|
method = "POST"
|
||||||
|
request = urllib.request.Request(
|
||||||
|
url, data=data, headers=headers, method=method)
|
||||||
|
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||||
|
body = response.read(MAX_AUDIO_BYTES + 1)
|
||||||
|
if len(body) > MAX_AUDIO_BYTES:
|
||||||
|
raise RuntimeError("upstream audio response is too large")
|
||||||
|
return body, response.headers.get_content_type()
|
||||||
|
|
||||||
|
|
||||||
|
def _json(url: str, timeout: float = 10) -> dict | list:
|
||||||
|
body, _ = _request(url, timeout=timeout)
|
||||||
|
return json.loads(body)
|
||||||
|
|
||||||
|
|
||||||
|
def _reachable(url: str, path: str, timeout: float = 2) -> bool:
|
||||||
|
try:
|
||||||
|
_request(f"{url}{path}", timeout=timeout)
|
||||||
|
return True
|
||||||
|
except (OSError, ValueError, RuntimeError, urllib.error.URLError):
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _looks_english(text: str) -> bool:
|
||||||
|
words = re.findall(r"[A-Za-zÀ-ÿ]+", text.lower())
|
||||||
|
if not words:
|
||||||
|
return False
|
||||||
|
german = sum(word in GERMAN_MARKERS for word in words)
|
||||||
|
english = sum(word in ENGLISH_MARKERS for word in words)
|
||||||
|
return english >= 2 and english > german * 1.5 and not re.search(r"[äöüß]", text.lower())
|
||||||
|
|
||||||
|
|
||||||
|
def segment_languages(text: str) -> list[tuple[str, str]]:
|
||||||
|
"""Return a compact German/English segment sequence."""
|
||||||
|
if _looks_english(text):
|
||||||
|
return [("en", text)]
|
||||||
|
if DEFAULT_LANGUAGE != "de":
|
||||||
|
return [(DEFAULT_LANGUAGE, text)]
|
||||||
|
|
||||||
|
segments: list[tuple[str, str]] = []
|
||||||
|
cursor = 0
|
||||||
|
for match in TERM_PATTERN.finditer(text):
|
||||||
|
if match.start() > cursor:
|
||||||
|
segments.append(("de", text[cursor:match.start()]))
|
||||||
|
segments.append(("en", match.group(0)))
|
||||||
|
cursor = match.end()
|
||||||
|
if cursor < len(text):
|
||||||
|
segments.append(("de", text[cursor:]))
|
||||||
|
if not segments:
|
||||||
|
return [("de", text)]
|
||||||
|
|
||||||
|
merged: list[tuple[str, str]] = []
|
||||||
|
for language, part in segments:
|
||||||
|
if not part:
|
||||||
|
continue
|
||||||
|
if merged and merged[-1][0] == language:
|
||||||
|
previous_language, previous_text = merged[-1]
|
||||||
|
merged[-1] = (previous_language, previous_text + part)
|
||||||
|
else:
|
||||||
|
merged.append((language, part))
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
def _speaker_conditioning() -> dict:
|
||||||
|
global SPEAKER_CONDITIONING
|
||||||
|
with SPEAKER_LOCK:
|
||||||
|
if SPEAKER_CONDITIONING is not None:
|
||||||
|
return SPEAKER_CONDITIONING
|
||||||
|
speakers = _json(f"{XTTS_URL}/studio_speakers", XTTS_TIMEOUT)
|
||||||
|
if not isinstance(speakers, dict) or XTTS_SPEAKER not in speakers:
|
||||||
|
raise RuntimeError("configured XTTS speaker is unavailable")
|
||||||
|
selected = speakers[XTTS_SPEAKER]
|
||||||
|
SPEAKER_CONDITIONING = {
|
||||||
|
"speaker_embedding": selected["speaker_embedding"],
|
||||||
|
"gpt_cond_latent": selected["gpt_cond_latent"],
|
||||||
|
}
|
||||||
|
return SPEAKER_CONDITIONING
|
||||||
|
|
||||||
|
|
||||||
|
def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes:
|
||||||
|
payload = {
|
||||||
|
**conditioning,
|
||||||
|
"text": text,
|
||||||
|
"language": language,
|
||||||
|
"add_wav_header": True,
|
||||||
|
"stream_chunk_size": "20",
|
||||||
|
}
|
||||||
|
audio, _ = _request(f"{XTTS_URL}/tts_stream", payload=payload,
|
||||||
|
timeout=XTTS_TIMEOUT)
|
||||||
|
if len(audio) < 44 or audio[:4] != b"RIFF" or audio[8:12] != b"WAVE":
|
||||||
|
raise RuntimeError("XTTS returned invalid WAV data")
|
||||||
|
return audio[44:]
|
||||||
|
|
||||||
|
|
||||||
|
def _wav(pcm: bytes) -> bytes:
|
||||||
|
output = io.BytesIO()
|
||||||
|
with wave.open(output, "wb") as wav_file:
|
||||||
|
wav_file.setnchannels(1)
|
||||||
|
wav_file.setsampwidth(2)
|
||||||
|
wav_file.setframerate(24000)
|
||||||
|
wav_file.writeframes(pcm)
|
||||||
|
return output.getvalue()
|
||||||
|
|
||||||
|
|
||||||
|
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||||
|
if output_format == "wav" and speed == 1.0:
|
||||||
|
return wav_bytes, "audio/wav"
|
||||||
|
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \
|
||||||
|
if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"]
|
||||||
|
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
|
||||||
|
"-i", "pipe:0"]
|
||||||
|
if speed != 1.0:
|
||||||
|
command.extend(["-filter:a", f"atempo={speed:.4f}"])
|
||||||
|
command.extend([*codec, "pipe:1"])
|
||||||
|
result = subprocess.run(
|
||||||
|
command, input=wav_bytes, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||||
|
check=False, timeout=120)
|
||||||
|
if result.returncode != 0 or not result.stdout:
|
||||||
|
raise RuntimeError("audio conversion failed")
|
||||||
|
return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav"
|
||||||
|
|
||||||
|
|
||||||
|
def synthesize_xtts(text: str, output_format: str,
|
||||||
|
speed: float) -> tuple[bytes, str]:
|
||||||
|
conditioning = _speaker_conditioning()
|
||||||
|
pcm_parts: list[bytes] = []
|
||||||
|
silence = b"\x00\x00" * int(24000 * max(0, SILENCE_MS) / 1000)
|
||||||
|
for language, segment in segment_languages(text):
|
||||||
|
if not segment.strip():
|
||||||
|
continue
|
||||||
|
pcm_parts.append(_xtts_pcm(segment, language, conditioning))
|
||||||
|
if silence:
|
||||||
|
pcm_parts.append(silence)
|
||||||
|
if pcm_parts and silence:
|
||||||
|
pcm_parts.pop()
|
||||||
|
if not pcm_parts:
|
||||||
|
raise RuntimeError("no speech segments generated")
|
||||||
|
return _convert(_wav(b"".join(pcm_parts)), output_format, speed)
|
||||||
|
|
||||||
|
|
||||||
|
def synthesize_piper(text: str, output_format: str,
|
||||||
|
speed: float) -> tuple[bytes, str]:
|
||||||
|
return _request(
|
||||||
|
f"{PIPER_URL}/tts",
|
||||||
|
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||||
|
"format": output_format},
|
||||||
|
timeout=PIPER_TIMEOUT,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||||
|
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
||||||
|
if acquired:
|
||||||
|
try:
|
||||||
|
audio = synthesize_xtts(text, output_format, speed)
|
||||||
|
with STATE_LOCK:
|
||||||
|
STATE["last_backend"] = "xtts-v2"
|
||||||
|
STATE["last_error"] = None
|
||||||
|
return audio
|
||||||
|
except Exception as exc: # fallback must cover all XTTS failures
|
||||||
|
with STATE_LOCK:
|
||||||
|
STATE["xtts_failures"] += 1
|
||||||
|
STATE["last_error"] = type(exc).__name__
|
||||||
|
finally:
|
||||||
|
SYNTHESIS_LOCK.release()
|
||||||
|
else:
|
||||||
|
with STATE_LOCK:
|
||||||
|
STATE["xtts_failures"] += 1
|
||||||
|
STATE["last_error"] = "queue-timeout"
|
||||||
|
|
||||||
|
audio = synthesize_piper(text, output_format, speed)
|
||||||
|
with STATE_LOCK:
|
||||||
|
STATE["last_backend"] = "piper"
|
||||||
|
STATE["piper_fallbacks"] += 1
|
||||||
|
return audio
|
||||||
|
|
||||||
|
|
||||||
|
class Handler(BaseHTTPRequestHandler):
|
||||||
|
protocol_version = "HTTP/1.1"
|
||||||
|
|
||||||
|
def log_message(self, fmt: str, *args: object) -> None:
|
||||||
|
# Never log request URLs, bodies, synthesized text or speaker vectors.
|
||||||
|
print(f"tts-gateway: {self.command} -> {args[1] if len(args) > 1 else '-'}")
|
||||||
|
|
||||||
|
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
|
||||||
|
self.send_response(status)
|
||||||
|
self.send_header("Content-Type", content_type)
|
||||||
|
self.send_header("Content-Length", str(len(body)))
|
||||||
|
self.send_header("Cache-Control", "no-store")
|
||||||
|
self.end_headers()
|
||||||
|
self.wfile.write(body)
|
||||||
|
|
||||||
|
def send_json(self, status: int, payload: dict) -> None:
|
||||||
|
self.send_bytes(status, json.dumps(payload, separators=(",", ":")).encode(),
|
||||||
|
"application/json")
|
||||||
|
|
||||||
|
def do_GET(self) -> None: # noqa: N802
|
||||||
|
if self.path != "/status":
|
||||||
|
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||||
|
return
|
||||||
|
primary_ready = _reachable(XTTS_URL, "/languages")
|
||||||
|
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||||
|
with STATE_LOCK:
|
||||||
|
state = dict(STATE)
|
||||||
|
self.send_json(
|
||||||
|
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||||
|
{
|
||||||
|
"ready": fallback_ready,
|
||||||
|
"engine": "xtts-v2-with-piper-fallback",
|
||||||
|
"model": "xtts-v2",
|
||||||
|
"voices": [VOICE_ALIAS],
|
||||||
|
"speaker": XTTS_SPEAKER,
|
||||||
|
"primary_ready": primary_ready,
|
||||||
|
"fallback_ready": fallback_ready,
|
||||||
|
**state,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
def do_POST(self) -> None: # noqa: N802
|
||||||
|
if self.path != "/tts":
|
||||||
|
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
length = int(self.headers.get("Content-Length", "0"))
|
||||||
|
except ValueError:
|
||||||
|
length = 0
|
||||||
|
if length <= 0 or length > MAX_REQUEST_BYTES:
|
||||||
|
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE,
|
||||||
|
{"error": "invalid request size"})
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
request = json.loads(self.rfile.read(length))
|
||||||
|
text = request.get("text", "")
|
||||||
|
voice = request.get("voice", VOICE_ALIAS)
|
||||||
|
output_format = request.get("format", "mp3")
|
||||||
|
speed = float(request.get("speed", 1.0))
|
||||||
|
except (json.JSONDecodeError, TypeError, ValueError):
|
||||||
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
|
||||||
|
return
|
||||||
|
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
|
||||||
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
|
||||||
|
return
|
||||||
|
if voice != VOICE_ALIAS:
|
||||||
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||||
|
return
|
||||||
|
if output_format not in {"wav", "mp3"}:
|
||||||
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||||
|
return
|
||||||
|
if not 0.5 <= speed <= 2.0:
|
||||||
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
|
||||||
|
return
|
||||||
|
started = time.monotonic()
|
||||||
|
try:
|
||||||
|
audio, content_type = synthesize(text.strip(), output_format, speed)
|
||||||
|
except Exception as exc:
|
||||||
|
with STATE_LOCK:
|
||||||
|
STATE["last_error"] = type(exc).__name__
|
||||||
|
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||||
|
{"error": "all local speech backends failed"})
|
||||||
|
return
|
||||||
|
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
|
||||||
|
f"{time.monotonic() - started:.2f}s")
|
||||||
|
self.send_bytes(HTTPStatus.OK, audio, content_type)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper")
|
||||||
|
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||||
@@ -172,10 +172,11 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
|
|||||||
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
||||||
cd "$STACK_DIR"
|
cd "$STACK_DIR"
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
||||||
router open-webui
|
xtts piper tts-gateway router open-webui
|
||||||
|
|
||||||
deadline=$((SECONDS + 180))
|
deadline=$((SECONDS + 180))
|
||||||
for container in mike-ai-router mike-ai-open-webui; do
|
for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \
|
||||||
|
mike-ai-router mike-ai-open-webui; do
|
||||||
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||||
"$container" 2>/dev/null || true) == healthy ]]; do
|
"$container" 2>/dev/null || true) == healthy ]]; do
|
||||||
(( SECONDS < deadline )) || die "$container wurde nicht rechtzeitig gesund."
|
(( SECONDS < deadline )) || die "$container wurde nicht rechtzeitig gesund."
|
||||||
|
|||||||
Executable
+20
@@ -0,0 +1,20 @@
|
|||||||
|
#!/bin/sh
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
backup_dir=${1:?usage: rollback-tts-production.sh BACKUP_DIR}
|
||||||
|
|
||||||
|
case "$backup_dir" in
|
||||||
|
/data/backups/mike-ai/*) ;;
|
||||||
|
*) echo "refusing unexpected backup path" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
test -f "$backup_dir/compose.yaml"
|
||||||
|
test -f "$backup_dir/stack.env"
|
||||||
|
|
||||||
|
cp -a "$backup_dir/compose.yaml" /opt/mike-ai/stack/compose.yaml
|
||||||
|
cp -a "$backup_dir/stack.env" /etc/mike-ai/stack.env
|
||||||
|
chmod 0600 /etc/mike-ai/stack.env
|
||||||
|
|
||||||
|
cd /opt/mike-ai/stack
|
||||||
|
docker compose --env-file /etc/mike-ai/stack.env up -d --no-deps --force-recreate router
|
||||||
|
docker rm -f mike-ai-tts-gateway mike-ai-xtts >/dev/null 2>&1 || true
|
||||||
Reference in New Issue
Block a user