From f90fc93f9cf3f304144715a37710f92aea59979f Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sun, 23 Aug 2026 12:22:59 +0200 Subject: [PATCH] Add XTTS primary voice with Piper fallback --- .env.example | 3 + README.md | 7 +- compose.yaml | 76 +++- config/install.env.example | 4 + docs/ARCHITECTURE.md | 22 +- docs/COMPONENTS.md | 4 +- docs/CURRENT_REFERENCE.md | 21 +- docs/DISASTER_RECOVERY.md | 7 +- docs/INSTALLATION.md | 14 +- docs/RECOVERY_REQUIREMENTS.md | 5 +- docs/XTTS_EVALUATION_2026-08-23.md | 113 ++++++ install.sh | 8 +- platform/checks/verify-platform.sh | 3 +- platform/docker/tts-gateway/Dockerfile | 16 + .../docker/tts-gateway/test_tts_gateway.py | 59 +++ platform/docker/tts-gateway/tts_gateway.py | 356 ++++++++++++++++++ .../migration/restore-reference-backup.sh | 5 +- platform/scripts/rollback-tts-production.sh | 20 + 18 files changed, 715 insertions(+), 28 deletions(-) create mode 100644 docs/XTTS_EVALUATION_2026-08-23.md create mode 100644 platform/docker/tts-gateway/Dockerfile create mode 100644 platform/docker/tts-gateway/test_tts_gateway.py create mode 100644 platform/docker/tts-gateway/tts_gateway.py create mode 100755 platform/scripts/rollback-tts-production.sh diff --git a/.env.example b/.env.example index 3bd90e9..10a12ec 100644 --- a/.env.example +++ b/.env.example @@ -7,6 +7,9 @@ WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER OPENWEBUI_IMAGE=ghcr.io/open-webui/open-webui:v0.9.5 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high +XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 +XTTS_CACHE_DIR=/data/models/xtts-v2-cache +XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b AI_DNS=192.168.1.1 FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf diff --git a/README.md b/README.md index 5ec9999..c5591cc 100644 --- a/README.md +++ b/README.md @@ -52,10 +52,13 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt. | llama.cpp | nur Docker-intern | Inferenz und integrierte Vision | | Profile Controller | nur Docker-intern | eng begrenzter Profil-/FLUX-Hot-Swap | | FLUX Worker | nur Docker-intern, normalerweise gestoppt | Bildgenerierung auf RTX 5080 | -| Piper | nur Docker-intern | lokale deutsche Sprachausgabe | +| XTTS-v2 | nur Docker-intern, RTX 3060 | primäre mehrsprachige Sprachausgabe | +| TTS Gateway | nur Docker-intern | Annmarie Nele, Queue und Piper-Fallback | +| Piper | nur Docker-intern, CPU | ausfallsichere deutsche Ersatzstimme | | MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid | -Piper-TTS und der FLUX.2-Klein-Hot-Swap sind reproduzierbare Kerndienste; STT +XTTS-v2, TTS-Gateway, Piper-Fallback und der FLUX.2-Klein-Hot-Swap sind +reproduzierbare Kerndienste; STT bleibt optional. Web-, Home-Assistant-, ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter `platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne diff --git a/compose.yaml b/compose.yaml index 14faa08..3eae357 100644 --- a/compose.yaml +++ b/compose.yaml @@ -529,7 +529,11 @@ services: CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" ENABLE_IMAGE_GENERATION: "true" ENABLE_TTS: "true" - TTS_WORKER_URL: http://piper:8085 + # Stable OpenAI compatibility names remain piper/alloy because an + # existing Open WebUI database persists those values. The gateway maps + # alloy to XTTS speaker Annmarie Nele and automatically falls back to + # Piper if XTTS is unavailable, busy or returns an error. + TTS_WORKER_URL: http://tts-gateway:8085 TTS_MODEL: piper TTS_VOICES: alloy TTS_DEFAULT_VOICE: alloy @@ -555,6 +559,8 @@ services: condition: service_healthy piper: condition: service_healthy + tts-gateway: + condition: service_healthy flux-worker: build: @@ -622,6 +628,74 @@ services: retries: 30 start_period: 120s + xtts: + image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90} + container_name: mike-ai-xtts + restart: unless-stopped + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: + - ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} + capabilities: [gpu] + read_only: true + shm_size: 1g + tmpfs: + - /tmp:size=1g,mode=1777 + - /root/.cache:size=2g,mode=0700 + volumes: + - "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts" + environment: + COQUI_TOS_AGREED: "1" + NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + CUDA_VISIBLE_DEVICES: "0" + NUM_THREADS: "4" + networks: [frontend] + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + healthcheck: + test: [CMD, curl, -fsS, "http://127.0.0.1/languages"] + interval: 10s + timeout: 5s + retries: 36 + start_period: 240s + + tts-gateway: + build: + context: platform/docker/tts-gateway + image: mike-ai/tts-gateway:local + container_name: mike-ai-tts-gateway + restart: unless-stopped + read_only: true + tmpfs: + - /tmp:size=256m,mode=1777 + environment: + TTS_GATEWAY_HOST: 0.0.0.0 + TTS_GATEWAY_PORT: "8085" + XTTS_URL: http://xtts:80 + PIPER_URL: http://piper:8085 + TTS_VOICE_ALIAS: alloy + XTTS_SPEAKER: Annmarie Nele + TTS_DEFAULT_LANGUAGE: de + XTTS_QUEUE_TIMEOUT: "15" + XTTS_TIMEOUT: "120" + PIPER_TIMEOUT: "120" + networks: [frontend] + depends_on: + piper: + condition: service_healthy + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + healthcheck: + test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"] + interval: 10s + timeout: 5s + retries: 12 + start_period: 10s + open-webui: image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5} container_name: mike-ai-open-webui diff --git a/config/install.env.example b/config/install.env.example index 61cabf3..9f0b65b 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -91,3 +91,7 @@ OPENWEBUI_ENABLE_SIGNUP=false OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high +XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 +XTTS_CACHE_DIR=/data/models/xtts-v2-cache +# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here. +XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index cbf78df..e939272 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -24,7 +24,9 @@ Heimnetz / VPN-Clients +-- llama-uncensored (80K, Abliterated, dual GPU) +-- llama-experimental +-- llama-ultra (256K, text-only, dual GPU) - +-- Piper-TTS (CPU, nur intern) + +-- TTS-Gateway + | +-- XTTS-v2 / Annmarie Nele (RTX 3060, primär) + | +-- Piper-TTS (CPU, automatischer Fallback) +-- internes MCP-Netz +-- Web-MCP + TinySearch + SearXNG +-- Home-Assistant-MCP-Relay @@ -41,7 +43,9 @@ Heimnetz / VPN-Clients | Profile Router | nur Docker-intern | OpenAI-API und Profilwahl | | Profile Controller | nein | startet ausschließlich fest erlaubte Profile | | llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision | -| Piper | nein | lokale deutsche Text-to-Speech-Ausgabe | +| XTTS-v2 | nein | primäre deutsche/englische Text-to-Speech-Ausgabe auf RTX 3060 | +| TTS-Gateway | nein | serialisiert XTTS, segmentiert Sprachwechsel und fällt auf Piper zurück | +| Piper | nein | CPU-basierte Text-to-Speech-Rückfallebene | | MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche | | SearXNG/TinySearch | nein | Suchbackend des Web-MCP | @@ -99,10 +103,16 @@ Ultra bleibt für maximalen Kontext bewusst text-only. ## Optionale Erweiterungen -Piper läuft als eigener CPU-Container und wird von Open WebUI über den Router -angesprochen. Sein Port wird nicht veröffentlicht. Das Stimmenmodell liegt im -persistenten Volume `piper-data` und wird beim ersten Start reproduzierbar -nachgeladen. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert. +Open WebUI spricht ausschließlich den Router an. Dieser reicht TTS intern an +das TTS-Gateway weiter. Das Gateway nutzt primär XTTS-v2 mit der Stimme +`Annmarie Nele` auf der RTX 3060. Deutsche Texte werden an bekannten +englischen IT-Begriffen segmentiert; reine englische Texte laufen vollständig +mit `language=en`. Da der offizielle XTTS-Streamingserver nur einen Auftrag +gleichzeitig unterstützt, serialisiert das Gateway die Aufträge. Bei Fehler, +Timeout oder belegter Queue übernimmt automatisch Piper auf der CPU. Kein +TTS-Port wird veröffentlicht. Der äußere Kompatibilitätsname bleibt bewusst +`piper/alloy`, damit persistente Open-WebUI-Einstellungen nach Updates und +Restores gültig bleiben. Bildgenerierung und Whisper bleiben im Basissystem deaktiviert. Home Assistant, ARR und Unraid sind vorbereitete MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen diff --git a/docs/COMPONENTS.md b/docs/COMPONENTS.md index 3f7906e..60297e0 100644 --- a/docs/COMPONENTS.md +++ b/docs/COMPONENTS.md @@ -12,7 +12,9 @@ | ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional | | Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional | | Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional | -| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | eigener interner CPU-Container, Stimme `de_DE-thorsten-high` | Kern | +| XTTS-v2 | Coqui, offizielles CUDA-12.1-Image per Digest | RTX-3060-Container, Stimme `Annmarie Nele`, CPML | Kern | +| TTS-Gateway | `platform/docker/tts-gateway/` | interne Queue, Deutsch/Englisch-Segmentierung und Piper-Fallback | Kern | +| Piper TTS | Open Home Foundation, `piper-tts` 1.6.0 | interner CPU-Fallback, Stimme `de_DE-thorsten-high` | Kern | | FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional | | LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional | | Glances | Distribution | nur Betriebsrolle dokumentiert | optional | diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index 8fc63cd..8c5aec3 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -126,15 +126,20 @@ leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus. ### XTTS -Dieser Abschnitt beschreibt ausschließlich den alten Referenzhost. Im neuen -Docker-Zielsystem ersetzt Piper (`de_DE-thorsten-high`) diesen Dienst. +Der aktuelle Docker-Stack nutzt Coqui XTTS-v2 als primäre Sprachausgabe. +Der isolierte Eignungs- und Ausfalltest ist in +[`XTTS_EVALUATION_2026-08-23.md`](XTTS_EVALUATION_2026-08-23.md) dokumentiert. -- Modell: Coqui XTTS-v2 -- CPU-only -- Stimme: `claribel` -- Deutsch und Englisch -- Port 8085, auf dem alten Host noch im LAN gebunden -- eigenes Python-3.11-Venv +- Modell: Coqui XTTS-v2, offizielles CUDA-12.1-Image per Digest gepinnt +- GPU: ausschließlich RTX 3060 über ihre stabile GPU-UUID +- Stimme: `Annmarie Nele` +- Deutsch und Englisch; bekannte englische IT-Begriffe werden segmentiert +- kein veröffentlichter Port, nur Docker-intern erreichbar +- serielles TTS-Gateway vor XTTS, weil der Server nur einen Auftrag zugleich + zuverlässig verarbeitet +- Piper mit `de_DE-thorsten-high` bleibt als automatischer CPU-Fallback aktiv +- OpenWebUI behält aus Kompatibilitätsgründen `model=piper` und `voice=alloy`; + der Router leitet diese Werte an das Gateway weiter ## Websuche diff --git a/docs/DISASTER_RECOVERY.md b/docs/DISASTER_RECOVERY.md index 3a9a241..6ba78ac 100644 --- a/docs/DISASTER_RECOVERY.md +++ b/docs/DISASTER_RECOVERY.md @@ -76,8 +76,11 @@ laufen, sondern alle fachlichen Funktionen geprüft wurden. - [ ] FLUX erzeugt Standard- und High-Bild - [ ] Qwen-Profil wird nach FLUX wiederhergestellt - [ ] Whisper transkribiert deutsche und englische Testdatei -- [ ] Piper ist gesund und erzeugt über den Router deutsche WAV- und MP3-Ausgabe -- [ ] Stimme und `piper-tts`-Version entsprechen der Installationskonfiguration +- [ ] XTTS-v2 läuft ausschließlich auf der RTX 3060 und meldet `Annmarie Nele` +- [ ] TTS-Gateway erzeugt über den Router deutsche und englische WAV-/MP3-Ausgabe +- [ ] englische IT-Begriffe im deutschen Satz werden sprachlich segmentiert +- [ ] gestopptes XTTS fällt ohne Router-/OpenWebUI-Neustart auf Piper zurück +- [ ] Piper-Fallback und `piper-tts`-Version entsprechen der Installationskonfiguration - [ ] STT/TTS blockieren das Textmodell nicht unzulässig ## Phase F – Sicherheitsprüfung diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md index b5b57a8..e920846 100644 --- a/docs/INSTALLATION.md +++ b/docs/INSTALLATION.md @@ -84,7 +84,9 @@ OpenSSH-Dienst des Hosts. Die TTS-Verbindung wird für eine frische Open-WebUI-Datenbank automatisch als OpenAI-kompatibler Audio-Endpunkt des Routers vorbelegt. Der Router reicht sie -intern an Piper weiter; Port 8085 wird nicht am Host veröffentlicht. Ein +intern an das TTS-Gateway weiter. Primär spricht XTTS-v2 mit `Annmarie Nele` +auf der RTX 3060; bei Fehlern oder Queue-Timeout übernimmt Piper auf der CPU. +Der Port 8085 wird nicht am Host veröffentlicht. Ein Restore setzt zusätzlich die vier persistenten Audiofelder gezielt neu, damit alte Werte wie `tts-1` oder `coral` die Compose-Vorgaben nicht überstimmen. Ein Ende-zu-Ende-Test ohne Ausgabe des API-Schlüssels: @@ -95,9 +97,17 @@ curl -fsS http://127.0.0.1:8081/v1/audio/speech \ -H "Authorization: Bearer $ROUTER_API_KEY" \ -H 'Content-Type: application/json' \ -d '{"model":"piper","voice":"alloy","input":"Hallo von Athena.","response_format":"mp3"}' \ - -o /tmp/athena-piper-test.mp3 + -o /tmp/athena-tts-test.mp3 ``` +Der beibehaltene API-Name `piper/alloy` ist eine Kompatibilitätsschnittstelle; +bei gesundem XTTS stammt die Ausgabe von `Annmarie Nele`. Der interne Status +des TTS-Gateways nennt `last_backend`, `primary_ready`, `fallback_ready` und +die Zahl der Piper-Rückfälle. Ein Fallback-Test stoppt ausschließlich XTTS, +erzeugt einen synthetischen Satz über denselben Router-Endpunkt und startet +XTTS anschließend wieder. OpenWebUI und Router müssen dafür nicht geändert +oder neu gestartet werden. + Zusätzlich prüfen: Standort-LAN sieht keine KI-Ports; Heimnetz erreicht beide; gestopptes VPN-Gateway lässt KI-Container nicht ins Internet; jeder Profilwechsel startet exakt einen llama-Container; Text, Tool Call, Bild und Sprachausgabe funktionieren. diff --git a/docs/RECOVERY_REQUIREMENTS.md b/docs/RECOVERY_REQUIREMENTS.md index ca36843..1eddf99 100644 --- a/docs/RECOVERY_REQUIREMENTS.md +++ b/docs/RECOVERY_REQUIREMENTS.md @@ -35,7 +35,10 @@ Pflichtrollen: - BF16 Vision-Projektor - Whisper large-v3-turbo - FLUX.2 klein -- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high` +- XTTS-v2, per Digest gepinntes CUDA-12.1-Image und CPML-Akzeptanz +- XTTS-Stimme `Annmarie Nele`, RTX-3060-UUID und persistenter Modellcache +- internes TTS-Gateway mit Queue, Sprachsegmentierung und Piper-Fallback +- Piper `piper-tts` 1.6.0 und Stimme `de_DE-thorsten-high` als CPU-Fallback ## 2. Externe Komponenten und Commits – teilweise gesichert diff --git a/docs/XTTS_EVALUATION_2026-08-23.md b/docs/XTTS_EVALUATION_2026-08-23.md new file mode 100644 index 0000000..ccf8182 --- /dev/null +++ b/docs/XTTS_EVALUATION_2026-08-23.md @@ -0,0 +1,113 @@ +# XTTS-v2 GPU evaluation on Athena (2026-08-23) + +## Purpose and safety boundary + +This was an isolated, reversible evaluation of Coqui XTTS-v2 as a possible +replacement for Piper. The user accepted the Coqui Public Model License for +this private test. + +- Official image: `ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121` +- Pulled digest: `sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90` +- Test container: `mike-ai-xtts-test` +- GPU visibility: RTX 3060 only +- Host binding: `127.0.0.1:18105` only +- Restart policy: `no` +- Model cache: `/data/xtts-test/cache` +- Piper, Open WebUI and the router were not reconfigured. + +The official server describes itself as a demo server. In particular, it does +not support concurrent streaming requests and is not an OpenAI-compatible +production endpoint. A queueing/OpenAI compatibility proxy is therefore +required before integration with Open WebUI. + +## XTTS resource use + +With the Medium profile already running, XTTS increased RTX 3060 use from +about 4,471 MiB to about 6,419 MiB. XTTS therefore occupied approximately +1,948 MiB and left about 5,492 MiB free. It did not use the RTX 5080. + +With Ultra (256K) and XTTS loaded together: + +| GPU | Used | Free | +|---|---:|---:| +| RTX 3060 12 GB | 8,669 MiB | 3,242 MiB | +| RTX 5080 16 GB | 15,770 MiB | 89 MiB | + +The combination loaded successfully without OOM. This confirms that XTTS fits +even beside the largest standard text profile. The RTX 5080 must remain +unavailable to XTTS because Ultra already fills it almost completely. + +## Streaming measurements + +The initial measurements used built-in female speaker `Ana Florence`. A +subsequent five-voice German comparison selected **`Annmarie Nele`** as the +production voice. Tests used harmless synthetic text. + +| Test | First audio | Generation time | Produced audio | RTF | +|---|---:|---:|---:|---:| +| German | 0.701 s | 2.889 s | 6.965 s | 0.415 | +| English | 0.305 s | 1.330 s | 3.989 s | 0.333 | +| German sentence with English IT terms | 0.309 s | 2.496 s | 7.339 s | 0.340 | + +After warm-up, audio starts after roughly 0.3 seconds and synthesis is around +2.4 to 3 times faster than real time. Perceived Open WebUI latency also +includes Qwen's time to finish the first sentence and proxy buffering. + +## Effect on Qwen throughput + +| Profile | XTTS state | Generation speed | +|---|---|---:| +| Medium 160K | loaded but idle | 71.92 token/s | +| Medium 160K | actively speaking | 59.90 token/s | +| Ultra 256K | loaded but idle | 66.89 token/s | +| Ultra 256K | actively speaking | 55.27 token/s | + +Active synthesis costs roughly 17% of Qwen generation speed because Qwen also +uses the RTX 3060. The slowdown ends with the speech request. Merely keeping +XTTS resident did not cause instability. + +## Result and recommendation + +XTTS-v2 is technically viable on the RTX 3060 and fits alongside every current +profile, including Ultra 256K. It provides early streaming and substantially +more natural multilingual speech than the current German-only Piper voice. + +The production design keeps Piper and adds a small internal proxy that provides: + +1. OpenAI-compatible `/v1/audio/speech` input and output. +2. A one-request queue because the official XTTS server has no concurrency. +3. German/English text segmentation so English product names are synthesized + with `language=en` while surrounding German remains `language=de`. +4. Cached speaker conditioning and a fixed allowlist of voices. +5. Health checks, bounded timeouts and automatic fallback to Piper. + +This gateway now lives under `platform/docker/tts-gateway/`. The externally +visible compatibility values remain `model=piper` and `voice=alloy`; internally +that alias selects `Annmarie Nele` whenever XTTS is healthy. + +## Production result and rollback + +After the isolated evaluation, the compatibility gateway was tested in three +stages and then deployed to production: + +1. Healthy XTTS produced valid WAV through the router-compatible endpoint. +2. XTTS was deliberately stopped; the same endpoint returned valid Piper WAV. +3. XTTS was restarted and automatically became the active backend again. + +The production services are `mike-ai-xtts` and `mike-ai-tts-gateway`, both +Docker-internal. Piper remained healthy throughout. OpenWebUI required no +configuration or database change. The router's public compatibility values +remain `model=piper` and `voice=alloy`. + +The initial Compose GPU declaration exposed both NVIDIA cards and caused XTTS +to select the nearly full RTX 5080. This was caught before the router switch. +The final declaration uses a Docker device reservation with the stable RTX +3060 UUID; inspecting the container must show exactly that UUID in +`DeviceRequests`. + +The verified pre-deployment state is backed up below +`/data/backups/mike-ai/20260823-xtts-production`. The reusable rollback helper +is `platform/scripts/rollback-tts-production.sh`; it restores the saved Compose +and environment files, recreates the old Piper-connected router and removes +only XTTS and its gateway. Both VPN and university-network SSH paths were +verified after deployment. diff --git a/install.sh b/install.sh index 279a0a0..7763cf0 100755 --- a/install.sh +++ b/install.sh @@ -283,7 +283,8 @@ setup_wireguard() { install_stack_files() { log "Stackdateien installieren" - install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$STATE_DIR/backups" + XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache} + install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups" rsync -a --delete --exclude .git --exclude '*.local.*' \ --exclude config/install.env "$ROOT_DIR/" "$STACK_DIR/" install -d -m 0700 "$SECRETS_DIR" @@ -313,6 +314,9 @@ OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false} OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0} PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high} +XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90} +XTTS_CACHE_DIR=$XTTS_CACHE_DIR +XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} AI_DNS=${WG_DNS:-1.1.1.1} FAST_MODEL_FILE=$FAST_MODEL_FILE MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE @@ -475,7 +479,7 @@ build_and_start() { llama-fast llama-medium llama-large llama-ultra llama-experimental docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ - profile-controller router open-webui + xtts piper tts-gateway profile-controller router open-webui if [[ ${WIREGUARD_MODE:-container} == container ]]; then systemctl restart mike-ai-container-vpn-guard.service diff --git a/platform/checks/verify-platform.sh b/platform/checks/verify-platform.sh index c431a58..236cf90 100755 --- a/platform/checks/verify-platform.sh +++ b/platform/checks/verify-platform.sh @@ -47,7 +47,8 @@ container_healthy() { [[ $state == running && ( -z $health || $health == healthy ) ]] } -for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai-open-webui; do +for container in mike-ai-profile-controller mike-ai-router mike-ai-xtts \ + mike-ai-piper mike-ai-tts-gateway mike-ai-open-webui; do if container_healthy "$container"; then pass "$container gesund" else diff --git a/platform/docker/tts-gateway/Dockerfile b/platform/docker/tts-gateway/Dockerfile new file mode 100644 index 0000000..45d095a --- /dev/null +++ b/platform/docker/tts-gateway/Dockerfile @@ -0,0 +1,16 @@ +FROM python:3.12-slim + +RUN apt-get update \ + && apt-get install --no-install-recommends -y curl ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && useradd --system --uid 10005 --home-dir /nonexistent --shell /usr/sbin/nologin tts + +COPY tts_gateway.py /app/tts_gateway.py + +USER 10005:10005 +EXPOSE 8085 + +HEALTHCHECK --interval=10s --timeout=5s --retries=12 \ + CMD curl -fsS http://127.0.0.1:8085/status || exit 1 + +ENTRYPOINT ["python", "/app/tts_gateway.py"] diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py new file mode 100644 index 0000000..71e8dc5 --- /dev/null +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -0,0 +1,59 @@ +import importlib.util +import pathlib +import unittest + + +MODULE_PATH = pathlib.Path(__file__).with_name("tts_gateway.py") +SPEC = importlib.util.spec_from_file_location("tts_gateway", MODULE_PATH) +gateway = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(gateway) + + +class LanguageSegmentationTests(unittest.TestCase): + def test_german_only(self): + self.assertEqual( + gateway.segment_languages("Guten Abend, wie warm ist es heute?"), + [("de", "Guten Abend, wie warm ist es heute?")], + ) + + def test_english_only(self): + text = "This is a short test and it is running on the local server." + self.assertEqual(gateway.segment_languages(text), [("en", text)]) + + def test_mixed_compounds(self): + text = "Ich öffne das Unraid-Dashboard und prüfe die Docker-Container." + self.assertEqual( + gateway.segment_languages(text), + [ + ("de", "Ich öffne das "), + ("en", "Unraid-Dashboard"), + ("de", " und prüfe die "), + ("en", "Docker-Container"), + ("de", "."), + ], + ) + + +class FallbackTests(unittest.TestCase): + def setUp(self): + self.original_xtts = gateway.synthesize_xtts + self.original_piper = gateway.synthesize_piper + + def tearDown(self): + gateway.synthesize_xtts = self.original_xtts + gateway.synthesize_piper = self.original_piper + + def test_piper_is_used_when_xtts_fails(self): + def fail(*_args): + raise RuntimeError("synthetic XTTS failure") + + gateway.synthesize_xtts = fail + gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav") + self.assertEqual( + gateway.synthesize("synthetic test", "wav", 1.0), + (b"piper", "audio/wav"), + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py new file mode 100644 index 0000000..05fc929 --- /dev/null +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -0,0 +1,356 @@ +#!/usr/bin/env python3 +"""Private XTTS-first TTS gateway with a Piper fallback. + +The gateway implements the narrow /status and /tts protocol already consumed +by the profile router. Request text is never logged or persisted. +""" + +from __future__ import annotations + +import io +import json +import os +import re +import subprocess +import threading +import time +import urllib.error +import urllib.request +import wave +from http import HTTPStatus +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0") +PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085")) +XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/") +PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/") +VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy") +XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele") +DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de") +MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000")) +MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536")) +MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024))) +XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120")) +PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120")) +QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15")) +SILENCE_MS = int(os.getenv("XTTS_SEGMENT_SILENCE_MS", "20")) + +SYNTHESIS_LOCK = threading.Lock() +STATE_LOCK = threading.Lock() +SPEAKER_LOCK = threading.Lock() +SPEAKER_CONDITIONING: dict | None = None +STATE = { + "last_backend": None, + "xtts_failures": 0, + "piper_fallbacks": 0, + "last_error": None, +} + +# Prefer full compounds to isolated terms. This keeps switches infrequent and +# avoids making mixed-language speech sound like a sequence of separate clips. +ENGLISH_TERMS = ( + "Home Assistant", "Open WebUI", "OpenWebUI", "Unraid Dashboard", + "Unraid-Dashboard", "Docker Container", "Docker-Container", + "Server Log", "Server-Log", "GitHub Repository", "GitHub Repo", + "WireGuard Tunnel", "Cron Job", "Cronjob", "Home Server", + "API Key", "Tool Calling", "Context Window", "Prompt Injection", + "Unraid", "Docker", "Container", "Dashboard", "Server", "Log", + "OpenAI", "GitHub", "WireGuard", "Linux", "Debian", "Frontend", + "Backend", "Router", "Browser", "Web", "Token", "Prompt", "Context", + "Model", "Image", "Tool", "Workflow", "Benchmark", "Streaming", + "SSH", "MCP", "API", "CPU", "GPU", "VRAM", "RAM", "HTTP", "HTTPS", +) +TERM_PATTERN = re.compile( + r"(? tuple[bytes, str]: + data = None + headers = {} + method = "GET" + if payload is not None: + data = json.dumps(payload, separators=(",", ":")).encode() + headers["Content-Type"] = "application/json" + method = "POST" + request = urllib.request.Request( + url, data=data, headers=headers, method=method) + with urllib.request.urlopen(request, timeout=timeout) as response: + body = response.read(MAX_AUDIO_BYTES + 1) + if len(body) > MAX_AUDIO_BYTES: + raise RuntimeError("upstream audio response is too large") + return body, response.headers.get_content_type() + + +def _json(url: str, timeout: float = 10) -> dict | list: + body, _ = _request(url, timeout=timeout) + return json.loads(body) + + +def _reachable(url: str, path: str, timeout: float = 2) -> bool: + try: + _request(f"{url}{path}", timeout=timeout) + return True + except (OSError, ValueError, RuntimeError, urllib.error.URLError): + return False + + +def _looks_english(text: str) -> bool: + words = re.findall(r"[A-Za-zÀ-ÿ]+", text.lower()) + if not words: + return False + german = sum(word in GERMAN_MARKERS for word in words) + english = sum(word in ENGLISH_MARKERS for word in words) + return english >= 2 and english > german * 1.5 and not re.search(r"[äöüß]", text.lower()) + + +def segment_languages(text: str) -> list[tuple[str, str]]: + """Return a compact German/English segment sequence.""" + if _looks_english(text): + return [("en", text)] + if DEFAULT_LANGUAGE != "de": + return [(DEFAULT_LANGUAGE, text)] + + segments: list[tuple[str, str]] = [] + cursor = 0 + for match in TERM_PATTERN.finditer(text): + if match.start() > cursor: + segments.append(("de", text[cursor:match.start()])) + segments.append(("en", match.group(0))) + cursor = match.end() + if cursor < len(text): + segments.append(("de", text[cursor:])) + if not segments: + return [("de", text)] + + merged: list[tuple[str, str]] = [] + for language, part in segments: + if not part: + continue + if merged and merged[-1][0] == language: + previous_language, previous_text = merged[-1] + merged[-1] = (previous_language, previous_text + part) + else: + merged.append((language, part)) + return merged + + +def _speaker_conditioning() -> dict: + global SPEAKER_CONDITIONING + with SPEAKER_LOCK: + if SPEAKER_CONDITIONING is not None: + return SPEAKER_CONDITIONING + speakers = _json(f"{XTTS_URL}/studio_speakers", XTTS_TIMEOUT) + if not isinstance(speakers, dict) or XTTS_SPEAKER not in speakers: + raise RuntimeError("configured XTTS speaker is unavailable") + selected = speakers[XTTS_SPEAKER] + SPEAKER_CONDITIONING = { + "speaker_embedding": selected["speaker_embedding"], + "gpt_cond_latent": selected["gpt_cond_latent"], + } + return SPEAKER_CONDITIONING + + +def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes: + payload = { + **conditioning, + "text": text, + "language": language, + "add_wav_header": True, + "stream_chunk_size": "20", + } + audio, _ = _request(f"{XTTS_URL}/tts_stream", payload=payload, + timeout=XTTS_TIMEOUT) + if len(audio) < 44 or audio[:4] != b"RIFF" or audio[8:12] != b"WAVE": + raise RuntimeError("XTTS returned invalid WAV data") + return audio[44:] + + +def _wav(pcm: bytes) -> bytes: + output = io.BytesIO() + with wave.open(output, "wb") as wav_file: + wav_file.setnchannels(1) + wav_file.setsampwidth(2) + wav_file.setframerate(24000) + wav_file.writeframes(pcm) + return output.getvalue() + + +def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]: + if output_format == "wav" and speed == 1.0: + return wav_bytes, "audio/wav" + codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \ + if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"] + command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav", + "-i", "pipe:0"] + if speed != 1.0: + command.extend(["-filter:a", f"atempo={speed:.4f}"]) + command.extend([*codec, "pipe:1"]) + result = subprocess.run( + command, input=wav_bytes, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + check=False, timeout=120) + if result.returncode != 0 or not result.stdout: + raise RuntimeError("audio conversion failed") + return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav" + + +def synthesize_xtts(text: str, output_format: str, + speed: float) -> tuple[bytes, str]: + conditioning = _speaker_conditioning() + pcm_parts: list[bytes] = [] + silence = b"\x00\x00" * int(24000 * max(0, SILENCE_MS) / 1000) + for language, segment in segment_languages(text): + if not segment.strip(): + continue + pcm_parts.append(_xtts_pcm(segment, language, conditioning)) + if silence: + pcm_parts.append(silence) + if pcm_parts and silence: + pcm_parts.pop() + if not pcm_parts: + raise RuntimeError("no speech segments generated") + return _convert(_wav(b"".join(pcm_parts)), output_format, speed) + + +def synthesize_piper(text: str, output_format: str, + speed: float) -> tuple[bytes, str]: + return _request( + f"{PIPER_URL}/tts", + payload={"text": text, "voice": "alloy", "speed": speed, + "format": output_format}, + timeout=PIPER_TIMEOUT, + ) + + +def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]: + acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT) + if acquired: + try: + audio = synthesize_xtts(text, output_format, speed) + with STATE_LOCK: + STATE["last_backend"] = "xtts-v2" + STATE["last_error"] = None + return audio + except Exception as exc: # fallback must cover all XTTS failures + with STATE_LOCK: + STATE["xtts_failures"] += 1 + STATE["last_error"] = type(exc).__name__ + finally: + SYNTHESIS_LOCK.release() + else: + with STATE_LOCK: + STATE["xtts_failures"] += 1 + STATE["last_error"] = "queue-timeout" + + audio = synthesize_piper(text, output_format, speed) + with STATE_LOCK: + STATE["last_backend"] = "piper" + STATE["piper_fallbacks"] += 1 + return audio + + +class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, fmt: str, *args: object) -> None: + # Never log request URLs, bodies, synthesized text or speaker vectors. + print(f"tts-gateway: {self.command} -> {args[1] if len(args) > 1 else '-'}") + + def send_bytes(self, status: int, body: bytes, content_type: str) -> None: + self.send_response(status) + self.send_header("Content-Type", content_type) + self.send_header("Content-Length", str(len(body))) + self.send_header("Cache-Control", "no-store") + self.end_headers() + self.wfile.write(body) + + def send_json(self, status: int, payload: dict) -> None: + self.send_bytes(status, json.dumps(payload, separators=(",", ":")).encode(), + "application/json") + + def do_GET(self) -> None: # noqa: N802 + if self.path != "/status": + self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"}) + return + primary_ready = _reachable(XTTS_URL, "/languages") + fallback_ready = _reachable(PIPER_URL, "/status") + with STATE_LOCK: + state = dict(STATE) + self.send_json( + HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE, + { + "ready": fallback_ready, + "engine": "xtts-v2-with-piper-fallback", + "model": "xtts-v2", + "voices": [VOICE_ALIAS], + "speaker": XTTS_SPEAKER, + "primary_ready": primary_ready, + "fallback_ready": fallback_ready, + **state, + }, + ) + + def do_POST(self) -> None: # noqa: N802 + if self.path != "/tts": + self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"}) + return + try: + length = int(self.headers.get("Content-Length", "0")) + except ValueError: + length = 0 + if length <= 0 or length > MAX_REQUEST_BYTES: + self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, + {"error": "invalid request size"}) + return + try: + request = json.loads(self.rfile.read(length)) + text = request.get("text", "") + voice = request.get("voice", VOICE_ALIAS) + output_format = request.get("format", "mp3") + speed = float(request.get("speed", 1.0)) + except (json.JSONDecodeError, TypeError, ValueError): + self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"}) + return + if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS: + self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"}) + return + if voice != VOICE_ALIAS: + self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"}) + return + if output_format not in {"wav", "mp3"}: + self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"}) + return + if not 0.5 <= speed <= 2.0: + self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"}) + return + started = time.monotonic() + try: + audio, content_type = synthesize(text.strip(), output_format, speed) + except Exception as exc: + with STATE_LOCK: + STATE["last_error"] = type(exc).__name__ + self.send_json(HTTPStatus.SERVICE_UNAVAILABLE, + {"error": "all local speech backends failed"}) + return + print(f"tts-gateway: synthesized via {STATE['last_backend']} in " + f"{time.monotonic() - started:.2f}s") + self.send_bytes(HTTPStatus.OK, audio, content_type) + + +if __name__ == "__main__": + print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper") + ThreadingHTTPServer((HOST, PORT), Handler).serve_forever() diff --git a/platform/migration/restore-reference-backup.sh b/platform/migration/restore-reference-backup.sh index f9b73a5..580c400 100755 --- a/platform/migration/restore-reference-backup.sh +++ b/platform/migration/restore-reference-backup.sh @@ -172,10 +172,11 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren" log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen" cd "$STACK_DIR" docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \ - router open-webui + xtts piper tts-gateway router open-webui deadline=$((SECONDS + 180)) -for container in mike-ai-router mike-ai-open-webui; do +for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \ + mike-ai-router mike-ai-open-webui; do until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \ "$container" 2>/dev/null || true) == healthy ]]; do (( SECONDS < deadline )) || die "$container wurde nicht rechtzeitig gesund." diff --git a/platform/scripts/rollback-tts-production.sh b/platform/scripts/rollback-tts-production.sh new file mode 100755 index 0000000..77c0145 --- /dev/null +++ b/platform/scripts/rollback-tts-production.sh @@ -0,0 +1,20 @@ +#!/bin/sh +set -eu + +backup_dir=${1:?usage: rollback-tts-production.sh BACKUP_DIR} + +case "$backup_dir" in + /data/backups/mike-ai/*) ;; + *) echo "refusing unexpected backup path" >&2; exit 2 ;; +esac + +test -f "$backup_dir/compose.yaml" +test -f "$backup_dir/stack.env" + +cp -a "$backup_dir/compose.yaml" /opt/mike-ai/stack/compose.yaml +cp -a "$backup_dir/stack.env" /etc/mike-ai/stack.env +chmod 0600 /etc/mike-ai/stack.env + +cd /opt/mike-ai/stack +docker compose --env-file /etc/mike-ai/stack.env up -d --no-deps --force-recreate router +docker rm -f mike-ai-tts-gateway mike-ai-xtts >/dev/null 2>&1 || true