Replace XTTS with Qwen3-TTS
This commit is contained in:
+4
-3
@@ -7,9 +7,10 @@ FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B
|
|||||||
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
|
||||||
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
|
||||||
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
|
||||||
|
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
AI_DNS=192.168.1.1
|
AI_DNS=192.168.1.1
|
||||||
DEFAULT_REASONING_EFFORT=off
|
DEFAULT_REASONING_EFFORT=off
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ Sie betreibt:
|
|||||||
- llama.cpp mit genau einem aktiven Qwen-Profil,
|
- llama.cpp mit genau einem aktiven Qwen-Profil,
|
||||||
- den OpenAI-kompatiblen Profile Router,
|
- den OpenAI-kompatiblen Profile Router,
|
||||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung,
|
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung,
|
||||||
- XTTS und Piper für Sprache,
|
- Qwen3-TTS und Piper für Sprache,
|
||||||
- das Athena-Dashboard,
|
- das Athena-Dashboard,
|
||||||
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
|
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
|
||||||
- WireGuard-Gateway und Datenbackup,
|
- WireGuard-Gateway und Datenbackup,
|
||||||
@@ -58,7 +58,7 @@ Qwen-Profil wird vom Profile Controller verwaltet.
|
|||||||
- Uncensored: separates lokales Profil
|
- Uncensored: separates lokales Profil
|
||||||
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach
|
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach
|
||||||
automatisch wiederhergestellt
|
automatisch wiederhergestellt
|
||||||
- XTTS: RTX 3060; Piper bleibt CPU-Fallback
|
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback
|
||||||
|
|
||||||
Die verbindlichen Werte stehen in `config/profile-matrix.json` und
|
Die verbindlichen Werte stehen in `config/profile-matrix.json` und
|
||||||
`docs/STANDARD_PROFILE_MATRIX.md`.
|
`docs/STANDARD_PROFILE_MATRIX.md`.
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
|||||||
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
|
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
|
||||||
- Profile Router auf Port 8081
|
- Profile Router auf Port 8081
|
||||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
||||||
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
|
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
||||||
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||||
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
||||||
|
|||||||
+20
-22
@@ -722,9 +722,9 @@ services:
|
|||||||
retries: 30
|
retries: 30
|
||||||
start_period: 120s
|
start_period: 120s
|
||||||
|
|
||||||
xtts:
|
qwen3-tts:
|
||||||
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
||||||
container_name: mike-ai-xtts
|
container_name: mike-ai-qwen3-tts
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
deploy:
|
deploy:
|
||||||
resources:
|
resources:
|
||||||
@@ -732,30 +732,33 @@ services:
|
|||||||
devices:
|
devices:
|
||||||
- driver: nvidia
|
- driver: nvidia
|
||||||
device_ids:
|
device_ids:
|
||||||
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
- ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
capabilities: [gpu]
|
capabilities: [gpu]
|
||||||
read_only: true
|
read_only: true
|
||||||
shm_size: 1g
|
shm_size: 1g
|
||||||
tmpfs:
|
tmpfs:
|
||||||
- /tmp:size=1g,mode=1777
|
- /tmp:size=1g,mode=1777
|
||||||
- /root/.cache:size=2g,mode=0700
|
|
||||||
volumes:
|
volumes:
|
||||||
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
|
- "${QWEN3_TTS_CACHE_DIR:-/data/models/qwen3-tts-cache}:/root/.cache/huggingface"
|
||||||
|
- "${QWEN3_TTS_VOICES_DIR:-/data/models/qwen3-tts-voices}:/data/voices"
|
||||||
environment:
|
environment:
|
||||||
COQUI_TOS_AGREED: "1"
|
NVIDIA_VISIBLE_DEVICES: ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
CUDA_VISIBLE_DEVICES: "0"
|
CUDA_VISIBLE_DEVICES: "0"
|
||||||
NUM_THREADS: "4"
|
HF_HOME: /root/.cache/huggingface
|
||||||
|
NUMBA_CACHE_DIR: /tmp/numba
|
||||||
|
QWEN3_TTS_MODEL_ID: Qwen/Qwen3-TTS-12Hz-1.7B-Base
|
||||||
|
QWEN3_TTS_DEFAULT_VOICE: serena
|
||||||
|
QWEN3_TTS_VOICES_DIR: /data/voices
|
||||||
networks: [frontend]
|
networks: [frontend]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
cap_drop: [ALL]
|
cap_drop: [ALL]
|
||||||
healthcheck:
|
healthcheck:
|
||||||
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
|
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8001/health', timeout=2)"]
|
||||||
interval: 10s
|
interval: 10s
|
||||||
timeout: 5s
|
timeout: 5s
|
||||||
retries: 36
|
retries: 60
|
||||||
start_period: 240s
|
start_period: 600s
|
||||||
|
|
||||||
tts-gateway:
|
tts-gateway:
|
||||||
build:
|
build:
|
||||||
@@ -769,22 +772,17 @@ services:
|
|||||||
environment:
|
environment:
|
||||||
TTS_GATEWAY_HOST: 0.0.0.0
|
TTS_GATEWAY_HOST: 0.0.0.0
|
||||||
TTS_GATEWAY_PORT: "8085"
|
TTS_GATEWAY_PORT: "8085"
|
||||||
XTTS_URL: http://xtts:80
|
QWEN_TTS_URL: http://qwen3-tts:8001
|
||||||
|
QWEN_TTS_MODEL: tts-1
|
||||||
|
QWEN_TTS_VOICE: serena
|
||||||
|
QWEN_TTS_LANGUAGE: German
|
||||||
|
QWEN_TTS_TIMEOUT: "120"
|
||||||
PIPER_URL: http://piper:8085
|
PIPER_URL: http://piper:8085
|
||||||
TTS_VOICE_ALIAS: alloy
|
TTS_VOICE_ALIAS: alloy
|
||||||
XTTS_SPEAKER: Annmarie Nele
|
|
||||||
TTS_DEFAULT_LANGUAGE: de
|
TTS_DEFAULT_LANGUAGE: de
|
||||||
# Mixed-language clip stitching caused long pauses and unintelligible
|
# Mixed-language clip stitching caused long pauses and unintelligible
|
||||||
# transitions. Keep full sentences in one stable German voice.
|
# transitions. Keep full sentences in one stable German voice.
|
||||||
TTS_CODE_SWITCH_ENABLED: "false"
|
TTS_CODE_SWITCH_ENABLED: "false"
|
||||||
XTTS_QUEUE_TIMEOUT: "15"
|
|
||||||
XTTS_TIMEOUT: "120"
|
|
||||||
# Keep normal sentences intact for natural prosody. This is only the
|
|
||||||
# safety ceiling for unusually long sentences.
|
|
||||||
XTTS_CHUNK_CHARS: "420"
|
|
||||||
# XTTS occasionally inserts multi-second silence inside a phrase.
|
|
||||||
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
|
|
||||||
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
|
|
||||||
PIPER_TIMEOUT: "120"
|
PIPER_TIMEOUT: "120"
|
||||||
networks: [frontend]
|
networks: [frontend]
|
||||||
depends_on:
|
depends_on:
|
||||||
|
|||||||
@@ -113,7 +113,8 @@ ULTRA_PARALLEL_SLOTS=1
|
|||||||
UNCENSORED_PARALLEL_SLOTS=1
|
UNCENSORED_PARALLEL_SLOTS=1
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
|
||||||
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
|
||||||
|
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
|
||||||
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
|
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
|
||||||
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
|
|||||||
+9
-6
@@ -286,8 +286,10 @@ setup_wireguard() {
|
|||||||
|
|
||||||
install_stack_files() {
|
install_stack_files() {
|
||||||
log "Stackdateien installieren"
|
log "Stackdateien installieren"
|
||||||
XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache}
|
QWEN3_TTS_CACHE_DIR=${QWEN3_TTS_CACHE_DIR:-$MODEL_DIR/qwen3-tts-cache}
|
||||||
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups"
|
QWEN3_TTS_VOICES_DIR=${QWEN3_TTS_VOICES_DIR:-$MODEL_DIR/qwen3-tts-voices}
|
||||||
|
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$QWEN3_TTS_CACHE_DIR" \
|
||||||
|
"$QWEN3_TTS_VOICES_DIR" "$STATE_DIR/backups"
|
||||||
# /opt/mike-ai/stack is both the live stack and the one canonical Git
|
# /opt/mike-ai/stack is both the live stack and the one canonical Git
|
||||||
# checkout. Copying everything except .git created two competing source
|
# checkout. Copying everything except .git created two competing source
|
||||||
# trees and made agents reconstruct deployment state on every change.
|
# trees and made agents reconstruct deployment state on every change.
|
||||||
@@ -314,9 +316,10 @@ ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
|
|||||||
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
|
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
|
||||||
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
||||||
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||||
XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
||||||
XTTS_CACHE_DIR=$XTTS_CACHE_DIR
|
QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR
|
||||||
XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR
|
||||||
|
QWEN3_TTS_GPU_DEVICE=${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||||
AI_DNS=${WG_DNS:-1.1.1.1}
|
AI_DNS=${WG_DNS:-1.1.1.1}
|
||||||
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
||||||
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
||||||
@@ -501,7 +504,7 @@ build_and_start() {
|
|||||||
docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null
|
docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer
|
docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||||
wireguard-gateway xtts piper tts-gateway profile-controller router llama-dashboard portainer backup
|
wireguard-gateway qwen3-tts piper tts-gateway profile-controller router llama-dashboard portainer backup
|
||||||
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
||||||
systemctl restart mike-ai-container-vpn-guard.service
|
systemctl restart mike-ai-container-vpn-guard.service
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -110,6 +110,22 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
self.assertIn("29 Kilometer pro Stunde", spoken)
|
self.assertIn("29 Kilometer pro Stunde", spoken)
|
||||||
self.assertNotIn("km", spoken.lower())
|
self.assertNotIn("km", spoken.lower())
|
||||||
|
|
||||||
|
def test_qwen_normalizes_ipv4_time_date_and_count(self):
|
||||||
|
spoken = gateway.prepare_for_qwen_speech(
|
||||||
|
"full_kiosk (192.168.1.5): 2× Timeout um 02:14 seit 01.09."
|
||||||
|
)
|
||||||
|
self.assertIn("full kiosk", spoken)
|
||||||
|
self.assertIn("192 Punkt 168 Punkt 1 Punkt 5", spoken)
|
||||||
|
self.assertIn("2 mal Timeout", spoken)
|
||||||
|
self.assertIn("2 Uhr 14", spoken)
|
||||||
|
self.assertIn("1. September", spoken)
|
||||||
|
|
||||||
|
def test_qwen_keeps_prosody_punctuation(self):
|
||||||
|
spoken = gateway.prepare_for_qwen_speech(
|
||||||
|
"Ist das gut? Ja! SarahTV: erreichbar."
|
||||||
|
)
|
||||||
|
self.assertEqual(spoken, "Ist das gut? Ja! SarahTV: erreichbar.")
|
||||||
|
|
||||||
def test_visual_punctuation_becomes_natural_pauses(self):
|
def test_visual_punctuation_becomes_natural_pauses(self):
|
||||||
spoken = gateway.clean_for_speech(
|
spoken = gateway.clean_for_speech(
|
||||||
"Status: stabil – keine Fehler; Docker-Container laufen."
|
"Status: stabil – keine Fehler; Docker-Container laufen."
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Private XTTS-first TTS gateway with a Piper fallback.
|
"""Private Qwen3-TTS-first gateway with a Piper fallback.
|
||||||
|
|
||||||
The gateway implements the narrow /status and /tts protocol already consumed
|
The gateway implements the narrow /status and /tts protocol already consumed
|
||||||
by the profile router. Request text is never logged or persisted.
|
by the profile router. Request text is never logged or persisted.
|
||||||
@@ -25,6 +25,11 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|||||||
|
|
||||||
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
|
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
|
||||||
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
|
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
|
||||||
|
QWEN_TTS_URL = os.getenv("QWEN_TTS_URL", "http://qwen3-tts:8001").rstrip("/")
|
||||||
|
QWEN_TTS_MODEL = os.getenv("QWEN_TTS_MODEL", "tts-1")
|
||||||
|
QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
|
||||||
|
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
|
||||||
|
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
|
||||||
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||||
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||||
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||||
@@ -223,8 +228,25 @@ def _spell_digits(value: str) -> str:
|
|||||||
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
||||||
|
|
||||||
|
|
||||||
|
def _spoken_ipv4(match: re.Match) -> str:
|
||||||
|
"""Keep IPv4 octets intact while making the separators pronounceable."""
|
||||||
|
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
|
||||||
|
|
||||||
|
|
||||||
def normalize_for_german_speech(text: str) -> str:
|
def normalize_for_german_speech(text: str) -> str:
|
||||||
"""Turn common visual notation into unambiguous spoken German."""
|
"""Turn common visual notation into unambiguous spoken German."""
|
||||||
|
# Run these before the date rule: otherwise 192.168.1.5 could be partly
|
||||||
|
# interpreted as a visual date.
|
||||||
|
text = re.sub(
|
||||||
|
r"\b(?:\d{1,3}\.){3}\d{1,3}\b",
|
||||||
|
_spoken_ipv4,
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
|
||||||
|
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
|
||||||
|
text,
|
||||||
|
)
|
||||||
text = re.sub(
|
text = re.sub(
|
||||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
||||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||||
@@ -281,9 +303,31 @@ def normalize_for_german_speech(text: str) -> str:
|
|||||||
flags=re.IGNORECASE,
|
flags=re.IGNORECASE,
|
||||||
)
|
)
|
||||||
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
|
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
|
||||||
|
text = re.sub(r"\b(\d+)\s*[×x]\s*", r"\1 mal ", text)
|
||||||
return text
|
return text
|
||||||
|
|
||||||
|
|
||||||
|
def prepare_for_qwen_speech(text: str) -> str:
|
||||||
|
"""Normalize technical display text without destroying Qwen's prosody."""
|
||||||
|
text = re.sub(r"```.*?```", " Codeblock. ", text, flags=re.DOTALL)
|
||||||
|
text = re.sub(r"`([^`]+)`", r"\1", text)
|
||||||
|
text = re.sub(r"!\[([^]]*)\]\([^)]+\)", r"\1", text)
|
||||||
|
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
|
||||||
|
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
|
||||||
|
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||||
|
text = text.replace("_", " ").replace("/", ", ")
|
||||||
|
text = text.replace("→", ". ").replace("←", ". ")
|
||||||
|
if DEFAULT_LANGUAGE == "de":
|
||||||
|
text = normalize_for_german_speech(text)
|
||||||
|
text = "".join(
|
||||||
|
char for char in text
|
||||||
|
if unicodedata.category(char) not in {"So", "Cs"}
|
||||||
|
)
|
||||||
|
text = re.sub(r"[ \t]+", " ", text)
|
||||||
|
text = re.sub(r"\s*\n+\s*", ". ", text)
|
||||||
|
return text.strip()
|
||||||
|
|
||||||
|
|
||||||
def clean_for_speech(text: str) -> str:
|
def clean_for_speech(text: str) -> str:
|
||||||
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
||||||
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
||||||
@@ -553,8 +597,17 @@ def _wav(pcm: bytes) -> bytes:
|
|||||||
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
|
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||||
if output_format == "wav" and speed == 1.0:
|
if output_format == "wav" and speed == 1.0:
|
||||||
return wav_bytes, "audio/wav"
|
return wav_bytes, "audio/wav"
|
||||||
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \
|
if output_format == "mp3":
|
||||||
if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"]
|
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"]
|
||||||
|
content_type = "audio/mpeg"
|
||||||
|
elif output_format == "pcm":
|
||||||
|
# Hermes' OpenAI streaming TTS client expects headerless 24 kHz,
|
||||||
|
# mono, signed 16-bit little-endian PCM chunks.
|
||||||
|
codec = ["-ac", "1", "-ar", "24000", "-codec:a", "pcm_s16le", "-f", "s16le"]
|
||||||
|
content_type = "application/octet-stream"
|
||||||
|
else:
|
||||||
|
codec = ["-codec:a", "pcm_s16le", "-f", "wav"]
|
||||||
|
content_type = "audio/wav"
|
||||||
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
|
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
|
||||||
"-i", "pipe:0"]
|
"-i", "pipe:0"]
|
||||||
if speed != 1.0:
|
if speed != 1.0:
|
||||||
@@ -565,7 +618,7 @@ def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes,
|
|||||||
check=False, timeout=120)
|
check=False, timeout=120)
|
||||||
if result.returncode != 0 or not result.stdout:
|
if result.returncode != 0 or not result.stdout:
|
||||||
raise RuntimeError("audio conversion failed")
|
raise RuntimeError("audio conversion failed")
|
||||||
return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav"
|
return result.stdout, content_type
|
||||||
|
|
||||||
|
|
||||||
def synthesize_xtts(text: str, output_format: str,
|
def synthesize_xtts(text: str, output_format: str,
|
||||||
@@ -583,24 +636,40 @@ def synthesize_xtts(text: str, output_format: str,
|
|||||||
|
|
||||||
def synthesize_piper(text: str, output_format: str,
|
def synthesize_piper(text: str, output_format: str,
|
||||||
speed: float) -> tuple[bytes, str]:
|
speed: float) -> tuple[bytes, str]:
|
||||||
return _request(
|
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||||
|
audio, content_type = _request(
|
||||||
f"{PIPER_URL}/tts",
|
f"{PIPER_URL}/tts",
|
||||||
payload={"text": text, "voice": "alloy", "speed": speed,
|
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||||
"format": output_format},
|
"format": upstream_format},
|
||||||
timeout=PIPER_TIMEOUT,
|
timeout=PIPER_TIMEOUT,
|
||||||
)
|
)
|
||||||
|
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||||
|
|
||||||
|
|
||||||
|
def synthesize_qwen(text: str, output_format: str,
|
||||||
|
speed: float) -> tuple[bytes, str]:
|
||||||
|
text = prepare_for_qwen_speech(text)
|
||||||
|
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||||
|
audio, content_type = _request(
|
||||||
|
f"{QWEN_TTS_URL}/v1/audio/speech",
|
||||||
|
payload={"model": QWEN_TTS_MODEL, "input": text,
|
||||||
|
"voice": QWEN_TTS_VOICE, "language": QWEN_TTS_LANGUAGE,
|
||||||
|
"response_format": upstream_format, "speed": speed},
|
||||||
|
timeout=QWEN_TTS_TIMEOUT,
|
||||||
|
)
|
||||||
|
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||||
|
|
||||||
|
|
||||||
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
|
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||||
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
||||||
if acquired:
|
if acquired:
|
||||||
try:
|
try:
|
||||||
audio = synthesize_xtts(text, output_format, speed)
|
audio = synthesize_qwen(text, output_format, speed)
|
||||||
with STATE_LOCK:
|
with STATE_LOCK:
|
||||||
STATE["last_backend"] = "xtts-v2"
|
STATE["last_backend"] = "qwen3-tts-1.7b"
|
||||||
STATE["last_error"] = None
|
STATE["last_error"] = None
|
||||||
return audio
|
return audio
|
||||||
except Exception as exc: # fallback must cover all XTTS failures
|
except Exception as exc: # fallback must cover all Qwen failures
|
||||||
with STATE_LOCK:
|
with STATE_LOCK:
|
||||||
STATE["xtts_failures"] += 1
|
STATE["xtts_failures"] += 1
|
||||||
STATE["last_error"] = type(exc).__name__
|
STATE["last_error"] = type(exc).__name__
|
||||||
@@ -641,7 +710,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
if self.path != "/status":
|
if self.path != "/status":
|
||||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||||
return
|
return
|
||||||
primary_ready = _reachable(XTTS_URL, "/languages")
|
primary_ready = _reachable(QWEN_TTS_URL, "/health")
|
||||||
fallback_ready = _reachable(PIPER_URL, "/status")
|
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||||
with STATE_LOCK:
|
with STATE_LOCK:
|
||||||
state = dict(STATE)
|
state = dict(STATE)
|
||||||
@@ -649,10 +718,10 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||||
{
|
{
|
||||||
"ready": fallback_ready,
|
"ready": fallback_ready,
|
||||||
"engine": "xtts-v2-with-piper-fallback",
|
"engine": "qwen3-tts-with-piper-fallback",
|
||||||
"model": "xtts-v2",
|
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||||
"voices": [VOICE_ALIAS],
|
"voices": [VOICE_ALIAS],
|
||||||
"speaker": XTTS_SPEAKER,
|
"speaker": QWEN_TTS_VOICE,
|
||||||
"primary_ready": primary_ready,
|
"primary_ready": primary_ready,
|
||||||
"fallback_ready": fallback_ready,
|
"fallback_ready": fallback_ready,
|
||||||
**state,
|
**state,
|
||||||
@@ -686,7 +755,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
if voice != VOICE_ALIAS:
|
if voice != VOICE_ALIAS:
|
||||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||||
return
|
return
|
||||||
if output_format not in {"wav", "mp3"}:
|
if output_format not in {"wav", "mp3", "pcm"}:
|
||||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||||
return
|
return
|
||||||
if not 0.5 <= speed <= 2.0:
|
if not 0.5 <= speed <= 2.0:
|
||||||
@@ -707,5 +776,5 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper")
|
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
|
||||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||||
|
|||||||
@@ -77,7 +77,6 @@ start_proxy 22 172.30.10.1:22
|
|||||||
start_proxy 8081 router:8081
|
start_proxy 8081 router:8081
|
||||||
start_proxy 8085 tts-gateway:8085
|
start_proxy 8085 tts-gateway:8085
|
||||||
start_proxy 8091 piper:8085
|
start_proxy 8091 piper:8085
|
||||||
start_proxy 8092 xtts:80
|
|
||||||
start_proxy 8202 mcp-athena-operator:8000
|
start_proxy 8202 mcp-athena-operator:8000
|
||||||
|
|
||||||
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
|
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
|
||||||
|
|||||||
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
|
|||||||
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
||||||
cd "$STACK_DIR"
|
cd "$STACK_DIR"
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
||||||
xtts piper tts-gateway router open-webui
|
qwen3-tts piper tts-gateway router open-webui
|
||||||
|
|
||||||
deadline=$((SECONDS + 180))
|
deadline=$((SECONDS + 180))
|
||||||
for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \
|
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
|
||||||
mike-ai-router mike-ai-open-webui; do
|
mike-ai-router mike-ai-open-webui; do
|
||||||
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||||
"$container" 2>/dev/null || true) == healthy ]]; do
|
"$container" 2>/dev/null || true) == healthy ]]; do
|
||||||
|
|||||||
@@ -177,7 +177,7 @@ TTS_VOICES = tuple(v.strip() for v in os.environ.get(
|
|||||||
"TTS_VOICES", "claribel").split(",") if v.strip())
|
"TTS_VOICES", "claribel").split(",") if v.strip())
|
||||||
TTS_DEFAULT_VOICE = os.environ.get(
|
TTS_DEFAULT_VOICE = os.environ.get(
|
||||||
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
|
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
|
||||||
TTS_FORMATS = ("mp3", "wav")
|
TTS_FORMATS = ("mp3", "wav", "pcm")
|
||||||
TTS_DEFAULT_FORMAT = "mp3"
|
TTS_DEFAULT_FORMAT = "mp3"
|
||||||
|
|
||||||
# --- Spracherkennung (whisper.cpp, deutsch, CPU-only) ---
|
# --- Spracherkennung (whisper.cpp, deutsch, CPU-only) ---
|
||||||
|
|||||||
+1
-1
@@ -28,7 +28,7 @@ for name in \
|
|||||||
mike-ai-profile-controller \
|
mike-ai-profile-controller \
|
||||||
mike-ai-router \
|
mike-ai-router \
|
||||||
mike-ai-piper \
|
mike-ai-piper \
|
||||||
mike-ai-xtts \
|
mike-ai-qwen3-tts \
|
||||||
mike-ai-tts-gateway \
|
mike-ai-tts-gateway \
|
||||||
mike-ai-llama-dashboard \
|
mike-ai-llama-dashboard \
|
||||||
mike-ai-mcp-athena-operator \
|
mike-ai-mcp-athena-operator \
|
||||||
|
|||||||
Reference in New Issue
Block a user