Replace XTTS with Qwen3-TTS
This commit is contained in:
+4
-3
@@ -7,9 +7,10 @@ FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B
|
||||
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
||||
PIPER_TTS_VERSION=1.6.0
|
||||
PIPER_VOICE=de_DE-thorsten-high
|
||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
||||
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
|
||||
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
|
||||
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
|
||||
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||
AI_DNS=192.168.1.1
|
||||
DEFAULT_REASONING_EFFORT=off
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ Sie betreibt:
|
||||
- llama.cpp mit genau einem aktiven Qwen-Profil,
|
||||
- den OpenAI-kompatiblen Profile Router,
|
||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung,
|
||||
- XTTS und Piper für Sprache,
|
||||
- Qwen3-TTS und Piper für Sprache,
|
||||
- das Athena-Dashboard,
|
||||
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
|
||||
- WireGuard-Gateway und Datenbackup,
|
||||
@@ -58,7 +58,7 @@ Qwen-Profil wird vom Profile Controller verwaltet.
|
||||
- Uncensored: separates lokales Profil
|
||||
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach
|
||||
automatisch wiederhergestellt
|
||||
- XTTS: RTX 3060; Piper bleibt CPU-Fallback
|
||||
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback
|
||||
|
||||
Die verbindlichen Werte stehen in `config/profile-matrix.json` und
|
||||
`docs/STANDARD_PROFILE_MATRIX.md`.
|
||||
|
||||
@@ -11,7 +11,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
||||
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
|
||||
- Profile Router auf Port 8081
|
||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
||||
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
|
||||
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
||||
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
||||
|
||||
+20
-22
@@ -722,9 +722,9 @@ services:
|
||||
retries: 30
|
||||
start_period: 120s
|
||||
|
||||
xtts:
|
||||
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
||||
container_name: mike-ai-xtts
|
||||
qwen3-tts:
|
||||
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
||||
container_name: mike-ai-qwen3-tts
|
||||
restart: unless-stopped
|
||||
deploy:
|
||||
resources:
|
||||
@@ -732,30 +732,33 @@ services:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids:
|
||||
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
- ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
capabilities: [gpu]
|
||||
read_only: true
|
||||
shm_size: 1g
|
||||
tmpfs:
|
||||
- /tmp:size=1g,mode=1777
|
||||
- /root/.cache:size=2g,mode=0700
|
||||
volumes:
|
||||
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
|
||||
- "${QWEN3_TTS_CACHE_DIR:-/data/models/qwen3-tts-cache}:/root/.cache/huggingface"
|
||||
- "${QWEN3_TTS_VOICES_DIR:-/data/models/qwen3-tts-voices}:/data/voices"
|
||||
environment:
|
||||
COQUI_TOS_AGREED: "1"
|
||||
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
NVIDIA_VISIBLE_DEVICES: ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
CUDA_VISIBLE_DEVICES: "0"
|
||||
NUM_THREADS: "4"
|
||||
HF_HOME: /root/.cache/huggingface
|
||||
NUMBA_CACHE_DIR: /tmp/numba
|
||||
QWEN3_TTS_MODEL_ID: Qwen/Qwen3-TTS-12Hz-1.7B-Base
|
||||
QWEN3_TTS_DEFAULT_VOICE: serena
|
||||
QWEN3_TTS_VOICES_DIR: /data/voices
|
||||
networks: [frontend]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
cap_drop: [ALL]
|
||||
healthcheck:
|
||||
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
|
||||
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8001/health', timeout=2)"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 36
|
||||
start_period: 240s
|
||||
retries: 60
|
||||
start_period: 600s
|
||||
|
||||
tts-gateway:
|
||||
build:
|
||||
@@ -769,22 +772,17 @@ services:
|
||||
environment:
|
||||
TTS_GATEWAY_HOST: 0.0.0.0
|
||||
TTS_GATEWAY_PORT: "8085"
|
||||
XTTS_URL: http://xtts:80
|
||||
QWEN_TTS_URL: http://qwen3-tts:8001
|
||||
QWEN_TTS_MODEL: tts-1
|
||||
QWEN_TTS_VOICE: serena
|
||||
QWEN_TTS_LANGUAGE: German
|
||||
QWEN_TTS_TIMEOUT: "120"
|
||||
PIPER_URL: http://piper:8085
|
||||
TTS_VOICE_ALIAS: alloy
|
||||
XTTS_SPEAKER: Annmarie Nele
|
||||
TTS_DEFAULT_LANGUAGE: de
|
||||
# Mixed-language clip stitching caused long pauses and unintelligible
|
||||
# transitions. Keep full sentences in one stable German voice.
|
||||
TTS_CODE_SWITCH_ENABLED: "false"
|
||||
XTTS_QUEUE_TIMEOUT: "15"
|
||||
XTTS_TIMEOUT: "120"
|
||||
# Keep normal sentences intact for natural prosody. This is only the
|
||||
# safety ceiling for unusually long sentences.
|
||||
XTTS_CHUNK_CHARS: "420"
|
||||
# XTTS occasionally inserts multi-second silence inside a phrase.
|
||||
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
|
||||
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
|
||||
PIPER_TIMEOUT: "120"
|
||||
networks: [frontend]
|
||||
depends_on:
|
||||
|
||||
@@ -113,7 +113,8 @@ ULTRA_PARALLEL_SLOTS=1
|
||||
UNCENSORED_PARALLEL_SLOTS=1
|
||||
PIPER_TTS_VERSION=1.6.0
|
||||
PIPER_VOICE=de_DE-thorsten-high
|
||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
||||
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
|
||||
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
|
||||
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
|
||||
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
|
||||
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||
|
||||
+9
-6
@@ -286,8 +286,10 @@ setup_wireguard() {
|
||||
|
||||
install_stack_files() {
|
||||
log "Stackdateien installieren"
|
||||
XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache}
|
||||
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups"
|
||||
QWEN3_TTS_CACHE_DIR=${QWEN3_TTS_CACHE_DIR:-$MODEL_DIR/qwen3-tts-cache}
|
||||
QWEN3_TTS_VOICES_DIR=${QWEN3_TTS_VOICES_DIR:-$MODEL_DIR/qwen3-tts-voices}
|
||||
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$QWEN3_TTS_CACHE_DIR" \
|
||||
"$QWEN3_TTS_VOICES_DIR" "$STATE_DIR/backups"
|
||||
# /opt/mike-ai/stack is both the live stack and the one canonical Git
|
||||
# checkout. Copying everything except .git created two competing source
|
||||
# trees and made agents reconstruct deployment state on every change.
|
||||
@@ -314,9 +316,10 @@ ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
|
||||
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
|
||||
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
||||
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||
XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
||||
XTTS_CACHE_DIR=$XTTS_CACHE_DIR
|
||||
XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
||||
QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR
|
||||
QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR
|
||||
QWEN3_TTS_GPU_DEVICE=${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
AI_DNS=${WG_DNS:-1.1.1.1}
|
||||
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
||||
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
||||
@@ -501,7 +504,7 @@ build_and_start() {
|
||||
docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||
wireguard-gateway xtts piper tts-gateway profile-controller router llama-dashboard portainer backup
|
||||
wireguard-gateway qwen3-tts piper tts-gateway profile-controller router llama-dashboard portainer backup
|
||||
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
||||
systemctl restart mike-ai-container-vpn-guard.service
|
||||
fi
|
||||
|
||||
@@ -110,6 +110,22 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertIn("29 Kilometer pro Stunde", spoken)
|
||||
self.assertNotIn("km", spoken.lower())
|
||||
|
||||
def test_qwen_normalizes_ipv4_time_date_and_count(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"full_kiosk (192.168.1.5): 2× Timeout um 02:14 seit 01.09."
|
||||
)
|
||||
self.assertIn("full kiosk", spoken)
|
||||
self.assertIn("192 Punkt 168 Punkt 1 Punkt 5", spoken)
|
||||
self.assertIn("2 mal Timeout", spoken)
|
||||
self.assertIn("2 Uhr 14", spoken)
|
||||
self.assertIn("1. September", spoken)
|
||||
|
||||
def test_qwen_keeps_prosody_punctuation(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"Ist das gut? Ja! SarahTV: erreichbar."
|
||||
)
|
||||
self.assertEqual(spoken, "Ist das gut? Ja! SarahTV: erreichbar.")
|
||||
|
||||
def test_visual_punctuation_becomes_natural_pauses(self):
|
||||
spoken = gateway.clean_for_speech(
|
||||
"Status: stabil – keine Fehler; Docker-Container laufen."
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private XTTS-first TTS gateway with a Piper fallback.
|
||||
"""Private Qwen3-TTS-first gateway with a Piper fallback.
|
||||
|
||||
The gateway implements the narrow /status and /tts protocol already consumed
|
||||
by the profile router. Request text is never logged or persisted.
|
||||
@@ -25,6 +25,11 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
|
||||
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
|
||||
QWEN_TTS_URL = os.getenv("QWEN_TTS_URL", "http://qwen3-tts:8001").rstrip("/")
|
||||
QWEN_TTS_MODEL = os.getenv("QWEN_TTS_MODEL", "tts-1")
|
||||
QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
|
||||
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
|
||||
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
|
||||
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||
@@ -223,8 +228,25 @@ def _spell_digits(value: str) -> str:
|
||||
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
||||
|
||||
|
||||
def _spoken_ipv4(match: re.Match) -> str:
|
||||
"""Keep IPv4 octets intact while making the separators pronounceable."""
|
||||
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
|
||||
|
||||
|
||||
def normalize_for_german_speech(text: str) -> str:
|
||||
"""Turn common visual notation into unambiguous spoken German."""
|
||||
# Run these before the date rule: otherwise 192.168.1.5 could be partly
|
||||
# interpreted as a visual date.
|
||||
text = re.sub(
|
||||
r"\b(?:\d{1,3}\.){3}\d{1,3}\b",
|
||||
_spoken_ipv4,
|
||||
text,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
|
||||
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
|
||||
text,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||
@@ -281,9 +303,31 @@ def normalize_for_german_speech(text: str) -> str:
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
|
||||
text = re.sub(r"\b(\d+)\s*[×x]\s*", r"\1 mal ", text)
|
||||
return text
|
||||
|
||||
|
||||
def prepare_for_qwen_speech(text: str) -> str:
|
||||
"""Normalize technical display text without destroying Qwen's prosody."""
|
||||
text = re.sub(r"```.*?```", " Codeblock. ", text, flags=re.DOTALL)
|
||||
text = re.sub(r"`([^`]+)`", r"\1", text)
|
||||
text = re.sub(r"!\[([^]]*)\]\([^)]+\)", r"\1", text)
|
||||
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
|
||||
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
|
||||
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||
text = text.replace("_", " ").replace("/", ", ")
|
||||
text = text.replace("→", ". ").replace("←", ". ")
|
||||
if DEFAULT_LANGUAGE == "de":
|
||||
text = normalize_for_german_speech(text)
|
||||
text = "".join(
|
||||
char for char in text
|
||||
if unicodedata.category(char) not in {"So", "Cs"}
|
||||
)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\s*\n+\s*", ". ", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def clean_for_speech(text: str) -> str:
|
||||
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
||||
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
||||
@@ -553,8 +597,17 @@ def _wav(pcm: bytes) -> bytes:
|
||||
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||
if output_format == "wav" and speed == 1.0:
|
||||
return wav_bytes, "audio/wav"
|
||||
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \
|
||||
if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"]
|
||||
if output_format == "mp3":
|
||||
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"]
|
||||
content_type = "audio/mpeg"
|
||||
elif output_format == "pcm":
|
||||
# Hermes' OpenAI streaming TTS client expects headerless 24 kHz,
|
||||
# mono, signed 16-bit little-endian PCM chunks.
|
||||
codec = ["-ac", "1", "-ar", "24000", "-codec:a", "pcm_s16le", "-f", "s16le"]
|
||||
content_type = "application/octet-stream"
|
||||
else:
|
||||
codec = ["-codec:a", "pcm_s16le", "-f", "wav"]
|
||||
content_type = "audio/wav"
|
||||
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
|
||||
"-i", "pipe:0"]
|
||||
if speed != 1.0:
|
||||
@@ -565,7 +618,7 @@ def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes,
|
||||
check=False, timeout=120)
|
||||
if result.returncode != 0 or not result.stdout:
|
||||
raise RuntimeError("audio conversion failed")
|
||||
return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav"
|
||||
return result.stdout, content_type
|
||||
|
||||
|
||||
def synthesize_xtts(text: str, output_format: str,
|
||||
@@ -583,24 +636,40 @@ def synthesize_xtts(text: str, output_format: str,
|
||||
|
||||
def synthesize_piper(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
return _request(
|
||||
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||
audio, content_type = _request(
|
||||
f"{PIPER_URL}/tts",
|
||||
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||
"format": output_format},
|
||||
"format": upstream_format},
|
||||
timeout=PIPER_TIMEOUT,
|
||||
)
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def synthesize_qwen(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
text = prepare_for_qwen_speech(text)
|
||||
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||
audio, content_type = _request(
|
||||
f"{QWEN_TTS_URL}/v1/audio/speech",
|
||||
payload={"model": QWEN_TTS_MODEL, "input": text,
|
||||
"voice": QWEN_TTS_VOICE, "language": QWEN_TTS_LANGUAGE,
|
||||
"response_format": upstream_format, "speed": speed},
|
||||
timeout=QWEN_TTS_TIMEOUT,
|
||||
)
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
||||
if acquired:
|
||||
try:
|
||||
audio = synthesize_xtts(text, output_format, speed)
|
||||
audio = synthesize_qwen(text, output_format, speed)
|
||||
with STATE_LOCK:
|
||||
STATE["last_backend"] = "xtts-v2"
|
||||
STATE["last_backend"] = "qwen3-tts-1.7b"
|
||||
STATE["last_error"] = None
|
||||
return audio
|
||||
except Exception as exc: # fallback must cover all XTTS failures
|
||||
except Exception as exc: # fallback must cover all Qwen failures
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
@@ -641,7 +710,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
if self.path != "/status":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
primary_ready = _reachable(XTTS_URL, "/languages")
|
||||
primary_ready = _reachable(QWEN_TTS_URL, "/health")
|
||||
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||
with STATE_LOCK:
|
||||
state = dict(STATE)
|
||||
@@ -649,10 +718,10 @@ class Handler(BaseHTTPRequestHandler):
|
||||
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{
|
||||
"ready": fallback_ready,
|
||||
"engine": "xtts-v2-with-piper-fallback",
|
||||
"model": "xtts-v2",
|
||||
"engine": "qwen3-tts-with-piper-fallback",
|
||||
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"voices": [VOICE_ALIAS],
|
||||
"speaker": XTTS_SPEAKER,
|
||||
"speaker": QWEN_TTS_VOICE,
|
||||
"primary_ready": primary_ready,
|
||||
"fallback_ready": fallback_ready,
|
||||
**state,
|
||||
@@ -686,7 +755,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
if voice != VOICE_ALIAS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||
return
|
||||
if output_format not in {"wav", "mp3"}:
|
||||
if output_format not in {"wav", "mp3", "pcm"}:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||
return
|
||||
if not 0.5 <= speed <= 2.0:
|
||||
@@ -707,5 +776,5 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper")
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
|
||||
@@ -77,7 +77,6 @@ start_proxy 22 172.30.10.1:22
|
||||
start_proxy 8081 router:8081
|
||||
start_proxy 8085 tts-gateway:8085
|
||||
start_proxy 8091 piper:8085
|
||||
start_proxy 8092 xtts:80
|
||||
start_proxy 8202 mcp-athena-operator:8000
|
||||
|
||||
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
|
||||
|
||||
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
|
||||
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
||||
cd "$STACK_DIR"
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
||||
xtts piper tts-gateway router open-webui
|
||||
qwen3-tts piper tts-gateway router open-webui
|
||||
|
||||
deadline=$((SECONDS + 180))
|
||||
for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \
|
||||
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
|
||||
mike-ai-router mike-ai-open-webui; do
|
||||
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||
"$container" 2>/dev/null || true) == healthy ]]; do
|
||||
|
||||
@@ -177,7 +177,7 @@ TTS_VOICES = tuple(v.strip() for v in os.environ.get(
|
||||
"TTS_VOICES", "claribel").split(",") if v.strip())
|
||||
TTS_DEFAULT_VOICE = os.environ.get(
|
||||
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
|
||||
TTS_FORMATS = ("mp3", "wav")
|
||||
TTS_FORMATS = ("mp3", "wav", "pcm")
|
||||
TTS_DEFAULT_FORMAT = "mp3"
|
||||
|
||||
# --- Spracherkennung (whisper.cpp, deutsch, CPU-only) ---
|
||||
|
||||
+1
-1
@@ -28,7 +28,7 @@ for name in \
|
||||
mike-ai-profile-controller \
|
||||
mike-ai-router \
|
||||
mike-ai-piper \
|
||||
mike-ai-xtts \
|
||||
mike-ai-qwen3-tts \
|
||||
mike-ai-tts-gateway \
|
||||
mike-ai-llama-dashboard \
|
||||
mike-ai-mcp-athena-operator \
|
||||
|
||||
Reference in New Issue
Block a user