Replace XTTS with Qwen3-TTS

This commit is contained in:
Mikei386
2026-09-05 13:30:14 +02:00
parent 44e1c1c50e
commit cc416150a8
12 changed files with 144 additions and 57 deletions
+4 -3
View File
@@ -7,9 +7,10 @@ FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
AI_DNS=192.168.1.1
DEFAULT_REASONING_EFFORT=off
+2 -2
View File
@@ -11,7 +11,7 @@ Sie betreibt:
- llama.cpp mit genau einem aktiven Qwen-Profil,
- den OpenAI-kompatiblen Profile Router,
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung,
- XTTS und Piper für Sprache,
- Qwen3-TTS und Piper für Sprache,
- das Athena-Dashboard,
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
- WireGuard-Gateway und Datenbackup,
@@ -58,7 +58,7 @@ Qwen-Profil wird vom Profile Controller verwaltet.
- Uncensored: separates lokales Profil
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach
automatisch wiederhergestellt
- XTTS: RTX 3060; Piper bleibt CPU-Fallback
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback
Die verbindlichen Werte stehen in `config/profile-matrix.json` und
`docs/STANDARD_PROFILE_MATRIX.md`.
+1 -1
View File
@@ -11,7 +11,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
- Profile Router auf Port 8081
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
- XTTS auf der RTX 3060 mit Piper als CPU-Fallback
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
- Portainer CE als optionale Container-Ansicht auf Port 9443
+20 -22
View File
@@ -722,9 +722,9 @@ services:
retries: 30
start_period: 120s
xtts:
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
container_name: mike-ai-xtts
qwen3-tts:
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
container_name: mike-ai-qwen3-tts
restart: unless-stopped
deploy:
resources:
@@ -732,30 +732,33 @@ services:
devices:
- driver: nvidia
device_ids:
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
- ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
capabilities: [gpu]
read_only: true
shm_size: 1g
tmpfs:
- /tmp:size=1g,mode=1777
- /root/.cache:size=2g,mode=0700
volumes:
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
- "${QWEN3_TTS_CACHE_DIR:-/data/models/qwen3-tts-cache}:/root/.cache/huggingface"
- "${QWEN3_TTS_VOICES_DIR:-/data/models/qwen3-tts-voices}:/data/voices"
environment:
COQUI_TOS_AGREED: "1"
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_VISIBLE_DEVICES: ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
CUDA_VISIBLE_DEVICES: "0"
NUM_THREADS: "4"
HF_HOME: /root/.cache/huggingface
NUMBA_CACHE_DIR: /tmp/numba
QWEN3_TTS_MODEL_ID: Qwen/Qwen3-TTS-12Hz-1.7B-Base
QWEN3_TTS_DEFAULT_VOICE: serena
QWEN3_TTS_VOICES_DIR: /data/voices
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8001/health', timeout=2)"]
interval: 10s
timeout: 5s
retries: 36
start_period: 240s
retries: 60
start_period: 600s
tts-gateway:
build:
@@ -769,22 +772,17 @@ services:
environment:
TTS_GATEWAY_HOST: 0.0.0.0
TTS_GATEWAY_PORT: "8085"
XTTS_URL: http://xtts:80
QWEN_TTS_URL: http://qwen3-tts:8001
QWEN_TTS_MODEL: tts-1
QWEN_TTS_VOICE: serena
QWEN_TTS_LANGUAGE: German
QWEN_TTS_TIMEOUT: "120"
PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy
XTTS_SPEAKER: Annmarie Nele
TTS_DEFAULT_LANGUAGE: de
# Mixed-language clip stitching caused long pauses and unintelligible
# transitions. Keep full sentences in one stable German voice.
TTS_CODE_SWITCH_ENABLED: "false"
XTTS_QUEUE_TIMEOUT: "15"
XTTS_TIMEOUT: "120"
# Keep normal sentences intact for natural prosody. This is only the
# safety ceiling for unusually long sentences.
XTTS_CHUNK_CHARS: "420"
# XTTS occasionally inserts multi-second silence inside a phrase.
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
PIPER_TIMEOUT: "120"
networks: [frontend]
depends_on:
+4 -3
View File
@@ -113,7 +113,8 @@ ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1
PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
XTTS_CACHE_DIR=/data/models/xtts-v2-cache
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
# Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here.
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
+9 -6
View File
@@ -286,8 +286,10 @@ setup_wireguard() {
install_stack_files() {
log "Stackdateien installieren"
XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache}
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups"
QWEN3_TTS_CACHE_DIR=${QWEN3_TTS_CACHE_DIR:-$MODEL_DIR/qwen3-tts-cache}
QWEN3_TTS_VOICES_DIR=${QWEN3_TTS_VOICES_DIR:-$MODEL_DIR/qwen3-tts-voices}
install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$QWEN3_TTS_CACHE_DIR" \
"$QWEN3_TTS_VOICES_DIR" "$STATE_DIR/backups"
# /opt/mike-ai/stack is both the live stack and the one canonical Git
# checkout. Copying everything except .git created two competing source
# trees and made agents reconstruct deployment state on every change.
@@ -314,9 +316,10 @@ ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
XTTS_CACHE_DIR=$XTTS_CACHE_DIR
XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR
QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR
QWEN3_TTS_GPU_DEVICE=${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
AI_DNS=${WG_DNS:-1.1.1.1}
FAST_MODEL_FILE=$FAST_MODEL_FILE
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
@@ -501,7 +504,7 @@ build_and_start() {
docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null
docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
wireguard-gateway xtts piper tts-gateway profile-controller router llama-dashboard portainer backup
wireguard-gateway qwen3-tts piper tts-gateway profile-controller router llama-dashboard portainer backup
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
systemctl restart mike-ai-container-vpn-guard.service
fi
@@ -110,6 +110,22 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertIn("29 Kilometer pro Stunde", spoken)
self.assertNotIn("km", spoken.lower())
def test_qwen_normalizes_ipv4_time_date_and_count(self):
spoken = gateway.prepare_for_qwen_speech(
"full_kiosk (192.168.1.5): 2× Timeout um 02:14 seit 01.09."
)
self.assertIn("full kiosk", spoken)
self.assertIn("192 Punkt 168 Punkt 1 Punkt 5", spoken)
self.assertIn("2 mal Timeout", spoken)
self.assertIn("2 Uhr 14", spoken)
self.assertIn("1. September", spoken)
def test_qwen_keeps_prosody_punctuation(self):
spoken = gateway.prepare_for_qwen_speech(
"Ist das gut? Ja! SarahTV: erreichbar."
)
self.assertEqual(spoken, "Ist das gut? Ja! SarahTV: erreichbar.")
def test_visual_punctuation_becomes_natural_pauses(self):
spoken = gateway.clean_for_speech(
"Status: stabil – keine Fehler; Docker-Container laufen."
+84 -15
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
"""Private XTTS-first TTS gateway with a Piper fallback.
"""Private Qwen3-TTS-first gateway with a Piper fallback.
The gateway implements the narrow /status and /tts protocol already consumed
by the profile router. Request text is never logged or persisted.
@@ -25,6 +25,11 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0")
PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085"))
QWEN_TTS_URL = os.getenv("QWEN_TTS_URL", "http://qwen3-tts:8001").rstrip("/")
QWEN_TTS_MODEL = os.getenv("QWEN_TTS_MODEL", "tts-1")
QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
@@ -223,8 +228,25 @@ def _spell_digits(value: str) -> str:
return " ".join(GERMAN_DIGITS[digit] for digit in value)
def _spoken_ipv4(match: re.Match) -> str:
"""Keep IPv4 octets intact while making the separators pronounceable."""
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
def normalize_for_german_speech(text: str) -> str:
"""Turn common visual notation into unambiguous spoken German."""
# Run these before the date rule: otherwise 192.168.1.5 could be partly
# interpreted as a visual date.
text = re.sub(
r"\b(?:\d{1,3}\.){3}\d{1,3}\b",
_spoken_ipv4,
text,
)
text = re.sub(
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
text,
)
text = re.sub(
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
@@ -281,9 +303,31 @@ def normalize_for_german_speech(text: str) -> str:
flags=re.IGNORECASE,
)
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
text = re.sub(r"\b(\d+)\s*[×x]\s*", r"\1 mal ", text)
return text
def prepare_for_qwen_speech(text: str) -> str:
"""Normalize technical display text without destroying Qwen's prosody."""
text = re.sub(r"```.*?```", " Codeblock. ", text, flags=re.DOTALL)
text = re.sub(r"`([^`]+)`", r"\1", text)
text = re.sub(r"!\[([^]]*)\]\([^)]+\)", r"\1", text)
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
text = text.replace("_", " ").replace("/", ", ")
text = text.replace("→", ". ").replace("←", ". ")
if DEFAULT_LANGUAGE == "de":
text = normalize_for_german_speech(text)
text = "".join(
char for char in text
if unicodedata.category(char) not in {"So", "Cs"}
)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\s*\n+\s*", ". ", text)
return text.strip()
def clean_for_speech(text: str) -> str:
"""Remove visual markup that makes long TTS output unstable or noisy."""
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
@@ -553,8 +597,17 @@ def _wav(pcm: bytes) -> bytes:
def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]:
if output_format == "wav" and speed == 1.0:
return wav_bytes, "audio/wav"
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \
if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"]
if output_format == "mp3":
codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"]
content_type = "audio/mpeg"
elif output_format == "pcm":
# Hermes' OpenAI streaming TTS client expects headerless 24 kHz,
# mono, signed 16-bit little-endian PCM chunks.
codec = ["-ac", "1", "-ar", "24000", "-codec:a", "pcm_s16le", "-f", "s16le"]
content_type = "application/octet-stream"
else:
codec = ["-codec:a", "pcm_s16le", "-f", "wav"]
content_type = "audio/wav"
command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav",
"-i", "pipe:0"]
if speed != 1.0:
@@ -565,7 +618,7 @@ def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes,
check=False, timeout=120)
if result.returncode != 0 or not result.stdout:
raise RuntimeError("audio conversion failed")
return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav"
return result.stdout, content_type
def synthesize_xtts(text: str, output_format: str,
@@ -583,24 +636,40 @@ def synthesize_xtts(text: str, output_format: str,
def synthesize_piper(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
return _request(
upstream_format = "wav" if output_format == "pcm" else output_format
audio, content_type = _request(
f"{PIPER_URL}/tts",
payload={"text": text, "voice": "alloy", "speed": speed,
"format": output_format},
"format": upstream_format},
timeout=PIPER_TIMEOUT,
)
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def synthesize_qwen(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
text = prepare_for_qwen_speech(text)
upstream_format = "wav" if output_format == "pcm" else output_format
audio, content_type = _request(
f"{QWEN_TTS_URL}/v1/audio/speech",
payload={"model": QWEN_TTS_MODEL, "input": text,
"voice": QWEN_TTS_VOICE, "language": QWEN_TTS_LANGUAGE,
"response_format": upstream_format, "speed": speed},
timeout=QWEN_TTS_TIMEOUT,
)
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if acquired:
try:
audio = synthesize_xtts(text, output_format, speed)
audio = synthesize_qwen(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "xtts-v2"
STATE["last_backend"] = "qwen3-tts-1.7b"
STATE["last_error"] = None
return audio
except Exception as exc: # fallback must cover all XTTS failures
except Exception as exc: # fallback must cover all Qwen failures
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["last_error"] = type(exc).__name__
@@ -641,7 +710,7 @@ class Handler(BaseHTTPRequestHandler):
if self.path != "/status":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
primary_ready = _reachable(XTTS_URL, "/languages")
primary_ready = _reachable(QWEN_TTS_URL, "/health")
fallback_ready = _reachable(PIPER_URL, "/status")
with STATE_LOCK:
state = dict(STATE)
@@ -649,10 +718,10 @@ class Handler(BaseHTTPRequestHandler):
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
{
"ready": fallback_ready,
"engine": "xtts-v2-with-piper-fallback",
"model": "xtts-v2",
"engine": "qwen3-tts-with-piper-fallback",
"model": "Qwen3-TTS-12Hz-1.7B-Base",
"voices": [VOICE_ALIAS],
"speaker": XTTS_SPEAKER,
"speaker": QWEN_TTS_VOICE,
"primary_ready": primary_ready,
"fallback_ready": fallback_ready,
**state,
@@ -686,7 +755,7 @@ class Handler(BaseHTTPRequestHandler):
if voice != VOICE_ALIAS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
return
if output_format not in {"wav", "mp3"}:
if output_format not in {"wav", "mp3", "pcm"}:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
return
if not 0.5 <= speed <= 2.0:
@@ -707,5 +776,5 @@ class Handler(BaseHTTPRequestHandler):
if __name__ == "__main__":
print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper")
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
@@ -77,7 +77,6 @@ start_proxy 22 172.30.10.1:22
start_proxy 8081 router:8081
start_proxy 8085 tts-gateway:8085
start_proxy 8091 piper:8085
start_proxy 8092 xtts:80
start_proxy 8202 mcp-athena-operator:8000
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
cd "$STACK_DIR"
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
xtts piper tts-gateway router open-webui
qwen3-tts piper tts-gateway router open-webui
deadline=$((SECONDS + 180))
for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
mike-ai-router mike-ai-open-webui; do
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
"$container" 2>/dev/null || true) == healthy ]]; do
+1 -1
View File
@@ -177,7 +177,7 @@ TTS_VOICES = tuple(v.strip() for v in os.environ.get(
"TTS_VOICES", "claribel").split(",") if v.strip())
TTS_DEFAULT_VOICE = os.environ.get(
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
TTS_FORMATS = ("mp3", "wav")
TTS_FORMATS = ("mp3", "wav", "pcm")
TTS_DEFAULT_FORMAT = "mp3"
# --- Spracherkennung (whisper.cpp, deutsch, CPU-only) ---
+1 -1
View File
@@ -28,7 +28,7 @@ for name in \
mike-ai-profile-controller \
mike-ai-router \
mike-ai-piper \
mike-ai-xtts \
mike-ai-qwen3-tts \
mike-ai-tts-gateway \
mike-ai-llama-dashboard \
mike-ai-mcp-athena-operator \