diff --git a/.env.example b/.env.example index ca3166f..288d6a7 100644 --- a/.env.example +++ b/.env.example @@ -7,9 +7,10 @@ FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high -XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 -XTTS_CACHE_DIR=/data/models/xtts-v2-cache -XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b +QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98 +QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache +QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices +QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b AI_DNS=192.168.1.1 DEFAULT_REASONING_EFFORT=off diff --git a/ATHENA.md b/ATHENA.md index b8e93e0..f2f966f 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -11,7 +11,7 @@ Sie betreibt: - llama.cpp mit genau einem aktiven Qwen-Profil, - den OpenAI-kompatiblen Profile Router, - FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung, -- XTTS und Piper für Sprache, +- Qwen3-TTS und Piper für Sprache, - das Athena-Dashboard, - Portainer CE als optionale Ansicht auf die laufenden Docker-Container, - WireGuard-Gateway und Datenbackup, @@ -58,7 +58,7 @@ Qwen-Profil wird vom Profile Controller verwaltet. - Uncensored: separates lokales Profil - FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach automatisch wiederhergestellt -- XTTS: RTX 3060; Piper bleibt CPU-Fallback +- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback Die verbindlichen Werte stehen in `config/profile-matrix.json` und `docs/STANDARD_PROFILE_MATRIX.md`. diff --git a/README.md b/README.md index edc9675..6c2a579 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored - Profile Router auf Port 8081 - FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080 -- XTTS auf der RTX 3060 mit Piper als CPU-Fallback +- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback - Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 - Portainer CE als optionale Container-Ansicht auf Port 9443 diff --git a/compose.yaml b/compose.yaml index e56a96b..5757bf4 100644 --- a/compose.yaml +++ b/compose.yaml @@ -722,9 +722,9 @@ services: retries: 30 start_period: 120s - xtts: - image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90} - container_name: mike-ai-xtts + qwen3-tts: + image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} + container_name: mike-ai-qwen3-tts restart: unless-stopped deploy: resources: @@ -732,30 +732,33 @@ services: devices: - driver: nvidia device_ids: - - ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} + - ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} capabilities: [gpu] read_only: true shm_size: 1g tmpfs: - /tmp:size=1g,mode=1777 - - /root/.cache:size=2g,mode=0700 volumes: - - "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts" + - "${QWEN3_TTS_CACHE_DIR:-/data/models/qwen3-tts-cache}:/root/.cache/huggingface" + - "${QWEN3_TTS_VOICES_DIR:-/data/models/qwen3-tts-voices}:/data/voices" environment: - COQUI_TOS_AGREED: "1" - NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} + NVIDIA_VISIBLE_DEVICES: ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} NVIDIA_DRIVER_CAPABILITIES: compute,utility CUDA_VISIBLE_DEVICES: "0" - NUM_THREADS: "4" + HF_HOME: /root/.cache/huggingface + NUMBA_CACHE_DIR: /tmp/numba + QWEN3_TTS_MODEL_ID: Qwen/Qwen3-TTS-12Hz-1.7B-Base + QWEN3_TTS_DEFAULT_VOICE: serena + QWEN3_TTS_VOICES_DIR: /data/voices networks: [frontend] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: - test: [CMD, curl, -fsS, "http://127.0.0.1/languages"] + test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8001/health', timeout=2)"] interval: 10s timeout: 5s - retries: 36 - start_period: 240s + retries: 60 + start_period: 600s tts-gateway: build: @@ -769,22 +772,17 @@ services: environment: TTS_GATEWAY_HOST: 0.0.0.0 TTS_GATEWAY_PORT: "8085" - XTTS_URL: http://xtts:80 + QWEN_TTS_URL: http://qwen3-tts:8001 + QWEN_TTS_MODEL: tts-1 + QWEN_TTS_VOICE: serena + QWEN_TTS_LANGUAGE: German + QWEN_TTS_TIMEOUT: "120" PIPER_URL: http://piper:8085 TTS_VOICE_ALIAS: alloy - XTTS_SPEAKER: Annmarie Nele TTS_DEFAULT_LANGUAGE: de # Mixed-language clip stitching caused long pauses and unintelligible # transitions. Keep full sentences in one stable German voice. TTS_CODE_SWITCH_ENABLED: "false" - XTTS_QUEUE_TIMEOUT: "15" - XTTS_TIMEOUT: "120" - # Keep normal sentences intact for natural prosody. This is only the - # safety ceiling for unusually long sentences. - XTTS_CHUNK_CHARS: "420" - # XTTS occasionally inserts multi-second silence inside a phrase. - XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650" - XTTS_INTERNAL_SILENCE_KEEP_MS: "220" PIPER_TIMEOUT: "120" networks: [frontend] depends_on: diff --git a/config/install.env.example b/config/install.env.example index 42f3b54..8c1a3ea 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -113,7 +113,8 @@ ULTRA_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high -XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 -XTTS_CACHE_DIR=/data/models/xtts-v2-cache +QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98 +QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache +QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices # Stable UUID of Athena's RTX 3060. Do not use a positional GPU index here. -XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b +QWEN3_TTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b diff --git a/install.sh b/install.sh index b9a356f..7aab107 100755 --- a/install.sh +++ b/install.sh @@ -286,8 +286,10 @@ setup_wireguard() { install_stack_files() { log "Stackdateien installieren" - XTTS_CACHE_DIR=${XTTS_CACHE_DIR:-$MODEL_DIR/xtts-v2-cache} - install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$XTTS_CACHE_DIR" "$STATE_DIR/backups" + QWEN3_TTS_CACHE_DIR=${QWEN3_TTS_CACHE_DIR:-$MODEL_DIR/qwen3-tts-cache} + QWEN3_TTS_VOICES_DIR=${QWEN3_TTS_VOICES_DIR:-$MODEL_DIR/qwen3-tts-voices} + install -d -m 0755 "$STACK_DIR" "$MODEL_DIR" "$QWEN3_TTS_CACHE_DIR" \ + "$QWEN3_TTS_VOICES_DIR" "$STATE_DIR/backups" # /opt/mike-ai/stack is both the live stack and the one canonical Git # checkout. Copying everything except .git created two competing source # trees and made agents reconstruct deployment state on every change. @@ -314,9 +316,10 @@ ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key) CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token) PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0} PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high} -XTTS_IMAGE=${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90} -XTTS_CACHE_DIR=$XTTS_CACHE_DIR -XTTS_GPU_DEVICE=${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} +QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} +QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR +QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR +QWEN3_TTS_GPU_DEVICE=${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} AI_DNS=${WG_DNS:-1.1.1.1} FAST_MODEL_FILE=$FAST_MODEL_FILE MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE @@ -501,7 +504,7 @@ build_and_start() { docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ - wireguard-gateway xtts piper tts-gateway profile-controller router llama-dashboard portainer backup + wireguard-gateway qwen3-tts piper tts-gateway profile-controller router llama-dashboard portainer backup if [[ ${WIREGUARD_MODE:-container} == container ]]; then systemctl restart mike-ai-container-vpn-guard.service fi diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 8d6993e..8ad4a53 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -110,6 +110,22 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertIn("29 Kilometer pro Stunde", spoken) self.assertNotIn("km", spoken.lower()) + def test_qwen_normalizes_ipv4_time_date_and_count(self): + spoken = gateway.prepare_for_qwen_speech( + "full_kiosk (192.168.1.5): 2× Timeout um 02:14 seit 01.09." + ) + self.assertIn("full kiosk", spoken) + self.assertIn("192 Punkt 168 Punkt 1 Punkt 5", spoken) + self.assertIn("2 mal Timeout", spoken) + self.assertIn("2 Uhr 14", spoken) + self.assertIn("1. September", spoken) + + def test_qwen_keeps_prosody_punctuation(self): + spoken = gateway.prepare_for_qwen_speech( + "Ist das gut? Ja! SarahTV: erreichbar." + ) + self.assertEqual(spoken, "Ist das gut? Ja! SarahTV: erreichbar.") + def test_visual_punctuation_becomes_natural_pauses(self): spoken = gateway.clean_for_speech( "Status: stabil – keine Fehler; Docker-Container laufen." diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index ef04587..b978034 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Private XTTS-first TTS gateway with a Piper fallback. +"""Private Qwen3-TTS-first gateway with a Piper fallback. The gateway implements the narrow /status and /tts protocol already consumed by the profile router. Request text is never logged or persisted. @@ -25,6 +25,11 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer HOST = os.getenv("TTS_GATEWAY_HOST", "0.0.0.0") PORT = int(os.getenv("TTS_GATEWAY_PORT", "8085")) +QWEN_TTS_URL = os.getenv("QWEN_TTS_URL", "http://qwen3-tts:8001").rstrip("/") +QWEN_TTS_MODEL = os.getenv("QWEN_TTS_MODEL", "tts-1") +QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena") +QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German") +QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120")) XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/") PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/") VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy") @@ -223,8 +228,25 @@ def _spell_digits(value: str) -> str: return " ".join(GERMAN_DIGITS[digit] for digit in value) +def _spoken_ipv4(match: re.Match) -> str: + """Keep IPv4 octets intact while making the separators pronounceable.""" + return " Punkt ".join(str(int(part)) for part in match.group(0).split(".")) + + def normalize_for_german_speech(text: str) -> str: """Turn common visual notation into unambiguous spoken German.""" + # Run these before the date rule: otherwise 192.168.1.5 could be partly + # interpreted as a visual date. + text = re.sub( + r"\b(?:\d{1,3}\.){3}\d{1,3}\b", + _spoken_ipv4, + text, + ) + text = re.sub( + r"\b([01]?\d|2[0-3]):([0-5]\d)\b", + lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}", + text, + ) text = re.sub( r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*" r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?", @@ -281,9 +303,31 @@ def normalize_for_german_speech(text: str) -> str: flags=re.IGNORECASE, ) text = re.sub(r"(\d)\s*%", r"\1 Prozent", text) + text = re.sub(r"\b(\d+)\s*[×x]\s*", r"\1 mal ", text) return text +def prepare_for_qwen_speech(text: str) -> str: + """Normalize technical display text without destroying Qwen's prosody.""" + text = re.sub(r"```.*?```", " Codeblock. ", text, flags=re.DOTALL) + text = re.sub(r"`([^`]+)`", r"\1", text) + text = re.sub(r"!\[([^]]*)\]\([^)]+\)", r"\1", text) + text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text) + text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text) + text = re.sub(r"(?m)^\s*[-*+]\s+", "", text) + text = text.replace("_", " ").replace("/", ", ") + text = text.replace("→", ". ").replace("←", ". ") + if DEFAULT_LANGUAGE == "de": + text = normalize_for_german_speech(text) + text = "".join( + char for char in text + if unicodedata.category(char) not in {"So", "Cs"} + ) + text = re.sub(r"[ \t]+", " ", text) + text = re.sub(r"\s*\n+\s*", ". ", text) + return text.strip() + + def clean_for_speech(text: str) -> str: """Remove visual markup that makes long TTS output unstable or noisy.""" text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL) @@ -553,8 +597,17 @@ def _wav(pcm: bytes) -> bytes: def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, str]: if output_format == "wav" and speed == 1.0: return wav_bytes, "audio/wav" - codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] \ - if output_format == "mp3" else ["-codec:a", "pcm_s16le", "-f", "wav"] + if output_format == "mp3": + codec = ["-codec:a", "libmp3lame", "-b:a", "96k", "-f", "mp3"] + content_type = "audio/mpeg" + elif output_format == "pcm": + # Hermes' OpenAI streaming TTS client expects headerless 24 kHz, + # mono, signed 16-bit little-endian PCM chunks. + codec = ["-ac", "1", "-ar", "24000", "-codec:a", "pcm_s16le", "-f", "s16le"] + content_type = "application/octet-stream" + else: + codec = ["-codec:a", "pcm_s16le", "-f", "wav"] + content_type = "audio/wav" command = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-f", "wav", "-i", "pipe:0"] if speed != 1.0: @@ -565,7 +618,7 @@ def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes, check=False, timeout=120) if result.returncode != 0 or not result.stdout: raise RuntimeError("audio conversion failed") - return result.stdout, "audio/mpeg" if output_format == "mp3" else "audio/wav" + return result.stdout, content_type def synthesize_xtts(text: str, output_format: str, @@ -583,24 +636,40 @@ def synthesize_xtts(text: str, output_format: str, def synthesize_piper(text: str, output_format: str, speed: float) -> tuple[bytes, str]: - return _request( + upstream_format = "wav" if output_format == "pcm" else output_format + audio, content_type = _request( f"{PIPER_URL}/tts", payload={"text": text, "voice": "alloy", "speed": speed, - "format": output_format}, + "format": upstream_format}, timeout=PIPER_TIMEOUT, ) + return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type) + + +def synthesize_qwen(text: str, output_format: str, + speed: float) -> tuple[bytes, str]: + text = prepare_for_qwen_speech(text) + upstream_format = "wav" if output_format == "pcm" else output_format + audio, content_type = _request( + f"{QWEN_TTS_URL}/v1/audio/speech", + payload={"model": QWEN_TTS_MODEL, "input": text, + "voice": QWEN_TTS_VOICE, "language": QWEN_TTS_LANGUAGE, + "response_format": upstream_format, "speed": speed}, + timeout=QWEN_TTS_TIMEOUT, + ) + return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type) def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]: acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT) if acquired: try: - audio = synthesize_xtts(text, output_format, speed) + audio = synthesize_qwen(text, output_format, speed) with STATE_LOCK: - STATE["last_backend"] = "xtts-v2" + STATE["last_backend"] = "qwen3-tts-1.7b" STATE["last_error"] = None return audio - except Exception as exc: # fallback must cover all XTTS failures + except Exception as exc: # fallback must cover all Qwen failures with STATE_LOCK: STATE["xtts_failures"] += 1 STATE["last_error"] = type(exc).__name__ @@ -641,7 +710,7 @@ class Handler(BaseHTTPRequestHandler): if self.path != "/status": self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"}) return - primary_ready = _reachable(XTTS_URL, "/languages") + primary_ready = _reachable(QWEN_TTS_URL, "/health") fallback_ready = _reachable(PIPER_URL, "/status") with STATE_LOCK: state = dict(STATE) @@ -649,10 +718,10 @@ class Handler(BaseHTTPRequestHandler): HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE, { "ready": fallback_ready, - "engine": "xtts-v2-with-piper-fallback", - "model": "xtts-v2", + "engine": "qwen3-tts-with-piper-fallback", + "model": "Qwen3-TTS-12Hz-1.7B-Base", "voices": [VOICE_ALIAS], - "speaker": XTTS_SPEAKER, + "speaker": QWEN_TTS_VOICE, "primary_ready": primary_ready, "fallback_ready": fallback_ready, **state, @@ -686,7 +755,7 @@ class Handler(BaseHTTPRequestHandler): if voice != VOICE_ALIAS: self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"}) return - if output_format not in {"wav", "mp3"}: + if output_format not in {"wav", "mp3", "pcm"}: self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"}) return if not 0.5 <= speed <= 2.0: @@ -707,5 +776,5 @@ class Handler(BaseHTTPRequestHandler): if __name__ == "__main__": - print(f"TTS gateway ready on {HOST}:{PORT}; primary={XTTS_SPEAKER}; fallback=Piper") + print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper") ThreadingHTTPServer((HOST, PORT), Handler).serve_forever() diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index c6f0e72..0dabbb5 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -77,7 +77,6 @@ start_proxy 22 172.30.10.1:22 start_proxy 8081 router:8081 start_proxy 8085 tts-gateway:8085 start_proxy 8091 piper:8085 -start_proxy 8092 xtts:80 start_proxy 8202 mcp-athena-operator:8000 wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}') diff --git a/platform/migration/restore-reference-backup.sh b/platform/migration/restore-reference-backup.sh index f554c7b..e03e571 100755 --- a/platform/migration/restore-reference-backup.sh +++ b/platform/migration/restore-reference-backup.sh @@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren" log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen" cd "$STACK_DIR" docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \ - xtts piper tts-gateway router open-webui + qwen3-tts piper tts-gateway router open-webui deadline=$((SECONDS + 180)) -for container in mike-ai-xtts mike-ai-piper mike-ai-tts-gateway \ +for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \ mike-ai-router mike-ai-open-webui; do until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \ "$container" 2>/dev/null || true) == healthy ]]; do diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index adf6bf4..4277d2d 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -177,7 +177,7 @@ TTS_VOICES = tuple(v.strip() for v in os.environ.get( "TTS_VOICES", "claribel").split(",") if v.strip()) TTS_DEFAULT_VOICE = os.environ.get( "TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel") -TTS_FORMATS = ("mp3", "wav") +TTS_FORMATS = ("mp3", "wav", "pcm") TTS_DEFAULT_FORMAT = "mp3" # --- Spracherkennung (whisper.cpp, deutsch, CPU-only) --- diff --git a/smoke-test.sh b/smoke-test.sh index aeb60ff..49ce016 100755 --- a/smoke-test.sh +++ b/smoke-test.sh @@ -28,7 +28,7 @@ for name in \ mike-ai-profile-controller \ mike-ai-router \ mike-ai-piper \ - mike-ai-xtts \ + mike-ai-qwen3-tts \ mike-ai-tts-gateway \ mike-ai-llama-dashboard \ mike-ai-mcp-athena-operator \