Remove Beta 1 and Piper fallback

This commit is contained in:
Mikei386
2026-09-10 13:17:57 +02:00
parent 52482627be
commit 08ff3d7c4e
38 changed files with 94 additions and 484 deletions
-24
View File
@@ -1,24 +0,0 @@
FROM python:3.12-slim-bookworm
ARG PIPER_TTS_VERSION=1.6.0
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates curl ffmpeg gosu \
&& python -m pip install --no-cache-dir "piper-tts==${PIPER_TTS_VERSION}" \
&& useradd --system --uid 10003 --home-dir /nonexistent --shell /usr/sbin/nologin piper \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
COPY piper_worker.py /app/piper_worker.py
COPY entrypoint.sh /usr/local/bin/mike-ai-piper-entrypoint
RUN chmod 0755 /usr/local/bin/mike-ai-piper-entrypoint
ENV PIPER_DATA_DIR=/data \
PIPER_VOICE=de_DE-thorsten-high \
PIPER_VOICE_ALIAS=alloy \
PIPER_HOST=0.0.0.0 \
PIPER_PORT=8085
VOLUME ["/data"]
EXPOSE 8085
ENTRYPOINT ["/usr/local/bin/mike-ai-piper-entrypoint"]
-15
View File
@@ -1,15 +0,0 @@
#!/bin/sh
set -eu
data_dir=${PIPER_DATA_DIR:-/data}
voice=${PIPER_VOICE:-de_DE-thorsten-high}
mkdir -p "$data_dir"
chown 10003:10003 "$data_dir"
if [ ! -s "$data_dir/$voice.onnx" ] || [ ! -s "$data_dir/$voice.onnx.json" ]; then
echo "Downloading Piper voice: $voice"
gosu piper python -m piper.download_voices --data-dir "$data_dir" "$voice"
fi
exec gosu piper python /app/piper_worker.py
-153
View File
@@ -1,153 +0,0 @@
#!/usr/bin/env python3
"""Small, private Piper worker for the Mike AI profile router.
The public OpenAI-compatible endpoint remains in the router. This worker only
accepts the narrow internal /status and /tts protocol and never logs input text.
"""
from __future__ import annotations
import io
import json
import os
import subprocess
import threading
import wave
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from piper import PiperVoice, SynthesisConfig
DATA_DIR = Path(os.getenv("PIPER_DATA_DIR", "/data"))
VOICE_NAME = os.getenv("PIPER_VOICE", "de_DE-thorsten-high")
VOICE_ALIAS = os.getenv("PIPER_VOICE_ALIAS", "alloy")
HOST = os.getenv("PIPER_HOST", "0.0.0.0")
PORT = int(os.getenv("PIPER_PORT", "8085"))
MAX_TEXT_CHARS = int(os.getenv("PIPER_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("PIPER_MAX_REQUEST_BYTES", "65536"))
VOICE_PATH = DATA_DIR / f"{VOICE_NAME}.onnx"
VOICE = PiperVoice.load(str(VOICE_PATH))
SYNTHESIS_LOCK = threading.Lock()
def synthesize_wav(text: str, speed: float) -> bytes:
"""Synthesize a complete WAV in memory without retaining the text."""
output = io.BytesIO()
config = SynthesisConfig(length_scale=1.0 / speed)
with SYNTHESIS_LOCK, wave.open(output, "wb") as wav_file:
VOICE.synthesize_wav(text, wav_file, syn_config=config)
return output.getvalue()
def wav_to_mp3(wav_bytes: bytes) -> bytes:
"""Convert Piper's WAV to the MP3 format Open WebUI requests by default."""
result = subprocess.run(
[
"ffmpeg", "-hide_banner", "-loglevel", "error",
"-f", "wav", "-i", "pipe:0",
"-codec:a", "libmp3lame", "-b:a", "96k",
"-f", "mp3", "pipe:1",
],
input=wav_bytes,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
timeout=120,
)
if result.returncode != 0:
raise RuntimeError("ffmpeg conversion failed")
return result.stdout
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def log_message(self, fmt: str, *args: object) -> None:
# Deliberately omit URLs and request bodies from the log.
print(f"piper-worker: {self.command} -> {args[1] if len(args) > 1 else '-'}")
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
self.send_response(status)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(body)))
self.send_header("Cache-Control", "no-store")
self.end_headers()
self.wfile.write(body)
def send_json(self, status: int, payload: dict) -> None:
self.send_bytes(
status,
json.dumps(payload, separators=(",", ":")).encode(),
"application/json",
)
def do_GET(self) -> None: # noqa: N802
if self.path != "/status":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
self.send_json(
HTTPStatus.OK,
{
"ready": True,
"engine": "piper",
"model": VOICE_NAME,
"voices": [VOICE_ALIAS],
},
)
def do_POST(self) -> None: # noqa: N802
if self.path != "/tts":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
try:
content_length = int(self.headers.get("Content-Length", "0"))
except ValueError:
content_length = 0
if content_length <= 0 or content_length > MAX_REQUEST_BYTES:
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, {"error": "invalid request size"})
return
try:
request = json.loads(self.rfile.read(content_length))
text = request.get("text", "")
voice = request.get("voice", VOICE_ALIAS)
output_format = request.get("format", "mp3")
speed = float(request.get("speed", 1.0))
except (json.JSONDecodeError, TypeError, ValueError):
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
return
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
return
if voice != VOICE_ALIAS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
return
if output_format not in {"wav", "mp3"}:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
return
if not 0.5 <= speed <= 2.0:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
return
try:
audio = synthesize_wav(text.strip(), speed)
if output_format == "mp3":
audio = wav_to_mp3(audio)
content_type = "audio/mpeg"
else:
content_type = "audio/wav"
except (OSError, RuntimeError, subprocess.SubprocessError):
self.send_json(HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "synthesis failed"})
return
self.send_bytes(HTTPStatus.OK, audio, content_type)
if __name__ == "__main__":
print(f"Piper worker ready: {VOICE_NAME} as {VOICE_ALIAS} on {HOST}:{PORT}")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
@@ -230,7 +230,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
for profile_item in containers().values():
stop_container(profile_item)
# The 9B beta text encoder temporarily borrows the RTX 3060 from
# Qwen3-TTS. The gateway retains Piper as a fallback meanwhile.
# Qwen3-TTS. TTS is unavailable during this exclusive GPU phase.
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
@@ -244,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
class FallbackTests(unittest.TestCase):
class BackendFailureTests(unittest.TestCase):
def setUp(self):
self.original_xtts = gateway.synthesize_xtts
self.original_piper = gateway.synthesize_piper
self.original_qwen = gateway.synthesize_qwen
def tearDown(self):
gateway.synthesize_xtts = self.original_xtts
gateway.synthesize_piper = self.original_piper
gateway.synthesize_qwen = self.original_qwen
def test_piper_is_used_when_xtts_fails(self):
def test_qwen_failure_is_reported_without_fallback(self):
def fail(*_args):
raise RuntimeError("synthetic XTTS failure")
raise RuntimeError("synthetic Qwen failure")
gateway.synthesize_xtts = fail
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
self.assertEqual(
gateway.synthesize("synthetic test", "wav", 1.0),
(b"piper", "audio/wav"),
)
gateway.synthesize_qwen = fail
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
gateway.synthesize("synthetic test", "wav", 1.0)
class AudioJoinTests(unittest.TestCase):
+15 -33
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
"""Private Qwen3-TTS-first gateway with a Piper fallback.
"""Private Qwen3-TTS gateway.
The gateway implements the narrow /status and /tts protocol already consumed
by the profile router. Request text is never logged or persisted.
@@ -33,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
@@ -43,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
# XTTS loses natural prosody when a sentence is synthesized as many tiny
# requests: every request starts a fresh utterance. Keep complete sentences
@@ -63,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
SPEAKER_CONDITIONING: dict | None = None
STATE = {
"last_backend": None,
"xtts_failures": 0,
"piper_fallbacks": 0,
"qwen_failures": 0,
"last_error": None,
}
@@ -738,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
def synthesize_piper(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
upstream_format = "wav" if output_format == "pcm" else output_format
audio, content_type = _request(
f"{PIPER_URL}/tts",
payload={"text": text, "voice": "alloy", "speed": speed,
"format": upstream_format},
timeout=PIPER_TIMEOUT,
)
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def synthesize_qwen(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
text = prepare_for_qwen_speech(text)
@@ -814,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
STATE["last_backend"] = "qwen3-tts-1.7b"
STATE["last_error"] = None
return audio
except Exception as exc: # fallback must cover all Qwen failures
except Exception as exc:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = type(exc).__name__
raise
finally:
SYNTHESIS_LOCK.release()
else:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = "queue-timeout"
audio = synthesize_piper(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "piper"
STATE["piper_fallbacks"] += 1
return audio
raise RuntimeError("speech queue timeout")
class Handler(BaseHTTPRequestHandler):
@@ -856,19 +837,20 @@ class Handler(BaseHTTPRequestHandler):
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
primary_ready = _reachable(QWEN_TTS_URL, "/health")
fallback_ready = _reachable(PIPER_URL, "/status")
with STATE_LOCK:
state = dict(STATE)
# This endpoint is also the container liveness check. Qwen3-TTS is
# deliberately stopped in exclusive GPU modes such as Applio, so the
# gateway itself must stay healthy while reporting ready=false.
self.send_json(
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
HTTPStatus.OK,
{
"ready": fallback_ready,
"engine": "qwen3-tts-with-piper-fallback",
"ready": primary_ready,
"engine": "qwen3-tts",
"model": "Qwen3-TTS-12Hz-1.7B-Base",
"voices": [VOICE_ALIAS],
"speaker": QWEN_TTS_VOICE,
"primary_ready": primary_ready,
"fallback_ready": fallback_ready,
**state,
},
)
@@ -916,7 +898,7 @@ class Handler(BaseHTTPRequestHandler):
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "all local speech backends failed"})
{"error": "local Qwen3-TTS backend failed"})
return
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
f"{time.monotonic() - started:.2f}s")
@@ -992,5 +974,5 @@ class Handler(BaseHTTPRequestHandler):
if __name__ == "__main__":
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
@@ -76,7 +76,6 @@ start_proxy() {
start_proxy 22 172.30.10.1:22
start_proxy 8081 router:8081
start_proxy 8085 tts-gateway:8085
start_proxy 8091 piper:8085
start_proxy 8099 llama-dashboard:8099
start_proxy 7861 music-ui:3000
start_proxy 7862 music-worker:7860
+2 -3
View File
@@ -144,13 +144,12 @@ stt:
model: "base"
language: "de"
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
# Reuse Athena's OpenAI-compatible Qwen3-TTS route.
tts:
provider: "openai"
speed: 1.0
openai:
model: "piper"
model: "qwen3-tts"
voice: "alloy"
speed: 1.0
base_url: "http://router:8081/v1"
@@ -25,7 +25,7 @@ image worker, then restores the previous LLM state.
Only one heavy GPU path may be active. Do not manually start a second GPU
worker around the controller. The lightweight dashboard, router, controller,
gateway, UI, CPU-STT, Piper, backup and operator containers may remain active.
gateway, UI, CPU-STT, backup and operator containers may remain active.
## Model profiles
@@ -33,7 +33,6 @@ gateway, UI, CPU-STT, Piper, backup and operator containers may remain active.
- Medium: Qwen3.8-27B IQ4_XS-pure, 160,000 tokens, vision.
- Large: the same Q4 model, 192,000 tokens, vision.
- Ultra: the same Q4 model, 262,144 tokens, no vision projector.
- Beta 1: Qwen3.8-27B GSQ-RCO IQ3_S-MTP, 112,000 tokens; experimental only.
- Uncensored: Abliterated Q4_K_M, 80,000 tokens, vision.
Medium, Large, Ultra and Beta distribute their runtime across both GPUs. Do not
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
cd "$STACK_DIR"
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
qwen3-tts piper tts-gateway router open-webui
qwen3-tts tts-gateway router open-webui
deadline=$((SECONDS + 180))
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
for container in mike-ai-qwen3-tts mike-ai-tts-gateway \
mike-ai-router mike-ai-open-webui; do
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
"$container" 2>/dev/null || true) == healthy ]]; do
+1 -1
View File
@@ -83,7 +83,7 @@ with con:
for key, value in (
("task.follow_up.enable", False),
("audio.tts.engine", "openai"),
("audio.tts.model", "piper"),
("audio.tts.model", "qwen3-tts"),
("audio.tts.voice", "alloy"),
("audio.tts.openai.api_base_url", "http://router:8081/v1"),
("web.search.enable", True),
+2 -2
View File
@@ -7,9 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
PROFILE="${1:-}"
case "$PROFILE" in
fast|medium|beta1|large|ultra|uncensored) ;;
fast|medium|large|ultra|uncensored) ;;
*)
echo "Usage: llama-profile {fast|medium|beta1|large|ultra|uncensored}" >&2
echo "Usage: llama-profile {fast|medium|large|ultra|uncensored}" >&2
exit 2
;;
esac