Remove Beta 1 and Piper fallback
This commit is contained in:
@@ -1,24 +0,0 @@
|
||||
FROM python:3.12-slim-bookworm
|
||||
|
||||
ARG PIPER_TTS_VERSION=1.6.0
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl ffmpeg gosu \
|
||||
&& python -m pip install --no-cache-dir "piper-tts==${PIPER_TTS_VERSION}" \
|
||||
&& useradd --system --uid 10003 --home-dir /nonexistent --shell /usr/sbin/nologin piper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
COPY piper_worker.py /app/piper_worker.py
|
||||
COPY entrypoint.sh /usr/local/bin/mike-ai-piper-entrypoint
|
||||
RUN chmod 0755 /usr/local/bin/mike-ai-piper-entrypoint
|
||||
|
||||
ENV PIPER_DATA_DIR=/data \
|
||||
PIPER_VOICE=de_DE-thorsten-high \
|
||||
PIPER_VOICE_ALIAS=alloy \
|
||||
PIPER_HOST=0.0.0.0 \
|
||||
PIPER_PORT=8085
|
||||
|
||||
VOLUME ["/data"]
|
||||
EXPOSE 8085
|
||||
ENTRYPOINT ["/usr/local/bin/mike-ai-piper-entrypoint"]
|
||||
@@ -1,15 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
data_dir=${PIPER_DATA_DIR:-/data}
|
||||
voice=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||
|
||||
mkdir -p "$data_dir"
|
||||
chown 10003:10003 "$data_dir"
|
||||
|
||||
if [ ! -s "$data_dir/$voice.onnx" ] || [ ! -s "$data_dir/$voice.onnx.json" ]; then
|
||||
echo "Downloading Piper voice: $voice"
|
||||
gosu piper python -m piper.download_voices --data-dir "$data_dir" "$voice"
|
||||
fi
|
||||
|
||||
exec gosu piper python /app/piper_worker.py
|
||||
@@ -1,153 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small, private Piper worker for the Mike AI profile router.
|
||||
|
||||
The public OpenAI-compatible endpoint remains in the router. This worker only
|
||||
accepts the narrow internal /status and /tts protocol and never logs input text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import threading
|
||||
import wave
|
||||
from http import HTTPStatus
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
|
||||
from piper import PiperVoice, SynthesisConfig
|
||||
|
||||
|
||||
DATA_DIR = Path(os.getenv("PIPER_DATA_DIR", "/data"))
|
||||
VOICE_NAME = os.getenv("PIPER_VOICE", "de_DE-thorsten-high")
|
||||
VOICE_ALIAS = os.getenv("PIPER_VOICE_ALIAS", "alloy")
|
||||
HOST = os.getenv("PIPER_HOST", "0.0.0.0")
|
||||
PORT = int(os.getenv("PIPER_PORT", "8085"))
|
||||
MAX_TEXT_CHARS = int(os.getenv("PIPER_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("PIPER_MAX_REQUEST_BYTES", "65536"))
|
||||
|
||||
VOICE_PATH = DATA_DIR / f"{VOICE_NAME}.onnx"
|
||||
VOICE = PiperVoice.load(str(VOICE_PATH))
|
||||
SYNTHESIS_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def synthesize_wav(text: str, speed: float) -> bytes:
|
||||
"""Synthesize a complete WAV in memory without retaining the text."""
|
||||
output = io.BytesIO()
|
||||
config = SynthesisConfig(length_scale=1.0 / speed)
|
||||
with SYNTHESIS_LOCK, wave.open(output, "wb") as wav_file:
|
||||
VOICE.synthesize_wav(text, wav_file, syn_config=config)
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def wav_to_mp3(wav_bytes: bytes) -> bytes:
|
||||
"""Convert Piper's WAV to the MP3 format Open WebUI requests by default."""
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-loglevel", "error",
|
||||
"-f", "wav", "-i", "pipe:0",
|
||||
"-codec:a", "libmp3lame", "-b:a", "96k",
|
||||
"-f", "mp3", "pipe:1",
|
||||
],
|
||||
input=wav_bytes,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
check=False,
|
||||
timeout=120,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError("ffmpeg conversion failed")
|
||||
return result.stdout
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
protocol_version = "HTTP/1.1"
|
||||
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
# Deliberately omit URLs and request bodies from the log.
|
||||
print(f"piper-worker: {self.command} -> {args[1] if len(args) > 1 else '-'}")
|
||||
|
||||
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", content_type)
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def send_json(self, status: int, payload: dict) -> None:
|
||||
self.send_bytes(
|
||||
status,
|
||||
json.dumps(payload, separators=(",", ":")).encode(),
|
||||
"application/json",
|
||||
)
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
if self.path != "/status":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
self.send_json(
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": True,
|
||||
"engine": "piper",
|
||||
"model": VOICE_NAME,
|
||||
"voices": [VOICE_ALIAS],
|
||||
},
|
||||
)
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.path != "/tts":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
|
||||
try:
|
||||
content_length = int(self.headers.get("Content-Length", "0"))
|
||||
except ValueError:
|
||||
content_length = 0
|
||||
if content_length <= 0 or content_length > MAX_REQUEST_BYTES:
|
||||
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, {"error": "invalid request size"})
|
||||
return
|
||||
|
||||
try:
|
||||
request = json.loads(self.rfile.read(content_length))
|
||||
text = request.get("text", "")
|
||||
voice = request.get("voice", VOICE_ALIAS)
|
||||
output_format = request.get("format", "mp3")
|
||||
speed = float(request.get("speed", 1.0))
|
||||
except (json.JSONDecodeError, TypeError, ValueError):
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
|
||||
return
|
||||
|
||||
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
|
||||
return
|
||||
if voice != VOICE_ALIAS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||
return
|
||||
if output_format not in {"wav", "mp3"}:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||
return
|
||||
if not 0.5 <= speed <= 2.0:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
|
||||
return
|
||||
|
||||
try:
|
||||
audio = synthesize_wav(text.strip(), speed)
|
||||
if output_format == "mp3":
|
||||
audio = wav_to_mp3(audio)
|
||||
content_type = "audio/mpeg"
|
||||
else:
|
||||
content_type = "audio/wav"
|
||||
except (OSError, RuntimeError, subprocess.SubprocessError):
|
||||
self.send_json(HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "synthesis failed"})
|
||||
return
|
||||
|
||||
self.send_bytes(HTTPStatus.OK, audio, content_type)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"Piper worker ready: {VOICE_NAME} as {VOICE_ALIAS} on {HOST}:{PORT}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
@@ -230,7 +230,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
|
||||
for profile_item in containers().values():
|
||||
stop_container(profile_item)
|
||||
# The 9B beta text encoder temporarily borrows the RTX 3060 from
|
||||
# Qwen3-TTS. The gateway retains Piper as a fallback meanwhile.
|
||||
# Qwen3-TTS. TTS is unavailable during this exclusive GPU phase.
|
||||
stop_container(tts_container(), timeout=30)
|
||||
stop_music_if_configured()
|
||||
stop_separator_if_configured()
|
||||
|
||||
@@ -244,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
|
||||
|
||||
|
||||
class FallbackTests(unittest.TestCase):
|
||||
class BackendFailureTests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.original_xtts = gateway.synthesize_xtts
|
||||
self.original_piper = gateway.synthesize_piper
|
||||
self.original_qwen = gateway.synthesize_qwen
|
||||
|
||||
def tearDown(self):
|
||||
gateway.synthesize_xtts = self.original_xtts
|
||||
gateway.synthesize_piper = self.original_piper
|
||||
gateway.synthesize_qwen = self.original_qwen
|
||||
|
||||
def test_piper_is_used_when_xtts_fails(self):
|
||||
def test_qwen_failure_is_reported_without_fallback(self):
|
||||
def fail(*_args):
|
||||
raise RuntimeError("synthetic XTTS failure")
|
||||
raise RuntimeError("synthetic Qwen failure")
|
||||
|
||||
gateway.synthesize_xtts = fail
|
||||
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
|
||||
self.assertEqual(
|
||||
gateway.synthesize("synthetic test", "wav", 1.0),
|
||||
(b"piper", "audio/wav"),
|
||||
)
|
||||
gateway.synthesize_qwen = fail
|
||||
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
|
||||
gateway.synthesize("synthetic test", "wav", 1.0)
|
||||
|
||||
|
||||
class AudioJoinTests(unittest.TestCase):
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private Qwen3-TTS-first gateway with a Piper fallback.
|
||||
"""Private Qwen3-TTS gateway.
|
||||
|
||||
The gateway implements the narrow /status and /tts protocol already consumed
|
||||
by the profile router. Request text is never logged or persisted.
|
||||
@@ -33,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
|
||||
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
|
||||
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
|
||||
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
|
||||
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
|
||||
@@ -43,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
|
||||
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
||||
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||
# XTTS loses natural prosody when a sentence is synthesized as many tiny
|
||||
# requests: every request starts a fresh utterance. Keep complete sentences
|
||||
@@ -63,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
|
||||
SPEAKER_CONDITIONING: dict | None = None
|
||||
STATE = {
|
||||
"last_backend": None,
|
||||
"xtts_failures": 0,
|
||||
"piper_fallbacks": 0,
|
||||
"qwen_failures": 0,
|
||||
"last_error": None,
|
||||
}
|
||||
|
||||
@@ -738,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
|
||||
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
|
||||
|
||||
|
||||
def synthesize_piper(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||
audio, content_type = _request(
|
||||
f"{PIPER_URL}/tts",
|
||||
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||
"format": upstream_format},
|
||||
timeout=PIPER_TIMEOUT,
|
||||
)
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def synthesize_qwen(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
text = prepare_for_qwen_speech(text)
|
||||
@@ -814,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
|
||||
STATE["last_backend"] = "qwen3-tts-1.7b"
|
||||
STATE["last_error"] = None
|
||||
return audio
|
||||
except Exception as exc: # fallback must cover all Qwen failures
|
||||
except Exception as exc:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
raise
|
||||
finally:
|
||||
SYNTHESIS_LOCK.release()
|
||||
else:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = "queue-timeout"
|
||||
|
||||
audio = synthesize_piper(text, output_format, speed)
|
||||
with STATE_LOCK:
|
||||
STATE["last_backend"] = "piper"
|
||||
STATE["piper_fallbacks"] += 1
|
||||
return audio
|
||||
raise RuntimeError("speech queue timeout")
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
@@ -856,19 +837,20 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
primary_ready = _reachable(QWEN_TTS_URL, "/health")
|
||||
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||
with STATE_LOCK:
|
||||
state = dict(STATE)
|
||||
# This endpoint is also the container liveness check. Qwen3-TTS is
|
||||
# deliberately stopped in exclusive GPU modes such as Applio, so the
|
||||
# gateway itself must stay healthy while reporting ready=false.
|
||||
self.send_json(
|
||||
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": fallback_ready,
|
||||
"engine": "qwen3-tts-with-piper-fallback",
|
||||
"ready": primary_ready,
|
||||
"engine": "qwen3-tts",
|
||||
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"voices": [VOICE_ALIAS],
|
||||
"speaker": QWEN_TTS_VOICE,
|
||||
"primary_ready": primary_ready,
|
||||
"fallback_ready": fallback_ready,
|
||||
**state,
|
||||
},
|
||||
)
|
||||
@@ -916,7 +898,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
with STATE_LOCK:
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{"error": "all local speech backends failed"})
|
||||
{"error": "local Qwen3-TTS backend failed"})
|
||||
return
|
||||
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
|
||||
f"{time.monotonic() - started:.2f}s")
|
||||
@@ -992,5 +974,5 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
|
||||
@@ -76,7 +76,6 @@ start_proxy() {
|
||||
start_proxy 22 172.30.10.1:22
|
||||
start_proxy 8081 router:8081
|
||||
start_proxy 8085 tts-gateway:8085
|
||||
start_proxy 8091 piper:8085
|
||||
start_proxy 8099 llama-dashboard:8099
|
||||
start_proxy 7861 music-ui:3000
|
||||
start_proxy 7862 music-worker:7860
|
||||
|
||||
@@ -144,13 +144,12 @@ stt:
|
||||
model: "base"
|
||||
language: "de"
|
||||
|
||||
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
|
||||
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
|
||||
# Reuse Athena's OpenAI-compatible Qwen3-TTS route.
|
||||
tts:
|
||||
provider: "openai"
|
||||
speed: 1.0
|
||||
openai:
|
||||
model: "piper"
|
||||
model: "qwen3-tts"
|
||||
voice: "alloy"
|
||||
speed: 1.0
|
||||
base_url: "http://router:8081/v1"
|
||||
|
||||
@@ -25,7 +25,7 @@ image worker, then restores the previous LLM state.
|
||||
|
||||
Only one heavy GPU path may be active. Do not manually start a second GPU
|
||||
worker around the controller. The lightweight dashboard, router, controller,
|
||||
gateway, UI, CPU-STT, Piper, backup and operator containers may remain active.
|
||||
gateway, UI, CPU-STT, backup and operator containers may remain active.
|
||||
|
||||
## Model profiles
|
||||
|
||||
@@ -33,7 +33,6 @@ gateway, UI, CPU-STT, Piper, backup and operator containers may remain active.
|
||||
- Medium: Qwen3.8-27B IQ4_XS-pure, 160,000 tokens, vision.
|
||||
- Large: the same Q4 model, 192,000 tokens, vision.
|
||||
- Ultra: the same Q4 model, 262,144 tokens, no vision projector.
|
||||
- Beta 1: Qwen3.8-27B GSQ-RCO IQ3_S-MTP, 112,000 tokens; experimental only.
|
||||
- Uncensored: Abliterated Q4_K_M, 80,000 tokens, vision.
|
||||
|
||||
Medium, Large, Ultra and Beta distribute their runtime across both GPUs. Do not
|
||||
|
||||
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
|
||||
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
||||
cd "$STACK_DIR"
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
||||
qwen3-tts piper tts-gateway router open-webui
|
||||
qwen3-tts tts-gateway router open-webui
|
||||
|
||||
deadline=$((SECONDS + 180))
|
||||
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
|
||||
for container in mike-ai-qwen3-tts mike-ai-tts-gateway \
|
||||
mike-ai-router mike-ai-open-webui; do
|
||||
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||
"$container" 2>/dev/null || true) == healthy ]]; do
|
||||
|
||||
@@ -83,7 +83,7 @@ with con:
|
||||
for key, value in (
|
||||
("task.follow_up.enable", False),
|
||||
("audio.tts.engine", "openai"),
|
||||
("audio.tts.model", "piper"),
|
||||
("audio.tts.model", "qwen3-tts"),
|
||||
("audio.tts.voice", "alloy"),
|
||||
("audio.tts.openai.api_base_url", "http://router:8081/v1"),
|
||||
("web.search.enable", True),
|
||||
|
||||
@@ -7,9 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||
PROFILE="${1:-}"
|
||||
|
||||
case "$PROFILE" in
|
||||
fast|medium|beta1|large|ultra|uncensored) ;;
|
||||
fast|medium|large|ultra|uncensored) ;;
|
||||
*)
|
||||
echo "Usage: llama-profile {fast|medium|beta1|large|ultra|uncensored}" >&2
|
||||
echo "Usage: llama-profile {fast|medium|large|ultra|uncensored}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
Reference in New Issue
Block a user