Synchronize repository with Athena deployment

This commit is contained in:
Mikei386
2026-09-13 20:01:36 +02:00
parent 040a2df48b
commit fce9900389
60 changed files with 2570 additions and 607 deletions
+27 -13
View File
@@ -132,6 +132,25 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertIn("5 bis 6 Uhr", spoken)
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
def test_qwen_speaks_aspect_ratios_as_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Cover sind im Hochformat (2:3), Screenshots im Querformat "
"(16:9), ein Quadrat im Seitenverhältnis 2:2 und 4:3-Format."
)
self.assertIn("Hochformat (2 zu 3)", spoken)
self.assertIn("Querformat (16 zu 9)", spoken)
self.assertIn("Seitenverhältnis 2 zu 2", spoken)
self.assertIn("4 zu 3-Format", spoken)
def test_qwen_keeps_clock_times_distinct_from_aspect_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Beginn um 16:09 Uhr, Fehler um 02:14; das Videoformat ist 16:9."
)
self.assertIn("16 Uhr 9", spoken)
self.assertNotIn("16 Uhr 9 Uhr", spoken)
self.assertIn("2 Uhr 14", spoken)
self.assertIn("Videoformat ist 16 zu 9", spoken)
def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self):
spoken = gateway.prepare_for_qwen_speech(
"Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026."
@@ -225,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
class FallbackTests(unittest.TestCase):
class BackendFailureTests(unittest.TestCase):
def setUp(self):
self.original_xtts = gateway.synthesize_xtts
self.original_piper = gateway.synthesize_piper
self.original_qwen = gateway.synthesize_qwen
def tearDown(self):
gateway.synthesize_xtts = self.original_xtts
gateway.synthesize_piper = self.original_piper
gateway.synthesize_qwen = self.original_qwen
def test_piper_is_used_when_xtts_fails(self):
def test_qwen_failure_is_reported_without_fallback(self):
def fail(*_args):
raise RuntimeError("synthetic XTTS failure")
raise RuntimeError("synthetic Qwen failure")
gateway.synthesize_xtts = fail
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
self.assertEqual(
gateway.synthesize("synthetic test", "wav", 1.0),
(b"piper", "audio/wav"),
)
gateway.synthesize_qwen = fail
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
gateway.synthesize("synthetic test", "wav", 1.0)
class AudioJoinTests(unittest.TestCase):
+166 -36
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
"""Private Qwen3-TTS-first gateway with a Piper fallback.
"""Private Qwen3-TTS gateway.
The gateway implements the narrow /status and /tts protocol already consumed
by the profile router. Request text is never logged or persisted.
@@ -8,6 +8,7 @@ by the profile router. Request text is never logged or persisted.
from __future__ import annotations
import io
import http.client
import json
import os
import re
@@ -17,6 +18,7 @@ import time
import unicodedata
import urllib.error
import urllib.request
import urllib.parse
import wave
from array import array
from http import HTTPStatus
@@ -31,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
@@ -41,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
# XTTS loses natural prosody when a sentence is synthesized as many tiny
# requests: every request starts a fresh utterance. Keep complete sentences
@@ -61,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
SPEAKER_CONDITIONING: dict | None = None
STATE = {
"last_backend": None,
"xtts_failures": 0,
"piper_fallbacks": 0,
"qwen_failures": 0,
"last_error": None,
}
@@ -267,6 +266,36 @@ def _spoken_ipv4(match: re.Match) -> str:
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
def _normalize_aspect_ratios(text: str) -> str:
"""Speak colon notation as a ratio only when the surrounding text says so.
A bare ``16:09`` remains a clock time. This deliberately avoids a global
replacement of common ratios because ``16:9`` can also be a valid time.
"""
cue = (
r"(?:Seitenverh[aä]ltnis|Bildseitenverh[aä]ltnis|Bildformat|"
r"Videoformat|Hochformat|Querformat|Format|Aspect[- ]?Ratio)"
)
text = re.sub(
rf"\b({cue}\b(?:\s+(?:von|im|ist|betr[aä]gt))?\s*[\(\[]?\s*)"
rf"(\d{{1,3}})\s*:\s*(\d{{1,3}})",
lambda match: (
f"{match.group(1)}{int(match.group(2))} zu {int(match.group(3))}"
),
text,
flags=re.IGNORECASE,
)
return re.sub(
rf"\b(\d{{1,3}})\s*:\s*(\d{{1,3}})"
rf"(\s*[-‐‑‒–—−]?\s*{cue}\b)",
lambda match: (
f"{int(match.group(1))} zu {int(match.group(2))}{match.group(3)}"
),
text,
flags=re.IGNORECASE,
)
def normalize_for_german_speech(text: str) -> str:
"""Turn common visual notation into unambiguous spoken German."""
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
@@ -279,10 +308,14 @@ def normalize_for_german_speech(text: str) -> str:
_spoken_ipv4,
text,
)
# A colon is ambiguous between an aspect ratio and a clock time. Resolve
# ratios first, but only when an explicit format cue is present.
text = _normalize_aspect_ratios(text)
text = re.sub(
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
r"\b([01]?\d|2[0-3]):([0-5]\d)\b(?:\s*Uhr\b)?",
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
text,
flags=re.IGNORECASE,
)
text = re.sub(
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"
@@ -702,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
def synthesize_piper(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
upstream_format = "wav" if output_format == "pcm" else output_format
audio, content_type = _request(
f"{PIPER_URL}/tts",
payload={"text": text, "voice": "alloy", "speed": speed,
"format": upstream_format},
timeout=PIPER_TIMEOUT,
)
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def synthesize_qwen(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
text = prepare_for_qwen_speech(text)
@@ -728,6 +749,47 @@ def synthesize_qwen(text: str, output_format: str,
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def open_qwen_pcm_stream(text: str, chunk_size: int = 4) \
-> tuple[http.client.HTTPConnection, http.client.HTTPResponse]:
"""Open Qwen's native token-level PCM stream without buffering it.
The upstream emits headerless 24 kHz mono signed 16-bit little-endian
PCM. Keeping this response streaming is what lets playback begin while
the remainder of the sentence is still being synthesized.
"""
parsed = urllib.parse.urlparse(QWEN_TTS_URL)
if parsed.scheme != "http" or not parsed.hostname:
raise RuntimeError("QWEN_TTS_URL must be an http URL")
port = parsed.port or 80
prefix = parsed.path.rstrip("/")
payload = json.dumps({
"model": QWEN_TTS_MODEL,
"input": prepare_for_qwen_speech(text),
"voice": QWEN_TTS_VOICE,
"language": QWEN_TTS_LANGUAGE,
"chunk_size": chunk_size,
}, separators=(",", ":")).encode()
connection = http.client.HTTPConnection(
parsed.hostname, port, timeout=QWEN_TTS_TIMEOUT)
try:
connection.request(
"POST",
f"{prefix}/v1/audio/speech/pcm-stream",
body=payload,
headers={"Content-Type": "application/json",
"Accept": "application/octet-stream"},
)
response = connection.getresponse()
if response.status != HTTPStatus.OK:
message = response.read(512).decode(errors="replace")
raise RuntimeError(
f"Qwen PCM stream failed ({response.status}): {message}")
return connection, response
except Exception:
connection.close()
raise
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if acquired:
@@ -737,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
STATE["last_backend"] = "qwen3-tts-1.7b"
STATE["last_error"] = None
return audio
except Exception as exc: # fallback must cover all Qwen failures
except Exception as exc:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = type(exc).__name__
raise
finally:
SYNTHESIS_LOCK.release()
else:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = "queue-timeout"
audio = synthesize_piper(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "piper"
STATE["piper_fallbacks"] += 1
return audio
raise RuntimeError("speech queue timeout")
class Handler(BaseHTTPRequestHandler):
@@ -779,25 +837,26 @@ class Handler(BaseHTTPRequestHandler):
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
primary_ready = _reachable(QWEN_TTS_URL, "/health")
fallback_ready = _reachable(PIPER_URL, "/status")
with STATE_LOCK:
state = dict(STATE)
# This endpoint is also the container liveness check. Qwen3-TTS is
# deliberately stopped in exclusive GPU modes such as Applio, so the
# gateway itself must stay healthy while reporting ready=false.
self.send_json(
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
HTTPStatus.OK,
{
"ready": fallback_ready,
"engine": "qwen3-tts-with-piper-fallback",
"ready": primary_ready,
"engine": "qwen3-tts",
"model": "Qwen3-TTS-12Hz-1.7B-Base",
"voices": [VOICE_ALIAS],
"speaker": QWEN_TTS_VOICE,
"primary_ready": primary_ready,
"fallback_ready": fallback_ready,
**state,
},
)
def do_POST(self) -> None: # noqa: N802
if self.path != "/tts":
if self.path not in {"/tts", "/tts/pcm-stream"}:
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
try:
@@ -810,7 +869,7 @@ class Handler(BaseHTTPRequestHandler):
return
try:
request = json.loads(self.rfile.read(length))
text = request.get("text", "")
text = request.get("input", request.get("text", ""))
voice = request.get("voice", VOICE_ALIAS)
output_format = request.get("format", "mp3")
speed = float(request.get("speed", 1.0))
@@ -829,6 +888,9 @@ class Handler(BaseHTTPRequestHandler):
if not 0.5 <= speed <= 2.0:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
return
if self.path == "/tts/pcm-stream":
self._stream_qwen_pcm(text.strip(), request)
return
started = time.monotonic()
try:
audio, content_type = synthesize(text.strip(), output_format, speed)
@@ -836,13 +898,81 @@ class Handler(BaseHTTPRequestHandler):
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "all local speech backends failed"})
{"error": "local Qwen3-TTS backend failed"})
return
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
f"{time.monotonic() - started:.2f}s")
self.send_bytes(HTTPStatus.OK, audio, content_type)
def _stream_qwen_pcm(self, text: str, request: dict) -> None:
"""Unframe Qwen's PCM frames and relay their audio immediately."""
try:
chunk_size = max(1, min(32, int(request.get("chunk_size", 4))))
except (TypeError, ValueError):
self.send_json(HTTPStatus.BAD_REQUEST,
{"error": "invalid chunk_size"})
return
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if not acquired:
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "speech queue timeout"})
return
connection = None
started = time.monotonic()
headers_sent = False
try:
connection, response = open_qwen_pcm_stream(text, chunk_size)
self.send_response(HTTPStatus.OK)
self.send_header("Content-Type", "application/octet-stream")
self.send_header("Cache-Control", "no-store")
self.send_header("Connection", "close")
self.end_headers()
headers_sent = True
first = True
while True:
frame_header = response.read(4)
if not frame_header:
break
if len(frame_header) != 4:
raise RuntimeError("truncated Qwen PCM frame header")
frame_length = int.from_bytes(frame_header, "big")
if frame_length == 0:
break
if frame_length > MAX_AUDIO_BYTES:
raise RuntimeError("Qwen PCM frame is too large")
remaining = frame_length
while remaining:
chunk = response.read(min(16384, remaining))
if not chunk:
raise RuntimeError("truncated Qwen PCM frame")
if first:
print("tts-gateway: first Qwen PCM chunk in "
f"{time.monotonic() - started:.2f}s")
first = False
self.wfile.write(chunk)
self.wfile.flush()
remaining -= len(chunk)
with STATE_LOCK:
STATE["last_backend"] = "qwen3-tts-1.7b-stream"
STATE["last_error"] = None
except Exception as exc:
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
# Once PCM started, simply close the truncated response. Sending
# JSON into the audio stream would produce loud corrupt samples.
if not headers_sent and not self.wfile.closed:
try:
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "local PCM stream failed"})
except (OSError, BrokenPipeError):
pass
finally:
if connection is not None:
connection.close()
SYNTHESIS_LOCK.release()
self.close_connection = True
if __name__ == "__main__":
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()