Remove Beta 1 and Piper fallback
This commit is contained in:
1 parent
52482627be
commit
08ff3d7c4e
38 files changed
+94
-484
No files matched your search
@@ -244,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
|
||||
|
||||
|
||||
class FallbackTests(unittest.TestCase):
|
||||
class BackendFailureTests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.original_xtts = gateway.synthesize_xtts
|
||||
self.original_piper = gateway.synthesize_piper
|
||||
self.original_qwen = gateway.synthesize_qwen
|
||||
|
||||
def tearDown(self):
|
||||
gateway.synthesize_xtts = self.original_xtts
|
||||
gateway.synthesize_piper = self.original_piper
|
||||
gateway.synthesize_qwen = self.original_qwen
|
||||
|
||||
def test_piper_is_used_when_xtts_fails(self):
|
||||
def test_qwen_failure_is_reported_without_fallback(self):
|
||||
def fail(*_args):
|
||||
raise RuntimeError("synthetic XTTS failure")
|
||||
raise RuntimeError("synthetic Qwen failure")
|
||||
|
||||
gateway.synthesize_xtts = fail
|
||||
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
|
||||
self.assertEqual(
|
||||
gateway.synthesize("synthetic test", "wav", 1.0),
|
||||
(b"piper", "audio/wav"),
|
||||
)
|
||||
gateway.synthesize_qwen = fail
|
||||
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
|
||||
gateway.synthesize("synthetic test", "wav", 1.0)
|
||||
|
||||
|
||||
class AudioJoinTests(unittest.TestCase):
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private Qwen3-TTS-first gateway with a Piper fallback.
|
||||
"""Private Qwen3-TTS gateway.
|
||||
|
||||
The gateway implements the narrow /status and /tts protocol already consumed
|
||||
by the profile router. Request text is never logged or persisted.
|
||||
@@ -33,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
|
||||
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
|
||||
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
|
||||
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
|
||||
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
|
||||
@@ -43,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
|
||||
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
||||
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||
# XTTS loses natural prosody when a sentence is synthesized as many tiny
|
||||
# requests: every request starts a fresh utterance. Keep complete sentences
|
||||
@@ -63,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
|
||||
SPEAKER_CONDITIONING: dict | None = None
|
||||
STATE = {
|
||||
"last_backend": None,
|
||||
"xtts_failures": 0,
|
||||
"piper_fallbacks": 0,
|
||||
"qwen_failures": 0,
|
||||
"last_error": None,
|
||||
}
|
||||
|
||||
@@ -738,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
|
||||
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
|
||||
|
||||
|
||||
def synthesize_piper(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||
audio, content_type = _request(
|
||||
f"{PIPER_URL}/tts",
|
||||
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||
"format": upstream_format},
|
||||
timeout=PIPER_TIMEOUT,
|
||||
)
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def synthesize_qwen(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
text = prepare_for_qwen_speech(text)
|
||||
@@ -814,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
|
||||
STATE["last_backend"] = "qwen3-tts-1.7b"
|
||||
STATE["last_error"] = None
|
||||
return audio
|
||||
except Exception as exc: # fallback must cover all Qwen failures
|
||||
except Exception as exc:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
raise
|
||||
finally:
|
||||
SYNTHESIS_LOCK.release()
|
||||
else:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = "queue-timeout"
|
||||
|
||||
audio = synthesize_piper(text, output_format, speed)
|
||||
with STATE_LOCK:
|
||||
STATE["last_backend"] = "piper"
|
||||
STATE["piper_fallbacks"] += 1
|
||||
return audio
|
||||
raise RuntimeError("speech queue timeout")
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
@@ -856,19 +837,20 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
primary_ready = _reachable(QWEN_TTS_URL, "/health")
|
||||
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||
with STATE_LOCK:
|
||||
state = dict(STATE)
|
||||
# This endpoint is also the container liveness check. Qwen3-TTS is
|
||||
# deliberately stopped in exclusive GPU modes such as Applio, so the
|
||||
# gateway itself must stay healthy while reporting ready=false.
|
||||
self.send_json(
|
||||
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": fallback_ready,
|
||||
"engine": "qwen3-tts-with-piper-fallback",
|
||||
"ready": primary_ready,
|
||||
"engine": "qwen3-tts",
|
||||
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"voices": [VOICE_ALIAS],
|
||||
"speaker": QWEN_TTS_VOICE,
|
||||
"primary_ready": primary_ready,
|
||||
"fallback_ready": fallback_ready,
|
||||
**state,
|
||||
},
|
||||
)
|
||||
@@ -916,7 +898,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
with STATE_LOCK:
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{"error": "all local speech backends failed"})
|
||||
{"error": "local Qwen3-TTS backend failed"})
|
||||
return
|
||||
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
|
||||
f"{time.monotonic() - started:.2f}s")
|
||||
@@ -992,5 +974,5 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
Reference in new issue
Block a user