From 744a207e5a4c1dffd43c2646738c5df82e887922 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Fri, 4 Sep 2026 09:30:04 +0200 Subject: [PATCH] Stabilize XTTS sentence endings --- platform/docker/tts-gateway/test_tts_gateway.py | 10 ++++++++++ platform/docker/tts-gateway/tts_gateway.py | 15 ++++++++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index eb3eba0..8d6993e 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -155,6 +155,16 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertEqual(len(segments), 1) self.assertIn("dir zu helfen. Klingt das gut.", segments[0][1]) + def test_xtts_receives_a_stable_terminal_stop_cue(self): + self.assertEqual( + gateway._stabilize_xtts_ending("Heute bleibt es trocken."), + "Heute bleibt es trocken;", + ) + self.assertEqual( + gateway._stabilize_xtts_ending("Klingt das gut."), + "Klingt das gut;", + ) + def test_normal_sentence_is_not_broken_into_word_sized_requests(self): text = ( "Die automatische Komprimierung ist in OpenClaw standardmäßig " diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 4ad97a9..ef04587 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -419,10 +419,23 @@ def _speaker_conditioning() -> dict: return SPEAKER_CONDITIONING +def _stabilize_xtts_ending(text: str) -> str: + """Give XTTS a reliable stop cue without changing the visible answer. + + XTTS v2 can continue with invented syllables after a terminal full stop, + especially in German. A terminal semicolon is tokenized as a stronger, + more reliable boundary while retaining neutral sentence intonation. + """ + spoken = text.rstrip() + if spoken.endswith((".", "!", "?", ";", ":")): + spoken = spoken[:-1].rstrip() + return f"{spoken};" + + def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes: payload = { **conditioning, - "text": text, + "text": _stabilize_xtts_ending(text), "language": language, "add_wav_header": True, "stream_chunk_size": "20",