diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 2169c63..eb3eba0 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertTrue(segments) self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments)) - def test_weather_summary_is_split_into_complete_sentences(self): + def test_weather_summary_keeps_complete_sentences_together(self): text = ( "Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. " "Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. " @@ -138,13 +138,23 @@ class LanguageSegmentationTests(unittest.TestCase): ) segments = gateway.prepare_segments(text) spoken = " ".join(part for _, part in segments) - self.assertEqual(len(segments), 5) + self.assertEqual(len(segments), 1) self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS for _, part in segments)) self.assertIn("29 Kilometer pro Stunde", spoken) self.assertIn("Kein Regen erwartet", spoken) self.assertIn("wetteronline Punkt de", spoken) + def test_short_followup_sentence_shares_the_same_xtts_request(self): + text = ( + "Ehrlich gesagt habe ich keine echten Gefühle wie Menschen, aber " + "ich bin wach, aufmerksam und motiviert, dir zu helfen. " + "Klingt das gut?" + ) + segments = gateway.prepare_segments(text) + self.assertEqual(len(segments), 1) + self.assertIn("dir zu helfen. Klingt das gut.", segments[0][1]) + def test_normal_sentence_is_not_broken_into_word_sized_requests(self): text = ( "Die automatische Komprimierung ist in OpenClaw standardmäßig " diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index a59498c..4ad97a9 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -318,12 +318,12 @@ def clean_for_speech(text: str) -> str: def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]: - """Return sentence-sized XTTS requests with a conservative hard ceiling. + """Return paragraph-sized XTTS requests with a conservative hard ceiling. - A sentence is deliberately never combined with the following sentence. - Overlong sentences are split at clause boundaries first and at words only - as a last resort. This preserves XTTS prosody without exposing it to an - unbounded paragraph. + Complete neighbouring sentences are combined while they fit. This avoids + restarting the generative XTTS decoder after every short sentence, which + can create invented tail syllables between sentences. Overlong sentences + are split at clause boundaries first and at words only as a last resort. """ text = text.strip() if not text: @@ -331,38 +331,50 @@ def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]: sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text) chunks: list[str] = [] + current = "" for sentence in sentences: sentence = sentence.strip() if not sentence: continue if len(sentence) <= limit: - chunks.append(sentence) + candidate = f"{current} {sentence}".strip() + if current and len(candidate) > limit: + chunks.append(current) + current = sentence + else: + current = candidate continue + if current: + chunks.append(current) + current = "" + # Retain commas in the preceding clause so XTTS can reproduce the # intended pause. Semicolons and colons were normalized earlier. clauses = re.split(r"(?<=,)\s+", sentence) - current = "" + long_current = "" for clause in clauses: clause = clause.strip() - candidate = f"{current} {clause}".strip() - if current and len(candidate) > limit: - chunks.append(current) - current = "" + candidate = f"{long_current} {clause}".strip() + if long_current and len(candidate) > limit: + chunks.append(long_current) + long_current = "" if len(clause) <= limit: - current = f"{current} {clause}".strip() + long_current = f"{long_current} {clause}".strip() continue # A clause without a usable pause can still exceed the ceiling. for word in clause.split(): - candidate = f"{current} {word}".strip() - if current and len(candidate) > limit: - chunks.append(current) - current = word + candidate = f"{long_current} {word}".strip() + if long_current and len(candidate) > limit: + chunks.append(long_current) + long_current = word else: - current = candidate - if current: - chunks.append(current) + long_current = candidate + if long_current: + chunks.append(long_current) + if current: + chunks.append(current) return chunks