diff --git a/compose.yaml b/compose.yaml index 92c2a07..43490e5 100644 --- a/compose.yaml +++ b/compose.yaml @@ -697,9 +697,9 @@ services: TTS_CODE_SWITCH_ENABLED: "false" XTTS_QUEUE_TIMEOUT: "15" XTTS_TIMEOUT: "120" - # Short sentence-sized requests avoid long generated silences and - # truncated weather/status summaries with Annmarie Nele. - XTTS_CHUNK_CHARS: "60" + # Keep normal sentences intact for natural prosody. This is only the + # safety ceiling for unusually long sentences. + XTTS_CHUNK_CHARS: "420" # XTTS occasionally inserts multi-second silence inside a phrase. XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650" XTTS_INTERNAL_SILENCE_KEEP_MS: "220" diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index c78bb8a..2169c63 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertTrue(segments) self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments)) - def test_weather_summary_is_split_into_short_complete_chunks(self): + def test_weather_summary_is_split_into_complete_sentences(self): text = ( "Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. " "Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. " @@ -138,12 +138,29 @@ class LanguageSegmentationTests(unittest.TestCase): ) segments = gateway.prepare_segments(text) spoken = " ".join(part for _, part in segments) - self.assertEqual(len(segments), 4) - self.assertTrue(all(len(part) <= 60 for _, part in segments)) + self.assertEqual(len(segments), 5) + self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS + for _, part in segments)) self.assertIn("29 Kilometer pro Stunde", spoken) self.assertIn("Kein Regen erwartet", spoken) self.assertIn("wetteronline Punkt de", spoken) + def test_normal_sentence_is_not_broken_into_word_sized_requests(self): + text = ( + "Die automatische Komprimierung ist in OpenClaw standardmäßig " + "aktiviert und lässt sich über die Konfigurationsdatei steuern." + ) + self.assertEqual(gateway.prepare_segments(text), [("de", text)]) + + def test_overlong_sentence_prefers_clause_boundaries(self): + clause = "dieser natürlich gesprochene Teilsatz bleibt zusammen," + text = " ".join([clause] * 12) + " und endet hier." + segments = gateway.prepare_segments(text) + self.assertGreater(len(segments), 1) + self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS + for _, part in segments)) + self.assertTrue(all(len(part.split()) > 4 for _, part in segments)) + class FallbackTests(unittest.TestCase): def setUp(self): diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 2ef5ae5..a59498c 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -38,10 +38,10 @@ MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024))) XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120")) PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120")) QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15")) -# XTTS can insert multi-second silences or truncate the remainder when a -# moderately long German paragraph is sent as one request. Keeping requests -# close to one sentence proved substantially more stable with Annmarie Nele. -CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "60")) +# XTTS loses natural prosody when a sentence is synthesized as many tiny +# requests: every request starts a fresh utterance. Keep complete sentences +# together and use this only as a safety ceiling for unusually long sentences. +CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "420")) TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90")) TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18")) CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8")) @@ -318,38 +318,51 @@ def clean_for_speech(text: str) -> str: def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]: - """Split at natural pauses and keep every XTTS request comfortably short.""" + """Return sentence-sized XTTS requests with a conservative hard ceiling. + + A sentence is deliberately never combined with the following sentence. + Overlong sentences are split at clause boundaries first and at words only + as a last resort. This preserves XTTS prosody without exposing it to an + unbounded paragraph. + """ text = text.strip() if not text: return [] - if len(text) <= limit: - return [text] - pieces = re.split(r"(?<=[.!?;:])\s+|\s+(?=\d+[.)]\s)", text) + sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text) chunks: list[str] = [] - current = "" - for piece in pieces: - piece = piece.strip() - if not piece: + for sentence in sentences: + sentence = sentence.strip() + if not sentence: continue - if len(piece) > limit: - words = piece.split() - for word in words: + if len(sentence) <= limit: + chunks.append(sentence) + continue + + # Retain commas in the preceding clause so XTTS can reproduce the + # intended pause. Semicolons and colons were normalized earlier. + clauses = re.split(r"(?<=,)\s+", sentence) + current = "" + for clause in clauses: + clause = clause.strip() + candidate = f"{current} {clause}".strip() + if current and len(candidate) > limit: + chunks.append(current) + current = "" + if len(clause) <= limit: + current = f"{current} {clause}".strip() + continue + + # A clause without a usable pause can still exceed the ceiling. + for word in clause.split(): candidate = f"{current} {word}".strip() if current and len(candidate) > limit: chunks.append(current) current = word else: current = candidate - continue - candidate = f"{current} {piece}".strip() - if current and len(candidate) > limit: + if current: chunks.append(current) - current = piece - else: - current = candidate - if current: - chunks.append(current) return chunks @@ -366,6 +379,14 @@ def prepare_segments(text: str) -> list[tuple[str, str]]: if language == "en": spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment) for chunk in _split_chunk(spoken): + if not re.search(r"\w", chunk, flags=re.UNICODE): + if prepared: + previous_language, previous_text = prepared[-1] + prepared[-1] = ( + previous_language, + previous_text.rstrip() + chunk.strip(), + ) + continue prepared.append((language, chunk)) return prepared