Preserve sentence prosody in local TTS

This commit is contained in:
Mikei386
2026-09-04 08:11:05 +02:00
parent 5afdf46a7c
commit be8a654f1e
3 changed files with 67 additions and 29 deletions
+3 -3
View File
@@ -697,9 +697,9 @@ services:
TTS_CODE_SWITCH_ENABLED: "false" TTS_CODE_SWITCH_ENABLED: "false"
XTTS_QUEUE_TIMEOUT: "15" XTTS_QUEUE_TIMEOUT: "15"
XTTS_TIMEOUT: "120" XTTS_TIMEOUT: "120"
# Short sentence-sized requests avoid long generated silences and # Keep normal sentences intact for natural prosody. This is only the
# truncated weather/status summaries with Annmarie Nele. # safety ceiling for unusually long sentences.
XTTS_CHUNK_CHARS: "60" XTTS_CHUNK_CHARS: "420"
# XTTS occasionally inserts multi-second silence inside a phrase. # XTTS occasionally inserts multi-second silence inside a phrase.
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650" XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
XTTS_INTERNAL_SILENCE_KEEP_MS: "220" XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
@@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertTrue(segments) self.assertTrue(segments)
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments)) self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
def test_weather_summary_is_split_into_short_complete_chunks(self): def test_weather_summary_is_split_into_complete_sentences(self):
text = ( text = (
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. " "Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. " "Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
@@ -138,12 +138,29 @@ class LanguageSegmentationTests(unittest.TestCase):
) )
segments = gateway.prepare_segments(text) segments = gateway.prepare_segments(text)
spoken = " ".join(part for _, part in segments) spoken = " ".join(part for _, part in segments)
self.assertEqual(len(segments), 4) self.assertEqual(len(segments), 5)
self.assertTrue(all(len(part) <= 60 for _, part in segments)) self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
for _, part in segments))
self.assertIn("29 Kilometer pro Stunde", spoken) self.assertIn("29 Kilometer pro Stunde", spoken)
self.assertIn("Kein Regen erwartet", spoken) self.assertIn("Kein Regen erwartet", spoken)
self.assertIn("wetteronline Punkt de", spoken) self.assertIn("wetteronline Punkt de", spoken)
def test_normal_sentence_is_not_broken_into_word_sized_requests(self):
text = (
"Die automatische Komprimierung ist in OpenClaw standardmäßig "
"aktiviert und lässt sich über die Konfigurationsdatei steuern."
)
self.assertEqual(gateway.prepare_segments(text), [("de", text)])
def test_overlong_sentence_prefers_clause_boundaries(self):
clause = "dieser natürlich gesprochene Teilsatz bleibt zusammen,"
text = " ".join([clause] * 12) + " und endet hier."
segments = gateway.prepare_segments(text)
self.assertGreater(len(segments), 1)
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
for _, part in segments))
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
class FallbackTests(unittest.TestCase): class FallbackTests(unittest.TestCase):
def setUp(self): def setUp(self):
+43 -22
View File
@@ -38,10 +38,10 @@ MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120")) XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120")) PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15")) QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
# XTTS can insert multi-second silences or truncate the remainder when a # XTTS loses natural prosody when a sentence is synthesized as many tiny
# moderately long German paragraph is sent as one request. Keeping requests # requests: every request starts a fresh utterance. Keep complete sentences
# close to one sentence proved substantially more stable with Annmarie Nele. # together and use this only as a safety ceiling for unusually long sentences.
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "60")) CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "420"))
TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90")) TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18")) TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8")) CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
@@ -318,36 +318,49 @@ def clean_for_speech(text: str) -> str:
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]: def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
"""Split at natural pauses and keep every XTTS request comfortably short.""" """Return sentence-sized XTTS requests with a conservative hard ceiling.
A sentence is deliberately never combined with the following sentence.
Overlong sentences are split at clause boundaries first and at words only
as a last resort. This preserves XTTS prosody without exposing it to an
unbounded paragraph.
"""
text = text.strip() text = text.strip()
if not text: if not text:
return [] return []
if len(text) <= limit:
return [text]
pieces = re.split(r"(?<=[.!?;:])\s+|\s+(?=\d+[.)]\s)", text) sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
chunks: list[str] = [] chunks: list[str] = []
current = "" for sentence in sentences:
for piece in pieces: sentence = sentence.strip()
piece = piece.strip() if not sentence:
if not piece:
continue continue
if len(piece) > limit: if len(sentence) <= limit:
words = piece.split() chunks.append(sentence)
for word in words: continue
# Retain commas in the preceding clause so XTTS can reproduce the
# intended pause. Semicolons and colons were normalized earlier.
clauses = re.split(r"(?<=,)\s+", sentence)
current = ""
for clause in clauses:
clause = clause.strip()
candidate = f"{current} {clause}".strip()
if current and len(candidate) > limit:
chunks.append(current)
current = ""
if len(clause) <= limit:
current = f"{current} {clause}".strip()
continue
# A clause without a usable pause can still exceed the ceiling.
for word in clause.split():
candidate = f"{current} {word}".strip() candidate = f"{current} {word}".strip()
if current and len(candidate) > limit: if current and len(candidate) > limit:
chunks.append(current) chunks.append(current)
current = word current = word
else: else:
current = candidate current = candidate
continue
candidate = f"{current} {piece}".strip()
if current and len(candidate) > limit:
chunks.append(current)
current = piece
else:
current = candidate
if current: if current:
chunks.append(current) chunks.append(current)
return chunks return chunks
@@ -366,6 +379,14 @@ def prepare_segments(text: str) -> list[tuple[str, str]]:
if language == "en": if language == "en":
spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment) spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment)
for chunk in _split_chunk(spoken): for chunk in _split_chunk(spoken):
if not re.search(r"\w", chunk, flags=re.UNICODE):
if prepared:
previous_language, previous_text = prepared[-1]
prepared[-1] = (
previous_language,
previous_text.rstrip() + chunk.strip(),
)
continue
prepared.append((language, chunk)) prepared.append((language, chunk))
return prepared return prepared