Reduce XTTS tail hallucinations

This commit is contained in:
Mikei386
2026-09-04 08:32:45 +02:00
parent be8a654f1e
commit 8dac735680
2 changed files with 43 additions and 21 deletions
@@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertTrue(segments) self.assertTrue(segments)
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments)) self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
def test_weather_summary_is_split_into_complete_sentences(self): def test_weather_summary_keeps_complete_sentences_together(self):
text = ( text = (
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. " "Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. " "Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
@@ -138,13 +138,23 @@ class LanguageSegmentationTests(unittest.TestCase):
) )
segments = gateway.prepare_segments(text) segments = gateway.prepare_segments(text)
spoken = " ".join(part for _, part in segments) spoken = " ".join(part for _, part in segments)
self.assertEqual(len(segments), 5) self.assertEqual(len(segments), 1)
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
for _, part in segments)) for _, part in segments))
self.assertIn("29 Kilometer pro Stunde", spoken) self.assertIn("29 Kilometer pro Stunde", spoken)
self.assertIn("Kein Regen erwartet", spoken) self.assertIn("Kein Regen erwartet", spoken)
self.assertIn("wetteronline Punkt de", spoken) self.assertIn("wetteronline Punkt de", spoken)
def test_short_followup_sentence_shares_the_same_xtts_request(self):
text = (
"Ehrlich gesagt habe ich keine echten Gefühle wie Menschen, aber "
"ich bin wach, aufmerksam und motiviert, dir zu helfen. "
"Klingt das gut?"
)
segments = gateway.prepare_segments(text)
self.assertEqual(len(segments), 1)
self.assertIn("dir zu helfen. Klingt das gut.", segments[0][1])
def test_normal_sentence_is_not_broken_into_word_sized_requests(self): def test_normal_sentence_is_not_broken_into_word_sized_requests(self):
text = ( text = (
"Die automatische Komprimierung ist in OpenClaw standardmäßig " "Die automatische Komprimierung ist in OpenClaw standardmäßig "
+29 -17
View File
@@ -318,12 +318,12 @@ def clean_for_speech(text: str) -> str:
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]: def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
"""Return sentence-sized XTTS requests with a conservative hard ceiling. """Return paragraph-sized XTTS requests with a conservative hard ceiling.
A sentence is deliberately never combined with the following sentence. Complete neighbouring sentences are combined while they fit. This avoids
Overlong sentences are split at clause boundaries first and at words only restarting the generative XTTS decoder after every short sentence, which
as a last resort. This preserves XTTS prosody without exposing it to an can create invented tail syllables between sentences. Overlong sentences
unbounded paragraph. are split at clause boundaries first and at words only as a last resort.
""" """
text = text.strip() text = text.strip()
if not text: if not text:
@@ -331,36 +331,48 @@ def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text) sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
chunks: list[str] = [] chunks: list[str] = []
current = ""
for sentence in sentences: for sentence in sentences:
sentence = sentence.strip() sentence = sentence.strip()
if not sentence: if not sentence:
continue continue
if len(sentence) <= limit: if len(sentence) <= limit:
chunks.append(sentence) candidate = f"{current} {sentence}".strip()
if current and len(candidate) > limit:
chunks.append(current)
current = sentence
else:
current = candidate
continue continue
if current:
chunks.append(current)
current = ""
# Retain commas in the preceding clause so XTTS can reproduce the # Retain commas in the preceding clause so XTTS can reproduce the
# intended pause. Semicolons and colons were normalized earlier. # intended pause. Semicolons and colons were normalized earlier.
clauses = re.split(r"(?<=,)\s+", sentence) clauses = re.split(r"(?<=,)\s+", sentence)
current = "" long_current = ""
for clause in clauses: for clause in clauses:
clause = clause.strip() clause = clause.strip()
candidate = f"{current} {clause}".strip() candidate = f"{long_current} {clause}".strip()
if current and len(candidate) > limit: if long_current and len(candidate) > limit:
chunks.append(current) chunks.append(long_current)
current = "" long_current = ""
if len(clause) <= limit: if len(clause) <= limit:
current = f"{current} {clause}".strip() long_current = f"{long_current} {clause}".strip()
continue continue
# A clause without a usable pause can still exceed the ceiling. # A clause without a usable pause can still exceed the ceiling.
for word in clause.split(): for word in clause.split():
candidate = f"{current} {word}".strip() candidate = f"{long_current} {word}".strip()
if current and len(candidate) > limit: if long_current and len(candidate) > limit:
chunks.append(current) chunks.append(long_current)
current = word long_current = word
else: else:
current = candidate long_current = candidate
if long_current:
chunks.append(long_current)
if current: if current:
chunks.append(current) chunks.append(current)
return chunks return chunks