Reduce XTTS tail hallucinations
This commit is contained in:
@@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
self.assertTrue(segments)
|
self.assertTrue(segments)
|
||||||
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
|
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
|
||||||
|
|
||||||
def test_weather_summary_is_split_into_complete_sentences(self):
|
def test_weather_summary_keeps_complete_sentences_together(self):
|
||||||
text = (
|
text = (
|
||||||
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
|
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
|
||||||
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
|
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
|
||||||
@@ -138,13 +138,23 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
segments = gateway.prepare_segments(text)
|
segments = gateway.prepare_segments(text)
|
||||||
spoken = " ".join(part for _, part in segments)
|
spoken = " ".join(part for _, part in segments)
|
||||||
self.assertEqual(len(segments), 5)
|
self.assertEqual(len(segments), 1)
|
||||||
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
|
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
|
||||||
for _, part in segments))
|
for _, part in segments))
|
||||||
self.assertIn("29 Kilometer pro Stunde", spoken)
|
self.assertIn("29 Kilometer pro Stunde", spoken)
|
||||||
self.assertIn("Kein Regen erwartet", spoken)
|
self.assertIn("Kein Regen erwartet", spoken)
|
||||||
self.assertIn("wetteronline Punkt de", spoken)
|
self.assertIn("wetteronline Punkt de", spoken)
|
||||||
|
|
||||||
|
def test_short_followup_sentence_shares_the_same_xtts_request(self):
|
||||||
|
text = (
|
||||||
|
"Ehrlich gesagt habe ich keine echten Gefühle wie Menschen, aber "
|
||||||
|
"ich bin wach, aufmerksam und motiviert, dir zu helfen. "
|
||||||
|
"Klingt das gut?"
|
||||||
|
)
|
||||||
|
segments = gateway.prepare_segments(text)
|
||||||
|
self.assertEqual(len(segments), 1)
|
||||||
|
self.assertIn("dir zu helfen. Klingt das gut.", segments[0][1])
|
||||||
|
|
||||||
def test_normal_sentence_is_not_broken_into_word_sized_requests(self):
|
def test_normal_sentence_is_not_broken_into_word_sized_requests(self):
|
||||||
text = (
|
text = (
|
||||||
"Die automatische Komprimierung ist in OpenClaw standardmäßig "
|
"Die automatische Komprimierung ist in OpenClaw standardmäßig "
|
||||||
|
|||||||
@@ -318,12 +318,12 @@ def clean_for_speech(text: str) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
||||||
"""Return sentence-sized XTTS requests with a conservative hard ceiling.
|
"""Return paragraph-sized XTTS requests with a conservative hard ceiling.
|
||||||
|
|
||||||
A sentence is deliberately never combined with the following sentence.
|
Complete neighbouring sentences are combined while they fit. This avoids
|
||||||
Overlong sentences are split at clause boundaries first and at words only
|
restarting the generative XTTS decoder after every short sentence, which
|
||||||
as a last resort. This preserves XTTS prosody without exposing it to an
|
can create invented tail syllables between sentences. Overlong sentences
|
||||||
unbounded paragraph.
|
are split at clause boundaries first and at words only as a last resort.
|
||||||
"""
|
"""
|
||||||
text = text.strip()
|
text = text.strip()
|
||||||
if not text:
|
if not text:
|
||||||
@@ -331,36 +331,48 @@ def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
|||||||
|
|
||||||
sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
|
sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
|
||||||
chunks: list[str] = []
|
chunks: list[str] = []
|
||||||
|
current = ""
|
||||||
for sentence in sentences:
|
for sentence in sentences:
|
||||||
sentence = sentence.strip()
|
sentence = sentence.strip()
|
||||||
if not sentence:
|
if not sentence:
|
||||||
continue
|
continue
|
||||||
if len(sentence) <= limit:
|
if len(sentence) <= limit:
|
||||||
chunks.append(sentence)
|
candidate = f"{current} {sentence}".strip()
|
||||||
|
if current and len(candidate) > limit:
|
||||||
|
chunks.append(current)
|
||||||
|
current = sentence
|
||||||
|
else:
|
||||||
|
current = candidate
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
if current:
|
||||||
|
chunks.append(current)
|
||||||
|
current = ""
|
||||||
|
|
||||||
# Retain commas in the preceding clause so XTTS can reproduce the
|
# Retain commas in the preceding clause so XTTS can reproduce the
|
||||||
# intended pause. Semicolons and colons were normalized earlier.
|
# intended pause. Semicolons and colons were normalized earlier.
|
||||||
clauses = re.split(r"(?<=,)\s+", sentence)
|
clauses = re.split(r"(?<=,)\s+", sentence)
|
||||||
current = ""
|
long_current = ""
|
||||||
for clause in clauses:
|
for clause in clauses:
|
||||||
clause = clause.strip()
|
clause = clause.strip()
|
||||||
candidate = f"{current} {clause}".strip()
|
candidate = f"{long_current} {clause}".strip()
|
||||||
if current and len(candidate) > limit:
|
if long_current and len(candidate) > limit:
|
||||||
chunks.append(current)
|
chunks.append(long_current)
|
||||||
current = ""
|
long_current = ""
|
||||||
if len(clause) <= limit:
|
if len(clause) <= limit:
|
||||||
current = f"{current} {clause}".strip()
|
long_current = f"{long_current} {clause}".strip()
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# A clause without a usable pause can still exceed the ceiling.
|
# A clause without a usable pause can still exceed the ceiling.
|
||||||
for word in clause.split():
|
for word in clause.split():
|
||||||
candidate = f"{current} {word}".strip()
|
candidate = f"{long_current} {word}".strip()
|
||||||
if current and len(candidate) > limit:
|
if long_current and len(candidate) > limit:
|
||||||
chunks.append(current)
|
chunks.append(long_current)
|
||||||
current = word
|
long_current = word
|
||||||
else:
|
else:
|
||||||
current = candidate
|
long_current = candidate
|
||||||
|
if long_current:
|
||||||
|
chunks.append(long_current)
|
||||||
if current:
|
if current:
|
||||||
chunks.append(current)
|
chunks.append(current)
|
||||||
return chunks
|
return chunks
|
||||||
|
|||||||
Reference in New Issue
Block a user