Preserve sentence prosody in local TTS
This commit is contained in:
+3
-3
@@ -697,9 +697,9 @@ services:
|
|||||||
TTS_CODE_SWITCH_ENABLED: "false"
|
TTS_CODE_SWITCH_ENABLED: "false"
|
||||||
XTTS_QUEUE_TIMEOUT: "15"
|
XTTS_QUEUE_TIMEOUT: "15"
|
||||||
XTTS_TIMEOUT: "120"
|
XTTS_TIMEOUT: "120"
|
||||||
# Short sentence-sized requests avoid long generated silences and
|
# Keep normal sentences intact for natural prosody. This is only the
|
||||||
# truncated weather/status summaries with Annmarie Nele.
|
# safety ceiling for unusually long sentences.
|
||||||
XTTS_CHUNK_CHARS: "60"
|
XTTS_CHUNK_CHARS: "420"
|
||||||
# XTTS occasionally inserts multi-second silence inside a phrase.
|
# XTTS occasionally inserts multi-second silence inside a phrase.
|
||||||
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
|
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
|
||||||
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
|
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
|
||||||
|
|||||||
@@ -130,7 +130,7 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
self.assertTrue(segments)
|
self.assertTrue(segments)
|
||||||
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
|
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
|
||||||
|
|
||||||
def test_weather_summary_is_split_into_short_complete_chunks(self):
|
def test_weather_summary_is_split_into_complete_sentences(self):
|
||||||
text = (
|
text = (
|
||||||
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
|
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
|
||||||
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
|
"Abends wolkiger, 16 bis 21 Grad. Böen bis 29 km/h. "
|
||||||
@@ -138,12 +138,29 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
segments = gateway.prepare_segments(text)
|
segments = gateway.prepare_segments(text)
|
||||||
spoken = " ".join(part for _, part in segments)
|
spoken = " ".join(part for _, part in segments)
|
||||||
self.assertEqual(len(segments), 4)
|
self.assertEqual(len(segments), 5)
|
||||||
self.assertTrue(all(len(part) <= 60 for _, part in segments))
|
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
|
||||||
|
for _, part in segments))
|
||||||
self.assertIn("29 Kilometer pro Stunde", spoken)
|
self.assertIn("29 Kilometer pro Stunde", spoken)
|
||||||
self.assertIn("Kein Regen erwartet", spoken)
|
self.assertIn("Kein Regen erwartet", spoken)
|
||||||
self.assertIn("wetteronline Punkt de", spoken)
|
self.assertIn("wetteronline Punkt de", spoken)
|
||||||
|
|
||||||
|
def test_normal_sentence_is_not_broken_into_word_sized_requests(self):
|
||||||
|
text = (
|
||||||
|
"Die automatische Komprimierung ist in OpenClaw standardmäßig "
|
||||||
|
"aktiviert und lässt sich über die Konfigurationsdatei steuern."
|
||||||
|
)
|
||||||
|
self.assertEqual(gateway.prepare_segments(text), [("de", text)])
|
||||||
|
|
||||||
|
def test_overlong_sentence_prefers_clause_boundaries(self):
|
||||||
|
clause = "dieser natürlich gesprochene Teilsatz bleibt zusammen,"
|
||||||
|
text = " ".join([clause] * 12) + " und endet hier."
|
||||||
|
segments = gateway.prepare_segments(text)
|
||||||
|
self.assertGreater(len(segments), 1)
|
||||||
|
self.assertTrue(all(len(part) <= gateway.CHUNK_CHARS
|
||||||
|
for _, part in segments))
|
||||||
|
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
|
||||||
|
|
||||||
|
|
||||||
class FallbackTests(unittest.TestCase):
|
class FallbackTests(unittest.TestCase):
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
|
|||||||
@@ -38,10 +38,10 @@ MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
|||||||
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||||
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||||
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||||
# XTTS can insert multi-second silences or truncate the remainder when a
|
# XTTS loses natural prosody when a sentence is synthesized as many tiny
|
||||||
# moderately long German paragraph is sent as one request. Keeping requests
|
# requests: every request starts a fresh utterance. Keep complete sentences
|
||||||
# close to one sentence proved substantially more stable with Annmarie Nele.
|
# together and use this only as a safety ceiling for unusually long sentences.
|
||||||
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "60"))
|
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "420"))
|
||||||
TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
|
TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
|
||||||
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
|
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
|
||||||
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
|
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
|
||||||
@@ -318,36 +318,49 @@ def clean_for_speech(text: str) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
||||||
"""Split at natural pauses and keep every XTTS request comfortably short."""
|
"""Return sentence-sized XTTS requests with a conservative hard ceiling.
|
||||||
|
|
||||||
|
A sentence is deliberately never combined with the following sentence.
|
||||||
|
Overlong sentences are split at clause boundaries first and at words only
|
||||||
|
as a last resort. This preserves XTTS prosody without exposing it to an
|
||||||
|
unbounded paragraph.
|
||||||
|
"""
|
||||||
text = text.strip()
|
text = text.strip()
|
||||||
if not text:
|
if not text:
|
||||||
return []
|
return []
|
||||||
if len(text) <= limit:
|
|
||||||
return [text]
|
|
||||||
|
|
||||||
pieces = re.split(r"(?<=[.!?;:])\s+|\s+(?=\d+[.)]\s)", text)
|
sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
|
||||||
chunks: list[str] = []
|
chunks: list[str] = []
|
||||||
current = ""
|
for sentence in sentences:
|
||||||
for piece in pieces:
|
sentence = sentence.strip()
|
||||||
piece = piece.strip()
|
if not sentence:
|
||||||
if not piece:
|
|
||||||
continue
|
continue
|
||||||
if len(piece) > limit:
|
if len(sentence) <= limit:
|
||||||
words = piece.split()
|
chunks.append(sentence)
|
||||||
for word in words:
|
continue
|
||||||
|
|
||||||
|
# Retain commas in the preceding clause so XTTS can reproduce the
|
||||||
|
# intended pause. Semicolons and colons were normalized earlier.
|
||||||
|
clauses = re.split(r"(?<=,)\s+", sentence)
|
||||||
|
current = ""
|
||||||
|
for clause in clauses:
|
||||||
|
clause = clause.strip()
|
||||||
|
candidate = f"{current} {clause}".strip()
|
||||||
|
if current and len(candidate) > limit:
|
||||||
|
chunks.append(current)
|
||||||
|
current = ""
|
||||||
|
if len(clause) <= limit:
|
||||||
|
current = f"{current} {clause}".strip()
|
||||||
|
continue
|
||||||
|
|
||||||
|
# A clause without a usable pause can still exceed the ceiling.
|
||||||
|
for word in clause.split():
|
||||||
candidate = f"{current} {word}".strip()
|
candidate = f"{current} {word}".strip()
|
||||||
if current and len(candidate) > limit:
|
if current and len(candidate) > limit:
|
||||||
chunks.append(current)
|
chunks.append(current)
|
||||||
current = word
|
current = word
|
||||||
else:
|
else:
|
||||||
current = candidate
|
current = candidate
|
||||||
continue
|
|
||||||
candidate = f"{current} {piece}".strip()
|
|
||||||
if current and len(candidate) > limit:
|
|
||||||
chunks.append(current)
|
|
||||||
current = piece
|
|
||||||
else:
|
|
||||||
current = candidate
|
|
||||||
if current:
|
if current:
|
||||||
chunks.append(current)
|
chunks.append(current)
|
||||||
return chunks
|
return chunks
|
||||||
@@ -366,6 +379,14 @@ def prepare_segments(text: str) -> list[tuple[str, str]]:
|
|||||||
if language == "en":
|
if language == "en":
|
||||||
spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment)
|
spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment)
|
||||||
for chunk in _split_chunk(spoken):
|
for chunk in _split_chunk(spoken):
|
||||||
|
if not re.search(r"\w", chunk, flags=re.UNICODE):
|
||||||
|
if prepared:
|
||||||
|
previous_language, previous_text = prepared[-1]
|
||||||
|
prepared[-1] = (
|
||||||
|
previous_language,
|
||||||
|
previous_text.rstrip() + chunk.strip(),
|
||||||
|
)
|
||||||
|
continue
|
||||||
prepared.append((language, chunk))
|
prepared.append((language, chunk))
|
||||||
return prepared
|
return prepared
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user