Normalize question marks before local TTS

This commit is contained in:
Mikei386 committed 2026-09-03 23:24:15 +02:00
1 parent 023c2ee40d
commit 384d81f6cd
2 files changed
+9

No files matched your search

@@ -120,6 +120,11 @@ class LanguageSegmentationTests(unittest.TestCase):
) )
self.assertNotRegex(spoken, r"[:;\-‐‑‒–—−]") self.assertNotRegex(spoken, r"[:;\-‐‑‒–—−]")
def test_question_and_exclamation_marks_are_not_sent_to_xtts(self):
spoken = gateway.clean_for_speech("Wie geht es dir? Wirklich gut!")
self.assertEqual(spoken, "Wie geht es dir. Wirklich gut.")
self.assertNotRegex(spoken, r"[!?]")
def test_punctuation_only_segments_are_never_synthesized(self): def test_punctuation_only_segments_are_never_synthesized(self):
segments = gateway.prepare_segments("Status: – alles läuft.") segments = gateway.prepare_segments("Status: – alles läuft.")
self.assertTrue(segments) self.assertTrue(segments)
@@ -298,6 +298,10 @@ def clean_for_speech(text: str) -> str:
text = re.sub(r"(?<=\w)[\-‐‑‒–—−](?=\w)", " ", text) text = re.sub(r"(?<=\w)[\-‐‑‒–—−](?=\w)", " ", text)
text = re.sub(r"\s+[\-‐‑‒–—−]\s+", ", ", text) text = re.sub(r"\s+[\-‐‑‒–—−]\s+", ", ", text)
text = re.sub(r"\s*[:;]+\s*", ", ", text) text = re.sub(r"\s*[:;]+\s*", ", ", text)
# XTTS can pronounce literal question/exclamation glyphs as short
# nonsense syllables (for example "?" as "nau"). Retain a sentence
# boundary for pacing, but never pass those glyphs to the model.
text = re.sub(r"\s*[!?]+\s*", ". ", text)
text = "".join( text = "".join(
char for char in text char for char in text
if unicodedata.category(char) not in {"So", "Cs"} if unicodedata.category(char) not in {"So", "Cs"}