diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 414fbc9..04ebcdc 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -120,6 +120,11 @@ class LanguageSegmentationTests(unittest.TestCase): ) self.assertNotRegex(spoken, r"[:;\-‐‑‒–—−]") + def test_question_and_exclamation_marks_are_not_sent_to_xtts(self): + spoken = gateway.clean_for_speech("Wie geht es dir? Wirklich gut!") + self.assertEqual(spoken, "Wie geht es dir. Wirklich gut.") + self.assertNotRegex(spoken, r"[!?]") + def test_punctuation_only_segments_are_never_synthesized(self): segments = gateway.prepare_segments("Status: – alles läuft.") self.assertTrue(segments) diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 9e9251f..55a963d 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -298,6 +298,10 @@ def clean_for_speech(text: str) -> str: text = re.sub(r"(?<=\w)[\-‐‑‒–—−](?=\w)", " ", text) text = re.sub(r"\s+[\-‐‑‒–—−]\s+", ", ", text) text = re.sub(r"\s*[:;]+\s*", ", ", text) + # XTTS can pronounce literal question/exclamation glyphs as short + # nonsense syllables (for example "?" as "nau"). Retain a sentence + # boundary for pacing, but never pass those glyphs to the model. + text = re.sub(r"\s*[!?]+\s*", ". ", text) text = "".join( char for char in text if unicodedata.category(char) not in {"So", "Cs"}