From dec4709ae906833c0d67ab6214e42bc02c3ae079 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Mon, 24 Aug 2026 19:26:07 +0200 Subject: [PATCH] Normalize punctuation before German TTS --- platform/docker/tts-gateway/test_tts_gateway.py | 15 +++++++++++++++ platform/docker/tts-gateway/tts_gateway.py | 10 ++++++++-- 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 721cac2..414fbc9 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -110,6 +110,21 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertIn("29 Kilometer pro Stunde", spoken) self.assertNotIn("km", spoken.lower()) + def test_visual_punctuation_becomes_natural_pauses(self): + spoken = gateway.clean_for_speech( + "Status: stabil – keine Fehler; Docker-Container laufen." + ) + self.assertEqual( + spoken, + "Status, stabil, keine Fehler, Docker Container laufen.", + ) + self.assertNotRegex(spoken, r"[:;\-‐‑‒–—−]") + + def test_punctuation_only_segments_are_never_synthesized(self): + segments = gateway.prepare_segments("Status: – alles läuft.") + self.assertTrue(segments) + self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments)) + def test_weather_summary_is_split_into_short_complete_chunks(self): text = ( "Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. " diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 7e53341..9e9251f 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -290,16 +290,22 @@ def clean_for_speech(text: str) -> str: text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text) text = re.sub(r"(?m)^\s*[-*+]\s+", "", text) text = text.replace("→", ". ").replace("←", ". ") - text = text.replace("–", " - ").replace("—", " - ") if DEFAULT_LANGUAGE == "de": text = normalize_for_german_speech(text) + # XTTS occasionally hallucinates syllables when visual punctuation is + # submitted literally or isolated at a chunk boundary. Preserve its pause + # semantics, but never ask the model to pronounce the glyph itself. + text = re.sub(r"(?<=\w)[\-‐‑‒–—−](?=\w)", " ", text) + text = re.sub(r"\s+[\-‐‑‒–—−]\s+", ", ", text) + text = re.sub(r"\s*[:;]+\s*", ", ", text) text = "".join( char for char in text if unicodedata.category(char) not in {"So", "Cs"} ) text = re.sub(r"[ \t]+", " ", text) text = re.sub(r"\s*\n+\s*", ". ", text) - text = re.sub(r"([:;])\s*\.", r"\1", text) + text = re.sub(r",\s*\.", ".", text) + text = re.sub(r"(?:,\s*){2,}", ", ", text) text = re.sub(r"(?:\.\s*){2,}", ". ", text) return text.strip()