Normalize punctuation before German TTS
This commit is contained in:
@@ -110,6 +110,21 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertIn("29 Kilometer pro Stunde", spoken)
|
||||
self.assertNotIn("km", spoken.lower())
|
||||
|
||||
def test_visual_punctuation_becomes_natural_pauses(self):
|
||||
spoken = gateway.clean_for_speech(
|
||||
"Status: stabil – keine Fehler; Docker-Container laufen."
|
||||
)
|
||||
self.assertEqual(
|
||||
spoken,
|
||||
"Status, stabil, keine Fehler, Docker Container laufen.",
|
||||
)
|
||||
self.assertNotRegex(spoken, r"[:;\-‐‑‒–—−]")
|
||||
|
||||
def test_punctuation_only_segments_are_never_synthesized(self):
|
||||
segments = gateway.prepare_segments("Status: – alles läuft.")
|
||||
self.assertTrue(segments)
|
||||
self.assertTrue(all(any(char.isalnum() for char in part) for _, part in segments))
|
||||
|
||||
def test_weather_summary_is_split_into_short_complete_chunks(self):
|
||||
text = (
|
||||
"Heute in Rastatt: teils sonnig, trocken, 10 bis 22 Grad. "
|
||||
|
||||
@@ -290,16 +290,22 @@ def clean_for_speech(text: str) -> str:
|
||||
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
|
||||
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||
text = text.replace("→", ". ").replace("←", ". ")
|
||||
text = text.replace("–", " - ").replace("—", " - ")
|
||||
if DEFAULT_LANGUAGE == "de":
|
||||
text = normalize_for_german_speech(text)
|
||||
# XTTS occasionally hallucinates syllables when visual punctuation is
|
||||
# submitted literally or isolated at a chunk boundary. Preserve its pause
|
||||
# semantics, but never ask the model to pronounce the glyph itself.
|
||||
text = re.sub(r"(?<=\w)[\-‐‑‒–—−](?=\w)", " ", text)
|
||||
text = re.sub(r"\s+[\-‐‑‒–—−]\s+", ", ", text)
|
||||
text = re.sub(r"\s*[:;]+\s*", ", ", text)
|
||||
text = "".join(
|
||||
char for char in text
|
||||
if unicodedata.category(char) not in {"So", "Cs"}
|
||||
)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\s*\n+\s*", ". ", text)
|
||||
text = re.sub(r"([:;])\s*\.", r"\1", text)
|
||||
text = re.sub(r",\s*\.", ".", text)
|
||||
text = re.sub(r"(?:,\s*){2,}", ", ", text)
|
||||
text = re.sub(r"(?:\.\s*){2,}", ". ", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user