Normalize German TTS abbreviations and units

This commit is contained in:
Mikei386
2026-09-05 15:11:57 +02:00
parent 63e7a93ee4
commit cbc312257c
3 changed files with 20 additions and 8 deletions
+7 -1
View File
@@ -235,6 +235,9 @@ def _spoken_ipv4(match: re.Match) -> str:
def normalize_for_german_speech(text: str) -> str:
"""Turn common visual notation into unambiguous spoken German."""
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
text = re.sub(r"\bca\.", "circa", text, flags=re.IGNORECASE)
text = re.sub(r"\bmax\.", "maximal", text, flags=re.IGNORECASE)
# Run these before the date rule: otherwise 192.168.1.5 could be partly
# interpreted as a visual date.
text = re.sub(
@@ -329,10 +332,13 @@ def prepare_for_qwen_speech(text: str) -> str:
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
text = text.replace("_", " ").replace("/", ", ")
text = text.replace("_", " ")
text = text.replace("→", ". ").replace("←", ". ")
if DEFAULT_LANGUAGE == "de":
text = normalize_for_german_speech(text)
# Preserve slashes until after unit normalization so km/h becomes
# "Kilometer pro Stunde" instead of the broken "Kilometer, h".
text = text.replace("/", ", ")
text = "".join(
char for char in text
if unicodedata.category(char) not in {"So", "Cs"}