Normalize structured German TTS input
This commit is contained in:
@@ -116,6 +116,16 @@ ENGLISH_PRONUNCIATIONS = {
|
||||
"gib": "gigabytes",
|
||||
}
|
||||
|
||||
GERMAN_MONTHS = {
|
||||
1: "Januar", 2: "Februar", 3: "März", 4: "April",
|
||||
5: "Mai", 6: "Juni", 7: "Juli", 8: "August",
|
||||
9: "September", 10: "Oktober", 11: "November", 12: "Dezember",
|
||||
}
|
||||
GERMAN_DIGITS = {
|
||||
"0": "null", "1": "eins", "2": "zwei", "3": "drei", "4": "vier",
|
||||
"5": "fünf", "6": "sechs", "7": "sieben", "8": "acht", "9": "neun",
|
||||
}
|
||||
|
||||
|
||||
def _request(url: str, *, payload: dict | None = None,
|
||||
timeout: float = 10) -> tuple[bytes, str]:
|
||||
@@ -190,6 +200,74 @@ def segment_languages(text: str) -> list[tuple[str, str]]:
|
||||
return merged
|
||||
|
||||
|
||||
def _spoken_date(match: re.Match) -> str:
|
||||
day = int(match.group(1))
|
||||
month = int(match.group(2))
|
||||
year = match.group(3)
|
||||
month_name = GERMAN_MONTHS.get(month)
|
||||
if not month_name or not 1 <= day <= 31:
|
||||
return match.group(0)
|
||||
result = f"{day}. {month_name}"
|
||||
if year:
|
||||
result += f" {year}"
|
||||
return result
|
||||
|
||||
|
||||
def _spell_digits(value: str) -> str:
|
||||
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
||||
|
||||
|
||||
def normalize_for_german_speech(text: str) -> str:
|
||||
"""Turn common visual notation into unambiguous spoken German."""
|
||||
text = re.sub(
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||
r"Höchstwert \1 Grad, Tiefstwert \2 Grad",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||
r"\1 Grad",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\b([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
||||
_spoken_date,
|
||||
text,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\bPLZ\s+(\d{5})\b",
|
||||
lambda match: "Postleitzahl " + _spell_digits(match.group(1)),
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\((\d{5})\)",
|
||||
lambda match: "(Postleitzahl " + _spell_digits(match.group(1)) + ")",
|
||||
text,
|
||||
)
|
||||
text = re.sub(
|
||||
r"https?://(?:www\.)?([^/\s)]+)(?:/[^\s)]*)?",
|
||||
lambda match: match.group(1),
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\b([A-Za-z0-9][A-Za-z0-9-]*(?:\.[A-Za-z0-9-]+)*)"
|
||||
r"\.(de|com|org|net|io|ai)\b",
|
||||
lambda match: (
|
||||
match.group(1).replace(".", " Punkt ")
|
||||
+ " Punkt " + " ".join(match.group(2).upper())
|
||||
),
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
|
||||
return text
|
||||
|
||||
|
||||
def clean_for_speech(text: str) -> str:
|
||||
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
||||
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
||||
@@ -200,12 +278,15 @@ def clean_for_speech(text: str) -> str:
|
||||
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||
text = text.replace("→", ". ").replace("←", ". ")
|
||||
text = text.replace("–", " - ").replace("—", " - ")
|
||||
if DEFAULT_LANGUAGE == "de":
|
||||
text = normalize_for_german_speech(text)
|
||||
text = "".join(
|
||||
char for char in text
|
||||
if unicodedata.category(char) not in {"So", "Cs"}
|
||||
)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\s*\n+\s*", ". ", text)
|
||||
text = re.sub(r"([:;])\s*\.", r"\1", text)
|
||||
text = re.sub(r"(?:\.\s*){2,}", ". ", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user