Normalize German TTS abbreviations and units
This commit is contained in:
1 parent
63e7a93ee4
commit
cbc312257c
3 files changed
+20
-8
No files matched your search
@@ -122,11 +122,15 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
|
||||
def test_qwen_speaks_temperature_and_time_ranges_with_bis(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"15°–23 °C, 93 %, ca. 12 mm, vor allem um 05–06 Uhr."
|
||||
"15°–23 °C, 93 %, ca. 12 mm, v.a. um 05–06 Uhr, "
|
||||
"Wind max. 16 km/h."
|
||||
)
|
||||
self.assertIn("15 bis 23 Grad", spoken)
|
||||
self.assertIn("93 Prozent", spoken)
|
||||
self.assertIn("circa 12 mm", spoken)
|
||||
self.assertIn("vor allem", spoken)
|
||||
self.assertIn("5 bis 6 Uhr", spoken)
|
||||
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
|
||||
|
||||
def test_qwen_keeps_prosody_punctuation(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
|
||||
@@ -235,6 +235,9 @@ def _spoken_ipv4(match: re.Match) -> str:
|
||||
|
||||
def normalize_for_german_speech(text: str) -> str:
|
||||
"""Turn common visual notation into unambiguous spoken German."""
|
||||
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"\bca\.", "circa", text, flags=re.IGNORECASE)
|
||||
text = re.sub(r"\bmax\.", "maximal", text, flags=re.IGNORECASE)
|
||||
# Run these before the date rule: otherwise 192.168.1.5 could be partly
|
||||
# interpreted as a visual date.
|
||||
text = re.sub(
|
||||
@@ -329,10 +332,13 @@ def prepare_for_qwen_speech(text: str) -> str:
|
||||
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
|
||||
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
|
||||
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||
text = text.replace("_", " ").replace("/", ", ")
|
||||
text = text.replace("_", " ")
|
||||
text = text.replace("→", ". ").replace("←", ". ")
|
||||
if DEFAULT_LANGUAGE == "de":
|
||||
text = normalize_for_german_speech(text)
|
||||
# Preserve slashes until after unit normalization so km/h becomes
|
||||
# "Kilometer pro Stunde" instead of the broken "Kilometer, h".
|
||||
text = text.replace("/", ", ")
|
||||
text = "".join(
|
||||
char for char in text
|
||||
if unicodedata.category(char) not in {"So", "Cs"}
|
||||
|
||||
Reference in new issue
Block a user