Improve German TTS range handling
This commit is contained in:
@@ -120,6 +120,14 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertIn("2 Uhr 14", spoken)
|
||||
self.assertIn("1. September", spoken)
|
||||
|
||||
def test_qwen_speaks_temperature_and_time_ranges_with_bis(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"15°–23 °C, 93 %, ca. 12 mm, vor allem um 05–06 Uhr."
|
||||
)
|
||||
self.assertIn("15 bis 23 Grad", spoken)
|
||||
self.assertIn("93 Prozent", spoken)
|
||||
self.assertIn("5 bis 6 Uhr", spoken)
|
||||
|
||||
def test_qwen_keeps_prosody_punctuation(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"Ist das gut? Ja! SarahTV: erreichbar."
|
||||
|
||||
@@ -247,6 +247,20 @@ def normalize_for_german_speech(text: str) -> str:
|
||||
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
|
||||
text,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"
|
||||
r"([01]?\d|2[0-3])\s*Uhr\b",
|
||||
lambda match: f"{int(match.group(1))} bis {int(match.group(2))} Uhr",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°?[ \t]*[-‐‑‒–—−][ \t]*"
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||
r"\1 bis \2 Grad",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
||||
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||
|
||||
Reference in New Issue
Block a user