Speak aspect ratios correctly in TTS

This commit is contained in:
Mikei386
2026-09-08 14:42:21 +02:00
parent f1ed51a302
commit 2724861224
2 changed files with 54 additions and 1 deletions
@@ -132,6 +132,25 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertIn("5 bis 6 Uhr", spoken)
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
def test_qwen_speaks_aspect_ratios_as_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Cover sind im Hochformat (2:3), Screenshots im Querformat "
"(16:9), ein Quadrat im Seitenverhältnis 2:2 und 4:3-Format."
)
self.assertIn("Hochformat (2 zu 3)", spoken)
self.assertIn("Querformat (16 zu 9)", spoken)
self.assertIn("Seitenverhältnis 2 zu 2", spoken)
self.assertIn("4 zu 3-Format", spoken)
def test_qwen_keeps_clock_times_distinct_from_aspect_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Beginn um 16:09 Uhr, Fehler um 02:14; das Videoformat ist 16:9."
)
self.assertIn("16 Uhr 9", spoken)
self.assertNotIn("16 Uhr 9 Uhr", spoken)
self.assertIn("2 Uhr 14", spoken)
self.assertIn("Videoformat ist 16 zu 9", spoken)
def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self):
spoken = gateway.prepare_for_qwen_speech(
"Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026."
+35 -1
View File
@@ -269,6 +269,36 @@ def _spoken_ipv4(match: re.Match) -> str:
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
def _normalize_aspect_ratios(text: str) -> str:
"""Speak colon notation as a ratio only when the surrounding text says so.
A bare ``16:09`` remains a clock time. This deliberately avoids a global
replacement of common ratios because ``16:9`` can also be a valid time.
"""
cue = (
r"(?:Seitenverh[aä]ltnis|Bildseitenverh[aä]ltnis|Bildformat|"
r"Videoformat|Hochformat|Querformat|Format|Aspect[- ]?Ratio)"
)
text = re.sub(
rf"\b({cue}\b(?:\s+(?:von|im|ist|betr[aä]gt))?\s*[\(\[]?\s*)"
rf"(\d{{1,3}})\s*:\s*(\d{{1,3}})",
lambda match: (
f"{match.group(1)}{int(match.group(2))} zu {int(match.group(3))}"
),
text,
flags=re.IGNORECASE,
)
return re.sub(
rf"\b(\d{{1,3}})\s*:\s*(\d{{1,3}})"
rf"(\s*[-‐‑‒–—−]?\s*{cue}\b)",
lambda match: (
f"{int(match.group(1))} zu {int(match.group(2))}{match.group(3)}"
),
text,
flags=re.IGNORECASE,
)
def normalize_for_german_speech(text: str) -> str:
"""Turn common visual notation into unambiguous spoken German."""
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
@@ -281,10 +311,14 @@ def normalize_for_german_speech(text: str) -> str:
_spoken_ipv4,
text,
)
# A colon is ambiguous between an aspect ratio and a clock time. Resolve
# ratios first, but only when an explicit format cue is present.
text = _normalize_aspect_ratios(text)
text = re.sub(
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
r"\b([01]?\d|2[0-3]):([0-5]\d)\b(?:\s*Uhr\b)?",
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
text,
flags=re.IGNORECASE,
)
text = re.sub(
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"