diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 1be3297..5ec5669 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -122,11 +122,15 @@ class LanguageSegmentationTests(unittest.TestCase): def test_qwen_speaks_temperature_and_time_ranges_with_bis(self): spoken = gateway.prepare_for_qwen_speech( - "15°–23 °C, 93 %, ca. 12 mm, vor allem um 05–06 Uhr." + "15°–23 °C, 93 %, ca. 12 mm, v.a. um 05–06 Uhr, " + "Wind max. 16 km/h." ) self.assertIn("15 bis 23 Grad", spoken) self.assertIn("93 Prozent", spoken) + self.assertIn("circa 12 mm", spoken) + self.assertIn("vor allem", spoken) self.assertIn("5 bis 6 Uhr", spoken) + self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken) def test_qwen_keeps_prosody_punctuation(self): spoken = gateway.prepare_for_qwen_speech( diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 8e6e98c..5b80788 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -235,6 +235,9 @@ def _spoken_ipv4(match: re.Match) -> str: def normalize_for_german_speech(text: str) -> str: """Turn common visual notation into unambiguous spoken German.""" + text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE) + text = re.sub(r"\bca\.", "circa", text, flags=re.IGNORECASE) + text = re.sub(r"\bmax\.", "maximal", text, flags=re.IGNORECASE) # Run these before the date rule: otherwise 192.168.1.5 could be partly # interpreted as a visual date. text = re.sub( @@ -329,10 +332,13 @@ def prepare_for_qwen_speech(text: str) -> str: text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text) text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text) text = re.sub(r"(?m)^\s*[-*+]\s+", "", text) - text = text.replace("_", " ").replace("/", ", ") + text = text.replace("_", " ") text = text.replace("→", ". ").replace("←", ". ") if DEFAULT_LANGUAGE == "de": text = normalize_for_german_speech(text) + # Preserve slashes until after unit normalization so km/h becomes + # "Kilometer pro Stunde" instead of the broken "Kilometer, h". + text = text.replace("/", ", ") text = "".join( char for char in text if unicodedata.category(char) not in {"So", "Cs"} diff --git a/platform/hermes/025-cron-profile-root-fix b/platform/hermes/025-cron-profile-root-fix index fa67682..be2beeb 100755 --- a/platform/hermes/025-cron-profile-root-fix +++ b/platform/hermes/025-cron-profile-root-fix @@ -17,18 +17,20 @@ ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes" ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes # Hermes streams each detected sentence to TTS separately. Its generic -# sentence splitter treats the period in the German abbreviation "ca." as a -# sentence ending, causing "ca." and the following measurement to be spoken -# as two unrelated utterances. Expand the abbreviation before splitting. +# sentence splitter treats periods in German abbreviations as sentence ends, +# causing the following words to become unrelated utterances. Expand these +# abbreviations before splitting. tts_target=/opt/hermes/tools/tts_streaming.py tts_old=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' tts_new=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta) - self.buf = re.sub(r"\bca\.(?=\s)", "circa", self.buf, flags=re.IGNORECASE)' + self.buf = re.sub(r"\bca\.(?=\s)", "circa", self.buf, flags=re.IGNORECASE) + self.buf = re.sub(r"\bv\.\s*a\.(?=\s)", "vor allem", self.buf, flags=re.IGNORECASE) + self.buf = re.sub(r"\bmax\.(?=\s)", "maximal", self.buf, flags=re.IGNORECASE)' if [ -f "$tts_target" ] && grep -Fq "$tts_old" "$tts_target"; then TTS_TARGET="$tts_target" TTS_OLD="$tts_old" TTS_NEW="$tts_new" python3 -c \ - 'import os; from pathlib import Path; p = Path(os.environ["TTS_TARGET"]); s = p.read_text(); p.write_text(s.replace(os.environ["TTS_OLD"], os.environ["TTS_NEW"], 1))' - echo "[tts-streaming-fix] expanded German ca. before sentence splitting" + 'import os; from functools import reduce; from pathlib import Path; p = Path(os.environ["TTS_TARGET"]); s = p.read_text(); lines = [" self.buf = re.sub(r\"\\bca\\.(?=\\s)\", \"circa\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bv\\.\\s*a\\.(?=\\s)\", \"vor allem\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bmax\\.(?=\\s)\", \"maximal\", self.buf, flags=re.IGNORECASE)"]; s = reduce(lambda value, line: value.replace("\n" + line, ""), lines, s); p.write_text(s.replace(os.environ["TTS_OLD"], os.environ["TTS_NEW"], 1))' + echo "[tts-streaming-fix] expanded German abbreviations before sentence splitting" else echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action" fi