Improve German TTS range handling

This commit is contained in:
Mikei386 committed 2026-09-05 14:07:44 +02:00
1 parent cc416150a8
commit 63e7a93ee4
3 files changed
+39

No files matched your search

@@ -120,6 +120,14 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertIn("2 Uhr 14", spoken) self.assertIn("2 Uhr 14", spoken)
self.assertIn("1. September", spoken) self.assertIn("1. September", spoken)
def test_qwen_speaks_temperature_and_time_ranges_with_bis(self):
spoken = gateway.prepare_for_qwen_speech(
"15°–23 °C, 93 %, ca. 12 mm, vor allem um 05–06 Uhr."
)
self.assertIn("15 bis 23 Grad", spoken)
self.assertIn("93 Prozent", spoken)
self.assertIn("5 bis 6 Uhr", spoken)
def test_qwen_keeps_prosody_punctuation(self): def test_qwen_keeps_prosody_punctuation(self):
spoken = gateway.prepare_for_qwen_speech( spoken = gateway.prepare_for_qwen_speech(
"Ist das gut? Ja! SarahTV: erreichbar." "Ist das gut? Ja! SarahTV: erreichbar."
@@ -247,6 +247,20 @@ def normalize_for_german_speech(text: str) -> str:
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}", lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
text, text,
) )
text = re.sub(
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"
r"([01]?\d|2[0-3])\s*Uhr\b",
lambda match: f"{int(match.group(1))} bis {int(match.group(2))} Uhr",
text,
flags=re.IGNORECASE,
)
text = re.sub(
r"(-?\d+(?:[,.]\d+)?)[ \t]*°?[ \t]*[-‐‑‒–—−][ \t]*"
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
r"\1 bis \2 Grad",
text,
flags=re.IGNORECASE,
)
text = re.sub( text = re.sub(
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*" r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?", r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
+17
View File
@@ -15,3 +15,20 @@ cli_dir="${HERMES_HOME:-/opt/data}/.local/bin"
mkdir -p "$cli_dir" mkdir -p "$cli_dir"
ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes" ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes"
ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes
# Hermes streams each detected sentence to TTS separately. Its generic
# sentence splitter treats the period in the German abbreviation "ca." as a
# sentence ending, causing "ca." and the following measurement to be spoken
# as two unrelated utterances. Expand the abbreviation before splitting.
tts_target=/opt/hermes/tools/tts_streaming.py
tts_old=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)'
tts_new=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)
self.buf = re.sub(r"\bca\.(?=\s)", "circa", self.buf, flags=re.IGNORECASE)'
if [ -f "$tts_target" ] && grep -Fq "$tts_old" "$tts_target"; then
TTS_TARGET="$tts_target" TTS_OLD="$tts_old" TTS_NEW="$tts_new" python3 -c \
'import os; from pathlib import Path; p = Path(os.environ["TTS_TARGET"]); s = p.read_text(); p.write_text(s.replace(os.environ["TTS_OLD"], os.environ["TTS_NEW"], 1))'
echo "[tts-streaming-fix] expanded German ca. before sentence splitting"
else
echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action"
fi