From 5f3d064bb403843e41459e06264d29a77c6ace67 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sat, 5 Sep 2026 15:28:11 +0200 Subject: [PATCH] Smooth Hermes TTS sentence transitions --- .../docker/tts-gateway/test_tts_gateway.py | 15 +++++ platform/docker/tts-gateway/tts_gateway.py | 48 +++++++++++++++ platform/hermes/025-cron-profile-root-fix | 60 +++++++++++++++---- 3 files changed, 110 insertions(+), 13 deletions(-) diff --git a/platform/docker/tts-gateway/test_tts_gateway.py b/platform/docker/tts-gateway/test_tts_gateway.py index 5ec5669..f1b9dfa 100644 --- a/platform/docker/tts-gateway/test_tts_gateway.py +++ b/platform/docker/tts-gateway/test_tts_gateway.py @@ -132,6 +132,21 @@ class LanguageSegmentationTests(unittest.TestCase): self.assertIn("5 bis 6 Uhr", spoken) self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken) + def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self): + spoken = gateway.prepare_for_qwen_speech( + "Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026." + ) + self.assertIn("4. bis 5. September", spoken) + self.assertIn("4. September bis 6. Oktober 2026", spoken) + + def test_qwen_does_not_treat_plain_number_ranges_as_dates(self): + spoken = gateway.prepare_for_qwen_speech( + "Version 3.8, Werte 11,3 bis 29,0 und Kontext 80–160K." + ) + self.assertIn("Version 3.8", spoken) + self.assertIn("11,3 bis 29,0", spoken) + self.assertIn("80–160K", spoken) + def test_qwen_keeps_prosody_punctuation(self): spoken = gateway.prepare_for_qwen_speech( "Ist das gut? Ja! SarahTV: erreichbar." diff --git a/platform/docker/tts-gateway/tts_gateway.py b/platform/docker/tts-gateway/tts_gateway.py index 5b80788..04315e1 100644 --- a/platform/docker/tts-gateway/tts_gateway.py +++ b/platform/docker/tts-gateway/tts_gateway.py @@ -224,6 +224,40 @@ def _spoken_date(match: re.Match) -> str: return result +def _spoken_short_date_range(match: re.Match) -> str: + first_day = int(match.group(1)) + last_day = int(match.group(2)) + month = int(match.group(3)) + year = match.group(4) + month_name = GERMAN_MONTHS.get(month) + if not month_name or not 1 <= first_day <= 31 or not 1 <= last_day <= 31: + return match.group(0) + result = f"{first_day}. bis {last_day}. {month_name}" + if year: + result += f" {year}" + return result + + +def _spoken_full_date_range(match: re.Match) -> str: + first_day = int(match.group(1)) + first_month = int(match.group(2)) + last_day = int(match.group(3)) + last_month = int(match.group(4)) + year = match.group(5) + first_name = GERMAN_MONTHS.get(first_month) + last_name = GERMAN_MONTHS.get(last_month) + if (not first_name or not last_name + or not 1 <= first_day <= 31 or not 1 <= last_day <= 31): + return match.group(0) + if first_month == last_month: + result = f"{first_day}. bis {last_day}. {last_name}" + else: + result = f"{first_day}. {first_name} bis {last_day}. {last_name}" + if year: + result += f" {year}" + return result + + def _spell_digits(value: str) -> str: return " ".join(GERMAN_DIGITS[digit] for digit in value) @@ -277,6 +311,20 @@ def normalize_for_german_speech(text: str) -> str: text, flags=re.IGNORECASE, ) + # Date ranges are deliberately strict: both variants require dotted + # calendar notation, so ordinary numeric ranges remain untouched. + text = re.sub( + r"\b([0-3]?\d)\.([01]?\d)\.\s*[-‐‑‒–—−]\s*" + r"([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?", + _spoken_full_date_range, + text, + ) + text = re.sub( + r"\b([0-3]?\d)\.\s*[-‐‑‒–—−]\s*" + r"([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?", + _spoken_short_date_range, + text, + ) text = re.sub( r"\b([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?", _spoken_date, diff --git a/platform/hermes/025-cron-profile-root-fix b/platform/hermes/025-cron-profile-root-fix index be2beeb..c7d905e 100755 --- a/platform/hermes/025-cron-profile-root-fix +++ b/platform/hermes/025-cron-profile-root-fix @@ -16,21 +16,55 @@ mkdir -p "$cli_dir" ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes" ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes -# Hermes streams each detected sentence to TTS separately. Its generic -# sentence splitter treats periods in German abbreviations as sentence ends, -# causing the following words to become unrelated utterances. Expand these -# abbreviations before splitting. +# Hermes' generic sentence splitter treats periods in German abbreviations as +# sentence ends. It also submits every sentence as a separate generative TTS +# request, which creates long gaps with local Qwen3-TTS. Keep the first sentence +# immediate, then group following complete sentences into modest ~240-character +# chunks so playback remains responsive but substantially more continuous. tts_target=/opt/hermes/tools/tts_streaming.py -tts_old=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' -tts_new=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta) - self.buf = re.sub(r"\bca\.(?=\s)", "circa", self.buf, flags=re.IGNORECASE) - self.buf = re.sub(r"\bv\.\s*a\.(?=\s)", "vor allem", self.buf, flags=re.IGNORECASE) - self.buf = re.sub(r"\bmax\.(?=\s)", "maximal", self.buf, flags=re.IGNORECASE)' +if [ -f "$tts_target" ] && grep -Fq 'self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' "$tts_target"; then + TTS_TARGET="$tts_target" python3 <<'PY' +import os +from pathlib import Path -if [ -f "$tts_target" ] && grep -Fq "$tts_old" "$tts_target"; then - TTS_TARGET="$tts_target" TTS_OLD="$tts_old" TTS_NEW="$tts_new" python3 -c \ - 'import os; from functools import reduce; from pathlib import Path; p = Path(os.environ["TTS_TARGET"]); s = p.read_text(); lines = [" self.buf = re.sub(r\"\\bca\\.(?=\\s)\", \"circa\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bv\\.\\s*a\\.(?=\\s)\", \"vor allem\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bmax\\.(?=\\s)\", \"maximal\", self.buf, flags=re.IGNORECASE)"]; s = reduce(lambda value, line: value.replace("\n" + line, ""), lines, s); p.write_text(s.replace(os.environ["TTS_OLD"], os.environ["TTS_NEW"], 1))' - echo "[tts-streaming-fix] expanded German abbreviations before sentence splitting" +p = Path(os.environ["TTS_TARGET"]) +s = p.read_text() + +base_feed = ' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' +abbr_lines = [ + ' self.buf = re.sub(r"\\bca\\.(?=\\s)", "circa", self.buf, flags=re.IGNORECASE)', + ' self.buf = re.sub(r"\\bv\\.\\s*a\\.(?=\\s)", "vor allem", self.buf, flags=re.IGNORECASE)', + ' self.buf = re.sub(r"\\bmax\\.(?=\\s)", "maximal", self.buf, flags=re.IGNORECASE)', +] +for line in abbr_lines: + s = s.replace("\n" + line, "") + +init_old = ' self.buf = ""' +init_extra = ( + '\n self.followup_target_len = 240' + '\n self._emitted_first = False' +) +s = s.replace(init_old + init_extra, init_old) + +threshold_old = ' if len(head.strip()) < self.min_len:' +threshold_new = ( + ' threshold = (self.min_len if not self._emitted_first ' + 'else self.followup_target_len)\n' + ' if len(head.strip()) < threshold:' +) +s = s.replace(threshold_new, threshold_old) + +append_old = ' out.append(head)' +append_new = append_old + '\n self._emitted_first = True' +s = s.replace(append_new, append_old) + +s = s.replace(base_feed, base_feed + "\n" + "\n".join(abbr_lines), 1) +s = s.replace(init_old, init_old + init_extra, 1) +s = s.replace(threshold_old, threshold_new, 1) +s = s.replace(append_old, append_new, 1) +p.write_text(s) +PY + echo "[tts-streaming-fix] enabled German normalization and adaptive sentence grouping" else echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action" fi