Smooth Hermes TTS sentence transitions
This commit is contained in:
@@ -132,6 +132,21 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
self.assertIn("5 bis 6 Uhr", spoken)
|
self.assertIn("5 bis 6 Uhr", spoken)
|
||||||
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
|
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
|
||||||
|
|
||||||
|
def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self):
|
||||||
|
spoken = gateway.prepare_for_qwen_speech(
|
||||||
|
"Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026."
|
||||||
|
)
|
||||||
|
self.assertIn("4. bis 5. September", spoken)
|
||||||
|
self.assertIn("4. September bis 6. Oktober 2026", spoken)
|
||||||
|
|
||||||
|
def test_qwen_does_not_treat_plain_number_ranges_as_dates(self):
|
||||||
|
spoken = gateway.prepare_for_qwen_speech(
|
||||||
|
"Version 3.8, Werte 11,3 bis 29,0 und Kontext 80–160K."
|
||||||
|
)
|
||||||
|
self.assertIn("Version 3.8", spoken)
|
||||||
|
self.assertIn("11,3 bis 29,0", spoken)
|
||||||
|
self.assertIn("80–160K", spoken)
|
||||||
|
|
||||||
def test_qwen_keeps_prosody_punctuation(self):
|
def test_qwen_keeps_prosody_punctuation(self):
|
||||||
spoken = gateway.prepare_for_qwen_speech(
|
spoken = gateway.prepare_for_qwen_speech(
|
||||||
"Ist das gut? Ja! SarahTV: erreichbar."
|
"Ist das gut? Ja! SarahTV: erreichbar."
|
||||||
|
|||||||
@@ -224,6 +224,40 @@ def _spoken_date(match: re.Match) -> str:
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _spoken_short_date_range(match: re.Match) -> str:
|
||||||
|
first_day = int(match.group(1))
|
||||||
|
last_day = int(match.group(2))
|
||||||
|
month = int(match.group(3))
|
||||||
|
year = match.group(4)
|
||||||
|
month_name = GERMAN_MONTHS.get(month)
|
||||||
|
if not month_name or not 1 <= first_day <= 31 or not 1 <= last_day <= 31:
|
||||||
|
return match.group(0)
|
||||||
|
result = f"{first_day}. bis {last_day}. {month_name}"
|
||||||
|
if year:
|
||||||
|
result += f" {year}"
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _spoken_full_date_range(match: re.Match) -> str:
|
||||||
|
first_day = int(match.group(1))
|
||||||
|
first_month = int(match.group(2))
|
||||||
|
last_day = int(match.group(3))
|
||||||
|
last_month = int(match.group(4))
|
||||||
|
year = match.group(5)
|
||||||
|
first_name = GERMAN_MONTHS.get(first_month)
|
||||||
|
last_name = GERMAN_MONTHS.get(last_month)
|
||||||
|
if (not first_name or not last_name
|
||||||
|
or not 1 <= first_day <= 31 or not 1 <= last_day <= 31):
|
||||||
|
return match.group(0)
|
||||||
|
if first_month == last_month:
|
||||||
|
result = f"{first_day}. bis {last_day}. {last_name}"
|
||||||
|
else:
|
||||||
|
result = f"{first_day}. {first_name} bis {last_day}. {last_name}"
|
||||||
|
if year:
|
||||||
|
result += f" {year}"
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
def _spell_digits(value: str) -> str:
|
def _spell_digits(value: str) -> str:
|
||||||
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
||||||
|
|
||||||
@@ -277,6 +311,20 @@ def normalize_for_german_speech(text: str) -> str:
|
|||||||
text,
|
text,
|
||||||
flags=re.IGNORECASE,
|
flags=re.IGNORECASE,
|
||||||
)
|
)
|
||||||
|
# Date ranges are deliberately strict: both variants require dotted
|
||||||
|
# calendar notation, so ordinary numeric ranges remain untouched.
|
||||||
|
text = re.sub(
|
||||||
|
r"\b([0-3]?\d)\.([01]?\d)\.\s*[-‐‑‒–—−]\s*"
|
||||||
|
r"([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
||||||
|
_spoken_full_date_range,
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\b([0-3]?\d)\.\s*[-‐‑‒–—−]\s*"
|
||||||
|
r"([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
||||||
|
_spoken_short_date_range,
|
||||||
|
text,
|
||||||
|
)
|
||||||
text = re.sub(
|
text = re.sub(
|
||||||
r"\b([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
r"\b([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
||||||
_spoken_date,
|
_spoken_date,
|
||||||
|
|||||||
@@ -16,21 +16,55 @@ mkdir -p "$cli_dir"
|
|||||||
ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes"
|
ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes"
|
||||||
ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes
|
ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes
|
||||||
|
|
||||||
# Hermes streams each detected sentence to TTS separately. Its generic
|
# Hermes' generic sentence splitter treats periods in German abbreviations as
|
||||||
# sentence splitter treats periods in German abbreviations as sentence ends,
|
# sentence ends. It also submits every sentence as a separate generative TTS
|
||||||
# causing the following words to become unrelated utterances. Expand these
|
# request, which creates long gaps with local Qwen3-TTS. Keep the first sentence
|
||||||
# abbreviations before splitting.
|
# immediate, then group following complete sentences into modest ~240-character
|
||||||
|
# chunks so playback remains responsive but substantially more continuous.
|
||||||
tts_target=/opt/hermes/tools/tts_streaming.py
|
tts_target=/opt/hermes/tools/tts_streaming.py
|
||||||
tts_old=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)'
|
if [ -f "$tts_target" ] && grep -Fq 'self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' "$tts_target"; then
|
||||||
tts_new=' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)
|
TTS_TARGET="$tts_target" python3 <<'PY'
|
||||||
self.buf = re.sub(r"\bca\.(?=\s)", "circa", self.buf, flags=re.IGNORECASE)
|
import os
|
||||||
self.buf = re.sub(r"\bv\.\s*a\.(?=\s)", "vor allem", self.buf, flags=re.IGNORECASE)
|
from pathlib import Path
|
||||||
self.buf = re.sub(r"\bmax\.(?=\s)", "maximal", self.buf, flags=re.IGNORECASE)'
|
|
||||||
|
|
||||||
if [ -f "$tts_target" ] && grep -Fq "$tts_old" "$tts_target"; then
|
p = Path(os.environ["TTS_TARGET"])
|
||||||
TTS_TARGET="$tts_target" TTS_OLD="$tts_old" TTS_NEW="$tts_new" python3 -c \
|
s = p.read_text()
|
||||||
'import os; from functools import reduce; from pathlib import Path; p = Path(os.environ["TTS_TARGET"]); s = p.read_text(); lines = [" self.buf = re.sub(r\"\\bca\\.(?=\\s)\", \"circa\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bv\\.\\s*a\\.(?=\\s)\", \"vor allem\", self.buf, flags=re.IGNORECASE)", " self.buf = re.sub(r\"\\bmax\\.(?=\\s)\", \"maximal\", self.buf, flags=re.IGNORECASE)"]; s = reduce(lambda value, line: value.replace("\n" + line, ""), lines, s); p.write_text(s.replace(os.environ["TTS_OLD"], os.environ["TTS_NEW"], 1))'
|
|
||||||
echo "[tts-streaming-fix] expanded German abbreviations before sentence splitting"
|
base_feed = ' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)'
|
||||||
|
abbr_lines = [
|
||||||
|
' self.buf = re.sub(r"\\bca\\.(?=\\s)", "circa", self.buf, flags=re.IGNORECASE)',
|
||||||
|
' self.buf = re.sub(r"\\bv\\.\\s*a\\.(?=\\s)", "vor allem", self.buf, flags=re.IGNORECASE)',
|
||||||
|
' self.buf = re.sub(r"\\bmax\\.(?=\\s)", "maximal", self.buf, flags=re.IGNORECASE)',
|
||||||
|
]
|
||||||
|
for line in abbr_lines:
|
||||||
|
s = s.replace("\n" + line, "")
|
||||||
|
|
||||||
|
init_old = ' self.buf = ""'
|
||||||
|
init_extra = (
|
||||||
|
'\n self.followup_target_len = 240'
|
||||||
|
'\n self._emitted_first = False'
|
||||||
|
)
|
||||||
|
s = s.replace(init_old + init_extra, init_old)
|
||||||
|
|
||||||
|
threshold_old = ' if len(head.strip()) < self.min_len:'
|
||||||
|
threshold_new = (
|
||||||
|
' threshold = (self.min_len if not self._emitted_first '
|
||||||
|
'else self.followup_target_len)\n'
|
||||||
|
' if len(head.strip()) < threshold:'
|
||||||
|
)
|
||||||
|
s = s.replace(threshold_new, threshold_old)
|
||||||
|
|
||||||
|
append_old = ' out.append(head)'
|
||||||
|
append_new = append_old + '\n self._emitted_first = True'
|
||||||
|
s = s.replace(append_new, append_old)
|
||||||
|
|
||||||
|
s = s.replace(base_feed, base_feed + "\n" + "\n".join(abbr_lines), 1)
|
||||||
|
s = s.replace(init_old, init_old + init_extra, 1)
|
||||||
|
s = s.replace(threshold_old, threshold_new, 1)
|
||||||
|
s = s.replace(append_old, append_new, 1)
|
||||||
|
p.write_text(s)
|
||||||
|
PY
|
||||||
|
echo "[tts-streaming-fix] enabled German normalization and adaptive sentence grouping"
|
||||||
else
|
else
|
||||||
echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action"
|
echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action"
|
||||||
fi
|
fi
|
||||||
|
|||||||
Reference in New Issue
Block a user