Normalize structured German TTS input
This commit is contained in:
1 parent
a067b79ed2
commit
d726afff70
4 files changed
+113
No files matched your search
@@ -122,6 +122,12 @@ Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
|
|||||||
plus Projektor. Nicht installierte Worker-Endpunkte antworten klar mit
|
plus Projektor. Nicht installierte Worker-Endpunkte antworten klar mit
|
||||||
`feature_disabled`, statt alte systemd-Pfade aufzurufen.
|
`feature_disabled`, statt alte systemd-Pfade aufzurufen.
|
||||||
|
|
||||||
|
Vor der Synthese wandelt das Gateway visuelle Schreibweisen in natürliche
|
||||||
|
deutsche Sprache um. Dazu gehören Datumsangaben, Temperaturbereiche,
|
||||||
|
Prozentwerte, fünfstellige Postleitzahlen und Internet-Domains. Das verhindert
|
||||||
|
Ausgaben wie „zweiundzwanzig Komma zehn“ für `22° / 10°`; Domain-Endungen
|
||||||
|
werden eindeutig buchstabiert.
|
||||||
|
|
||||||
## Zentrale MCP-Werkzeugebene
|
## Zentrale MCP-Werkzeugebene
|
||||||
|
|
||||||
Werkzeuge werden nicht in llama.cpp eingebaut. Sie laufen als eigene,
|
Werkzeuge werden nicht in llama.cpp eingebaut. Sie laufen als eigene,
|
||||||
|
|||||||
@@ -95,6 +95,11 @@ ausgesprochen, die Ausgabe bleibt jedoch flüssig und verständlich. Reine
|
|||||||
englische Texte erkennt das Gateway weiterhin automatisch. Ein Ende-zu-Ende-Test
|
englische Texte erkennt das Gateway weiterhin automatisch. Ein Ende-zu-Ende-Test
|
||||||
ohne Ausgabe des API-Schlüssels:
|
ohne Ausgabe des API-Schlüssels:
|
||||||
|
|
||||||
|
Für die Sprachausgabe normalisiert das Gateway außerdem Datumsangaben,
|
||||||
|
Temperaturen, Prozentwerte, Postleitzahlen und Domains. Beispielsweise wird
|
||||||
|
`22° / 10°` als „Höchstwert 22 Grad, Tiefstwert 10 Grad“ und `wetter.com` als
|
||||||
|
„Wetter Punkt C O M“ gesprochen. Die sichtbare Chatantwort wird nicht verändert.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
set -a; source /etc/mike-ai/stack.env; set +a
|
set -a; source /etc/mike-ai/stack.env; set +a
|
||||||
curl -fsS http://127.0.0.1:8081/v1/audio/speech \
|
curl -fsS http://127.0.0.1:8081/v1/audio/speech \
|
||||||
|
|||||||
@@ -78,6 +78,27 @@ class LanguageSegmentationTests(unittest.TestCase):
|
|||||||
text = "Docker-Container laufen, die Health-Checks melden healthy."
|
text = "Docker-Container laufen, die Health-Checks melden healthy."
|
||||||
self.assertEqual(gateway.segment_languages(text), [("de", text)])
|
self.assertEqual(gateway.segment_languages(text), [("de", text)])
|
||||||
|
|
||||||
|
def test_weather_notation_is_spoken_naturally(self):
|
||||||
|
text = (
|
||||||
|
"Rastatt (76437), Sonntag 23.08.: 22° / 10°, "
|
||||||
|
"0 % Regen. Quellen: wetter.com und wetteronline.de"
|
||||||
|
)
|
||||||
|
spoken = gateway.clean_for_speech(text)
|
||||||
|
self.assertIn("Postleitzahl sieben sechs vier drei sieben", spoken)
|
||||||
|
self.assertIn("23. August", spoken)
|
||||||
|
self.assertIn("Höchstwert 22 Grad, Tiefstwert 10 Grad", spoken)
|
||||||
|
self.assertIn("0 Prozent Regen", spoken)
|
||||||
|
self.assertIn("wetter Punkt C O M", spoken)
|
||||||
|
self.assertIn("wetteronline Punkt D E", spoken)
|
||||||
|
|
||||||
|
def test_full_date_and_raw_url_are_normalized(self):
|
||||||
|
spoken = gateway.clean_for_speech(
|
||||||
|
"Am 23.08.2026 steht es auf https://www.example.org/path?q=1."
|
||||||
|
)
|
||||||
|
self.assertIn("23. August 2026", spoken)
|
||||||
|
self.assertIn("example Punkt O R G", spoken)
|
||||||
|
self.assertNotIn("https", spoken)
|
||||||
|
|
||||||
|
|
||||||
class FallbackTests(unittest.TestCase):
|
class FallbackTests(unittest.TestCase):
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
|
|||||||
@@ -116,6 +116,16 @@ ENGLISH_PRONUNCIATIONS = {
|
|||||||
"gib": "gigabytes",
|
"gib": "gigabytes",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
GERMAN_MONTHS = {
|
||||||
|
1: "Januar", 2: "Februar", 3: "März", 4: "April",
|
||||||
|
5: "Mai", 6: "Juni", 7: "Juli", 8: "August",
|
||||||
|
9: "September", 10: "Oktober", 11: "November", 12: "Dezember",
|
||||||
|
}
|
||||||
|
GERMAN_DIGITS = {
|
||||||
|
"0": "null", "1": "eins", "2": "zwei", "3": "drei", "4": "vier",
|
||||||
|
"5": "fünf", "6": "sechs", "7": "sieben", "8": "acht", "9": "neun",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def _request(url: str, *, payload: dict | None = None,
|
def _request(url: str, *, payload: dict | None = None,
|
||||||
timeout: float = 10) -> tuple[bytes, str]:
|
timeout: float = 10) -> tuple[bytes, str]:
|
||||||
@@ -190,6 +200,74 @@ def segment_languages(text: str) -> list[tuple[str, str]]:
|
|||||||
return merged
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
def _spoken_date(match: re.Match) -> str:
|
||||||
|
day = int(match.group(1))
|
||||||
|
month = int(match.group(2))
|
||||||
|
year = match.group(3)
|
||||||
|
month_name = GERMAN_MONTHS.get(month)
|
||||||
|
if not month_name or not 1 <= day <= 31:
|
||||||
|
return match.group(0)
|
||||||
|
result = f"{day}. {month_name}"
|
||||||
|
if year:
|
||||||
|
result += f" {year}"
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _spell_digits(value: str) -> str:
|
||||||
|
return " ".join(GERMAN_DIGITS[digit] for digit in value)
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_for_german_speech(text: str) -> str:
|
||||||
|
"""Turn common visual notation into unambiguous spoken German."""
|
||||||
|
text = re.sub(
|
||||||
|
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?[ \t]*/[ \t]*"
|
||||||
|
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||||
|
r"Höchstwert \1 Grad, Tiefstwert \2 Grad",
|
||||||
|
text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"(-?\d+(?:[,.]\d+)?)[ \t]*°[ \t]*(?:C)?",
|
||||||
|
r"\1 Grad",
|
||||||
|
text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\b([0-3]?\d)\.([01]?\d)\.(?:(\d{4})\b)?",
|
||||||
|
_spoken_date,
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\bPLZ\s+(\d{5})\b",
|
||||||
|
lambda match: "Postleitzahl " + _spell_digits(match.group(1)),
|
||||||
|
text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\((\d{5})\)",
|
||||||
|
lambda match: "(Postleitzahl " + _spell_digits(match.group(1)) + ")",
|
||||||
|
text,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"https?://(?:www\.)?([^/\s)]+)(?:/[^\s)]*)?",
|
||||||
|
lambda match: match.group(1),
|
||||||
|
text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
)
|
||||||
|
text = re.sub(
|
||||||
|
r"\b([A-Za-z0-9][A-Za-z0-9-]*(?:\.[A-Za-z0-9-]+)*)"
|
||||||
|
r"\.(de|com|org|net|io|ai)\b",
|
||||||
|
lambda match: (
|
||||||
|
match.group(1).replace(".", " Punkt ")
|
||||||
|
+ " Punkt " + " ".join(match.group(2).upper())
|
||||||
|
),
|
||||||
|
text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
)
|
||||||
|
text = re.sub(r"(\d)\s*%", r"\1 Prozent", text)
|
||||||
|
return text
|
||||||
|
|
||||||
|
|
||||||
def clean_for_speech(text: str) -> str:
|
def clean_for_speech(text: str) -> str:
|
||||||
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
"""Remove visual markup that makes long TTS output unstable or noisy."""
|
||||||
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
|
||||||
@@ -200,12 +278,15 @@ def clean_for_speech(text: str) -> str:
|
|||||||
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
|
||||||
text = text.replace("→", ". ").replace("←", ". ")
|
text = text.replace("→", ". ").replace("←", ". ")
|
||||||
text = text.replace("–", " - ").replace("—", " - ")
|
text = text.replace("–", " - ").replace("—", " - ")
|
||||||
|
if DEFAULT_LANGUAGE == "de":
|
||||||
|
text = normalize_for_german_speech(text)
|
||||||
text = "".join(
|
text = "".join(
|
||||||
char for char in text
|
char for char in text
|
||||||
if unicodedata.category(char) not in {"So", "Cs"}
|
if unicodedata.category(char) not in {"So", "Cs"}
|
||||||
)
|
)
|
||||||
text = re.sub(r"[ \t]+", " ", text)
|
text = re.sub(r"[ \t]+", " ", text)
|
||||||
text = re.sub(r"\s*\n+\s*", ". ", text)
|
text = re.sub(r"\s*\n+\s*", ". ", text)
|
||||||
|
text = re.sub(r"([:;])\s*\.", r"\1", text)
|
||||||
text = re.sub(r"(?:\.\s*){2,}", ". ", text)
|
text = re.sub(r"(?:\.\s*){2,}", ". ", text)
|
||||||
return text.strip()
|
return text.strip()
|
||||||
|
|
||||||
|
|||||||
Reference in new issue
Block a user