Stabilize German XTTS speech output

This commit is contained in:
Mikei386
2026-08-23 13:16:17 +02:00
parent f90fc93f9c
commit a067b79ed2
6 changed files with 264 additions and 17 deletions
+184 -11
View File
@@ -14,9 +14,11 @@ import re
import subprocess
import threading
import time
import unicodedata
import urllib.error
import urllib.request
import wave
from array import array
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -28,13 +30,19 @@ PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
CODE_SWITCH_ENABLED = os.getenv("TTS_CODE_SWITCH_ENABLED", "false").lower() \
in {"1", "true", "yes", "on"}
MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
SILENCE_MS = int(os.getenv("XTTS_SEGMENT_SILENCE_MS", "20"))
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "220"))
TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
SENTENCE_PAUSE_MS = int(os.getenv("XTTS_SENTENCE_PAUSE_MS", "65"))
SYNTHESIS_LOCK = threading.Lock()
STATE_LOCK = threading.Lock()
@@ -50,16 +58,27 @@ STATE = {
# Prefer full compounds to isolated terms. This keeps switches infrequent and
# avoids making mixed-language speech sound like a sequence of separate clips.
ENGLISH_TERMS = (
"unsupported image format", "premature end of JPEG", "incomplete scan",
"Docker Containers", "Docker-Containers", "Docker Containern",
"Docker-Containern", "Health Checks", "Health-Checks",
"Restart Loops", "Restart-Loops", "False Positive", "Delivery Errors",
"DeliveryErrors", "Ack Problem", "Ack-Problem", "I/O timeout",
"Parity Check", "Parity-Check", "Disk disabled", "Disk invalid",
"Home Assistant", "Open WebUI", "OpenWebUI", "Unraid Dashboard",
"Unraid-Dashboard", "Docker Container", "Docker-Container",
"Server Log", "Server-Log", "GitHub Repository", "GitHub Repo",
"WireGuard Tunnel", "Cron Job", "Cronjob", "Home Server",
"API Key", "Tool Calling", "Context Window", "Prompt Injection",
"Unraid", "Docker", "Container", "Dashboard", "Server", "Log",
"Unraid", "Docker", "Containern", "Containers", "Container",
"Crashes", "healthy", "disabled", "invalid", "Dashboard", "Server",
"Logs", "Log", "Matches", "up",
"OpenAI", "GitHub", "WireGuard", "Linux", "Debian", "Frontend",
"Backend", "Router", "Browser", "Web", "Token", "Prompt", "Context",
"Model", "Image", "Tool", "Workflow", "Benchmark", "Streaming",
"SSH", "MCP", "API", "CPU", "GPU", "VRAM", "RAM", "HTTP", "HTTPS",
"HomeAssistant", "ESPHome", "iGotify", "Immich", "Vaultwarden",
"UniFi", "UptimeKuma", "go2rtc", "Zigbee", "SONOFF", "eWeLink",
"RTSP", "JPEG", "NVMe", "GiB",
)
TERM_PATTERN = re.compile(
r"(?<![\w])(" + "|".join(
@@ -77,6 +96,26 @@ ENGLISH_MARKERS = {
"of", "on", "or", "please", "the", "this", "to", "with", "you",
}
ENGLISH_PRONUNCIATIONS = {
"containern": "containers",
"docker containern": "Docker containers",
"docker-containern": "Docker containers",
"docker-container": "Docker container",
"docker-containers": "Docker containers",
"health-checks": "health checks",
"restart-loops": "restart loops",
"deliveryerrors": "delivery errors",
"ack-problem": "ack problem",
"i/o timeout": "I O timeout",
"parity-check": "parity check",
"homeassistant": "Home Assistant",
"openwebui": "Open Web U I",
"rtsp": "R T S P",
"jpeg": "J peg",
"nvme": "N V M E",
"gib": "gigabytes",
}
def _request(url: str, *, payload: dict | None = None,
timeout: float = 10) -> tuple[bytes, str]:
@@ -124,6 +163,8 @@ def segment_languages(text: str) -> list[tuple[str, str]]:
return [("en", text)]
if DEFAULT_LANGUAGE != "de":
return [(DEFAULT_LANGUAGE, text)]
if not CODE_SWITCH_ENABLED:
return [("de", text)]
segments: list[tuple[str, str]] = []
cursor = 0
@@ -149,6 +190,79 @@ def segment_languages(text: str) -> list[tuple[str, str]]:
return merged
def clean_for_speech(text: str) -> str:
"""Remove visual markup that makes long TTS output unstable or noisy."""
text = re.sub(r"```.*?```", " Code block. ", text, flags=re.DOTALL)
text = re.sub(r"`([^`]+)`", r"\1", text)
text = re.sub(r"!\[([^]]*)\]\([^)]+\)", r"\1", text)
text = re.sub(r"\[([^]]+)\]\([^)]+\)", r"\1", text)
text = re.sub(r"(?m)^\s{0,3}#{1,6}\s*", "", text)
text = re.sub(r"(?m)^\s*[-*+]\s+", "", text)
text = text.replace("→", ". ").replace("←", ". ")
text = text.replace("–", " - ").replace("—", " - ")
text = "".join(
char for char in text
if unicodedata.category(char) not in {"So", "Cs"}
)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\s*\n+\s*", ". ", text)
text = re.sub(r"(?:\.\s*){2,}", ". ", text)
return text.strip()
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
"""Split at natural pauses and keep every XTTS request comfortably short."""
text = text.strip()
if not text:
return []
if len(text) <= limit:
return [text]
pieces = re.split(r"(?<=[.!?;:])\s+|\s+(?=\d+[.)]\s)", text)
chunks: list[str] = []
current = ""
for piece in pieces:
piece = piece.strip()
if not piece:
continue
if len(piece) > limit:
words = piece.split()
for word in words:
candidate = f"{current} {word}".strip()
if current and len(candidate) > limit:
chunks.append(current)
current = word
else:
current = candidate
continue
candidate = f"{current} {piece}".strip()
if current and len(candidate) > limit:
chunks.append(current)
current = piece
else:
current = candidate
if current:
chunks.append(current)
return chunks
def prepare_segments(text: str) -> list[tuple[str, str]]:
"""Prepare short, deterministic German/English XTTS requests."""
prepared: list[tuple[str, str]] = []
for language, segment in segment_languages(clean_for_speech(text)):
if not re.search(r"\w", segment, flags=re.UNICODE):
if prepared:
previous_language, previous_text = prepared[-1]
prepared[-1] = (previous_language, previous_text + segment.strip())
continue
spoken = segment
if language == "en":
spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment)
for chunk in _split_chunk(spoken):
prepared.append((language, chunk))
return prepared
def _speaker_conditioning() -> dict:
global SPEAKER_CONDITIONING
with SPEAKER_LOCK:
@@ -180,6 +294,70 @@ def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes:
return audio[44:]
def _trim_pcm(pcm: bytes) -> bytes:
"""Trim generated edge silence while retaining a small safety padding."""
samples = array("h")
samples.frombytes(pcm)
if not samples:
return pcm
first = next((i for i, value in enumerate(samples)
if abs(value) >= TRIM_THRESHOLD), 0)
last = next((i for i in range(len(samples) - 1, -1, -1)
if abs(samples[i]) >= TRIM_THRESHOLD), len(samples) - 1)
padding = int(24000 * max(0, TRIM_PADDING_MS) / 1000)
first = max(0, first - padding)
last = min(len(samples) - 1, last + padding)
return samples[first:last + 1].tobytes()
def _fade_edge(pcm: bytes, *, fade_in: bool = False,
fade_out: bool = False) -> bytes:
samples = array("h")
samples.frombytes(pcm)
count = min(len(samples), int(24000 * max(0, CROSSFADE_MS) / 1000))
if count <= 1:
return pcm
if fade_in:
for index in range(count):
samples[index] = int(samples[index] * index / (count - 1))
if fade_out:
start = len(samples) - count
for index in range(count):
samples[start + index] = int(
samples[start + index] * (count - 1 - index) / (count - 1))
return samples.tobytes()
def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
"""Join clips without clicks; pause only at real sentence boundaries."""
if not parts:
return b""
output = bytearray()
sentence_silence = b"\x00\x00" * int(
24000 * max(0, SENTENCE_PAUSE_MS) / 1000)
for index, (text, pcm) in enumerate(parts):
pcm = _trim_pcm(pcm)
previous_ends_sentence = index > 0 and bool(
re.search(r"[.!?;:]\s*$", parts[index - 1][0]))
if index == 0:
output.extend(_fade_edge(pcm, fade_in=True))
elif previous_ends_sentence:
if output:
faded = _fade_edge(bytes(output), fade_out=True)
output[:] = faded
output.extend(sentence_silence)
output.extend(_fade_edge(pcm, fade_in=True))
else:
# Language switches inside a sentence get no artificial pause.
# Small fades remove the discontinuity that otherwise sounds like
# a high click or beep between independently generated clips.
if output:
faded = _fade_edge(bytes(output), fade_out=True)
output[:] = faded
output.extend(_fade_edge(pcm, fade_in=True))
return bytes(output)
def _wav(pcm: bytes) -> bytes:
output = io.BytesIO()
with wave.open(output, "wb") as wav_file:
@@ -211,19 +389,14 @@ def _convert(wav_bytes: bytes, output_format: str, speed: float) -> tuple[bytes,
def synthesize_xtts(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
conditioning = _speaker_conditioning()
pcm_parts: list[bytes] = []
silence = b"\x00\x00" * int(24000 * max(0, SILENCE_MS) / 1000)
for language, segment in segment_languages(text):
pcm_parts: list[tuple[str, bytes]] = []
for language, segment in prepare_segments(text):
if not segment.strip():
continue
pcm_parts.append(_xtts_pcm(segment, language, conditioning))
if silence:
pcm_parts.append(silence)
if pcm_parts and silence:
pcm_parts.pop()
pcm_parts.append((segment, _xtts_pcm(segment, language, conditioning)))
if not pcm_parts:
raise RuntimeError("no speech segments generated")
return _convert(_wav(b"".join(pcm_parts)), output_format, speed)
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
def synthesize_piper(text: str, output_format: str,