Stabilize XTTS sentence endings

This commit is contained in:
Mikei386
2026-09-04 09:30:04 +02:00
parent 8dac735680
commit 744a207e5a
2 changed files with 24 additions and 1 deletions
+14 -1
View File
@@ -419,10 +419,23 @@ def _speaker_conditioning() -> dict:
return SPEAKER_CONDITIONING
def _stabilize_xtts_ending(text: str) -> str:
"""Give XTTS a reliable stop cue without changing the visible answer.
XTTS v2 can continue with invented syllables after a terminal full stop,
especially in German. A terminal semicolon is tokenized as a stronger,
more reliable boundary while retaining neutral sentence intonation.
"""
spoken = text.rstrip()
if spoken.endswith((".", "!", "?", ";", ":")):
spoken = spoken[:-1].rstrip()
return f"{spoken};"
def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes:
payload = {
**conditioning,
"text": text,
"text": _stabilize_xtts_ending(text),
"language": language,
"add_wav_header": True,
"stream_chunk_size": "20",