Stabilize XTTS sentence endings
This commit is contained in:
@@ -419,10 +419,23 @@ def _speaker_conditioning() -> dict:
|
||||
return SPEAKER_CONDITIONING
|
||||
|
||||
|
||||
def _stabilize_xtts_ending(text: str) -> str:
|
||||
"""Give XTTS a reliable stop cue without changing the visible answer.
|
||||
|
||||
XTTS v2 can continue with invented syllables after a terminal full stop,
|
||||
especially in German. A terminal semicolon is tokenized as a stronger,
|
||||
more reliable boundary while retaining neutral sentence intonation.
|
||||
"""
|
||||
spoken = text.rstrip()
|
||||
if spoken.endswith((".", "!", "?", ";", ":")):
|
||||
spoken = spoken[:-1].rstrip()
|
||||
return f"{spoken};"
|
||||
|
||||
|
||||
def _xtts_pcm(text: str, language: str, conditioning: dict) -> bytes:
|
||||
payload = {
|
||||
**conditioning,
|
||||
"text": text,
|
||||
"text": _stabilize_xtts_ending(text),
|
||||
"language": language,
|
||||
"add_wav_header": True,
|
||||
"stream_chunk_size": "20",
|
||||
|
||||
Reference in New Issue
Block a user