Preserve sentence prosody in local TTS
This commit is contained in:
@@ -38,10 +38,10 @@ MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
||||
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||
# XTTS can insert multi-second silences or truncate the remainder when a
|
||||
# moderately long German paragraph is sent as one request. Keeping requests
|
||||
# close to one sentence proved substantially more stable with Annmarie Nele.
|
||||
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "60"))
|
||||
# XTTS loses natural prosody when a sentence is synthesized as many tiny
|
||||
# requests: every request starts a fresh utterance. Keep complete sentences
|
||||
# together and use this only as a safety ceiling for unusually long sentences.
|
||||
CHUNK_CHARS = int(os.getenv("XTTS_CHUNK_CHARS", "420"))
|
||||
TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
|
||||
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
|
||||
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
|
||||
@@ -318,38 +318,51 @@ def clean_for_speech(text: str) -> str:
|
||||
|
||||
|
||||
def _split_chunk(text: str, limit: int = CHUNK_CHARS) -> list[str]:
|
||||
"""Split at natural pauses and keep every XTTS request comfortably short."""
|
||||
"""Return sentence-sized XTTS requests with a conservative hard ceiling.
|
||||
|
||||
A sentence is deliberately never combined with the following sentence.
|
||||
Overlong sentences are split at clause boundaries first and at words only
|
||||
as a last resort. This preserves XTTS prosody without exposing it to an
|
||||
unbounded paragraph.
|
||||
"""
|
||||
text = text.strip()
|
||||
if not text:
|
||||
return []
|
||||
if len(text) <= limit:
|
||||
return [text]
|
||||
|
||||
pieces = re.split(r"(?<=[.!?;:])\s+|\s+(?=\d+[.)]\s)", text)
|
||||
sentences = re.split(r"(?<=[.!?])\s+|\s+(?=\d+[.)]\s)", text)
|
||||
chunks: list[str] = []
|
||||
current = ""
|
||||
for piece in pieces:
|
||||
piece = piece.strip()
|
||||
if not piece:
|
||||
for sentence in sentences:
|
||||
sentence = sentence.strip()
|
||||
if not sentence:
|
||||
continue
|
||||
if len(piece) > limit:
|
||||
words = piece.split()
|
||||
for word in words:
|
||||
if len(sentence) <= limit:
|
||||
chunks.append(sentence)
|
||||
continue
|
||||
|
||||
# Retain commas in the preceding clause so XTTS can reproduce the
|
||||
# intended pause. Semicolons and colons were normalized earlier.
|
||||
clauses = re.split(r"(?<=,)\s+", sentence)
|
||||
current = ""
|
||||
for clause in clauses:
|
||||
clause = clause.strip()
|
||||
candidate = f"{current} {clause}".strip()
|
||||
if current and len(candidate) > limit:
|
||||
chunks.append(current)
|
||||
current = ""
|
||||
if len(clause) <= limit:
|
||||
current = f"{current} {clause}".strip()
|
||||
continue
|
||||
|
||||
# A clause without a usable pause can still exceed the ceiling.
|
||||
for word in clause.split():
|
||||
candidate = f"{current} {word}".strip()
|
||||
if current and len(candidate) > limit:
|
||||
chunks.append(current)
|
||||
current = word
|
||||
else:
|
||||
current = candidate
|
||||
continue
|
||||
candidate = f"{current} {piece}".strip()
|
||||
if current and len(candidate) > limit:
|
||||
if current:
|
||||
chunks.append(current)
|
||||
current = piece
|
||||
else:
|
||||
current = candidate
|
||||
if current:
|
||||
chunks.append(current)
|
||||
return chunks
|
||||
|
||||
|
||||
@@ -366,6 +379,14 @@ def prepare_segments(text: str) -> list[tuple[str, str]]:
|
||||
if language == "en":
|
||||
spoken = ENGLISH_PRONUNCIATIONS.get(segment.strip().lower(), segment)
|
||||
for chunk in _split_chunk(spoken):
|
||||
if not re.search(r"\w", chunk, flags=re.UNICODE):
|
||||
if prepared:
|
||||
previous_language, previous_text = prepared[-1]
|
||||
prepared[-1] = (
|
||||
previous_language,
|
||||
previous_text.rstrip() + chunk.strip(),
|
||||
)
|
||||
continue
|
||||
prepared.append((language, chunk))
|
||||
return prepared
|
||||
|
||||
|
||||
Reference in New Issue
Block a user