Compress hallucinated XTTS silence

This commit is contained in:
Mikei386 committed 2026-09-03 23:46:32 +02:00
1 parent 435c59da41
commit 5afdf46a7c
3 files changed
+49 -1

No files matched your search

@@ -184,6 +184,19 @@ class AudioJoinTests(unittest.TestCase):
sentence = gateway._join_pcm([("Fertig.", spoken), ("Weiter", spoken)])
self.assertGreater(len(sentence), len(inline))
def test_long_internal_silence_is_shortened(self):
spoken = self.pcm([900] * 1000)
long_silence = self.pcm([0] * int(24000 * 1.5))
result = gateway._compress_internal_silence(spoken + long_silence + spoken)
expected_keep = int(24000 * gateway.INTERNAL_SILENCE_KEEP_MS / 1000) * 2
self.assertEqual(len(result), len(spoken) * 2 + expected_keep)
def test_natural_short_pause_is_preserved(self):
spoken = self.pcm([900] * 1000)
short_silence = self.pcm([0] * int(24000 * 0.3))
original = spoken + short_silence + spoken
self.assertEqual(gateway._compress_internal_silence(original), original)
if __name__ == "__main__":
unittest.main()
+33 -1
View File
@@ -46,6 +46,9 @@ TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
SENTENCE_PAUSE_MS = int(os.getenv("XTTS_SENTENCE_PAUSE_MS", "65"))
INTERNAL_SILENCE_THRESHOLD = int(os.getenv("XTTS_INTERNAL_SILENCE_THRESHOLD", "512"))
INTERNAL_SILENCE_TRIGGER_MS = int(os.getenv("XTTS_INTERNAL_SILENCE_TRIGGER_MS", "650"))
INTERNAL_SILENCE_KEEP_MS = int(os.getenv("XTTS_INTERNAL_SILENCE_KEEP_MS", "220"))
SYNTHESIS_LOCK = threading.Lock()
STATE_LOCK = threading.Lock()
@@ -432,6 +435,35 @@ def _fade_edge(pcm: bytes, *, fade_in: bool = False,
return samples.tobytes()
def _compress_internal_silence(pcm: bytes) -> bytes:
"""Shorten XTTS silence hallucinations while preserving normal pauses."""
samples = array("h")
samples.frombytes(pcm)
if not samples:
return pcm
trigger = int(24000 * max(0, INTERNAL_SILENCE_TRIGGER_MS) / 1000)
keep = int(24000 * max(0, INTERNAL_SILENCE_KEEP_MS) / 1000)
if trigger <= 0 or keep >= trigger:
return pcm
output = array("h")
index = 0
while index < len(samples):
if abs(samples[index]) > INTERNAL_SILENCE_THRESHOLD:
output.append(samples[index])
index += 1
continue
end = index + 1
while end < len(samples) and abs(samples[end]) <= INTERNAL_SILENCE_THRESHOLD:
end += 1
run = end - index
if run >= trigger:
output.extend(samples[index:index + keep])
else:
output.extend(samples[index:end])
index = end
return output.tobytes()
def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
"""Join clips without clicks; pause only at real sentence boundaries."""
if not parts:
@@ -459,7 +491,7 @@ def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
faded = _fade_edge(bytes(output), fade_out=True)
output[:] = faded
output.extend(_fade_edge(pcm, fade_in=True))
return bytes(output)
return _compress_internal_silence(bytes(output))
def _wav(pcm: bytes) -> bytes: