Compress hallucinated XTTS silence
This commit is contained in:
@@ -700,6 +700,9 @@ services:
|
|||||||
# Short sentence-sized requests avoid long generated silences and
|
# Short sentence-sized requests avoid long generated silences and
|
||||||
# truncated weather/status summaries with Annmarie Nele.
|
# truncated weather/status summaries with Annmarie Nele.
|
||||||
XTTS_CHUNK_CHARS: "60"
|
XTTS_CHUNK_CHARS: "60"
|
||||||
|
# XTTS occasionally inserts multi-second silence inside a phrase.
|
||||||
|
XTTS_INTERNAL_SILENCE_TRIGGER_MS: "650"
|
||||||
|
XTTS_INTERNAL_SILENCE_KEEP_MS: "220"
|
||||||
PIPER_TIMEOUT: "120"
|
PIPER_TIMEOUT: "120"
|
||||||
networks: [frontend]
|
networks: [frontend]
|
||||||
depends_on:
|
depends_on:
|
||||||
|
|||||||
@@ -184,6 +184,19 @@ class AudioJoinTests(unittest.TestCase):
|
|||||||
sentence = gateway._join_pcm([("Fertig.", spoken), ("Weiter", spoken)])
|
sentence = gateway._join_pcm([("Fertig.", spoken), ("Weiter", spoken)])
|
||||||
self.assertGreater(len(sentence), len(inline))
|
self.assertGreater(len(sentence), len(inline))
|
||||||
|
|
||||||
|
def test_long_internal_silence_is_shortened(self):
|
||||||
|
spoken = self.pcm([900] * 1000)
|
||||||
|
long_silence = self.pcm([0] * int(24000 * 1.5))
|
||||||
|
result = gateway._compress_internal_silence(spoken + long_silence + spoken)
|
||||||
|
expected_keep = int(24000 * gateway.INTERNAL_SILENCE_KEEP_MS / 1000) * 2
|
||||||
|
self.assertEqual(len(result), len(spoken) * 2 + expected_keep)
|
||||||
|
|
||||||
|
def test_natural_short_pause_is_preserved(self):
|
||||||
|
spoken = self.pcm([900] * 1000)
|
||||||
|
short_silence = self.pcm([0] * int(24000 * 0.3))
|
||||||
|
original = spoken + short_silence + spoken
|
||||||
|
self.assertEqual(gateway._compress_internal_silence(original), original)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
@@ -46,6 +46,9 @@ TRIM_THRESHOLD = int(os.getenv("XTTS_TRIM_THRESHOLD", "90"))
|
|||||||
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
|
TRIM_PADDING_MS = int(os.getenv("XTTS_TRIM_PADDING_MS", "18"))
|
||||||
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
|
CROSSFADE_MS = int(os.getenv("XTTS_CROSSFADE_MS", "8"))
|
||||||
SENTENCE_PAUSE_MS = int(os.getenv("XTTS_SENTENCE_PAUSE_MS", "65"))
|
SENTENCE_PAUSE_MS = int(os.getenv("XTTS_SENTENCE_PAUSE_MS", "65"))
|
||||||
|
INTERNAL_SILENCE_THRESHOLD = int(os.getenv("XTTS_INTERNAL_SILENCE_THRESHOLD", "512"))
|
||||||
|
INTERNAL_SILENCE_TRIGGER_MS = int(os.getenv("XTTS_INTERNAL_SILENCE_TRIGGER_MS", "650"))
|
||||||
|
INTERNAL_SILENCE_KEEP_MS = int(os.getenv("XTTS_INTERNAL_SILENCE_KEEP_MS", "220"))
|
||||||
|
|
||||||
SYNTHESIS_LOCK = threading.Lock()
|
SYNTHESIS_LOCK = threading.Lock()
|
||||||
STATE_LOCK = threading.Lock()
|
STATE_LOCK = threading.Lock()
|
||||||
@@ -432,6 +435,35 @@ def _fade_edge(pcm: bytes, *, fade_in: bool = False,
|
|||||||
return samples.tobytes()
|
return samples.tobytes()
|
||||||
|
|
||||||
|
|
||||||
|
def _compress_internal_silence(pcm: bytes) -> bytes:
|
||||||
|
"""Shorten XTTS silence hallucinations while preserving normal pauses."""
|
||||||
|
samples = array("h")
|
||||||
|
samples.frombytes(pcm)
|
||||||
|
if not samples:
|
||||||
|
return pcm
|
||||||
|
trigger = int(24000 * max(0, INTERNAL_SILENCE_TRIGGER_MS) / 1000)
|
||||||
|
keep = int(24000 * max(0, INTERNAL_SILENCE_KEEP_MS) / 1000)
|
||||||
|
if trigger <= 0 or keep >= trigger:
|
||||||
|
return pcm
|
||||||
|
output = array("h")
|
||||||
|
index = 0
|
||||||
|
while index < len(samples):
|
||||||
|
if abs(samples[index]) > INTERNAL_SILENCE_THRESHOLD:
|
||||||
|
output.append(samples[index])
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
end = index + 1
|
||||||
|
while end < len(samples) and abs(samples[end]) <= INTERNAL_SILENCE_THRESHOLD:
|
||||||
|
end += 1
|
||||||
|
run = end - index
|
||||||
|
if run >= trigger:
|
||||||
|
output.extend(samples[index:index + keep])
|
||||||
|
else:
|
||||||
|
output.extend(samples[index:end])
|
||||||
|
index = end
|
||||||
|
return output.tobytes()
|
||||||
|
|
||||||
|
|
||||||
def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
|
def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
|
||||||
"""Join clips without clicks; pause only at real sentence boundaries."""
|
"""Join clips without clicks; pause only at real sentence boundaries."""
|
||||||
if not parts:
|
if not parts:
|
||||||
@@ -459,7 +491,7 @@ def _join_pcm(parts: list[tuple[str, bytes]]) -> bytes:
|
|||||||
faded = _fade_edge(bytes(output), fade_out=True)
|
faded = _fade_edge(bytes(output), fade_out=True)
|
||||||
output[:] = faded
|
output[:] = faded
|
||||||
output.extend(_fade_edge(pcm, fade_in=True))
|
output.extend(_fade_edge(pcm, fade_in=True))
|
||||||
return bytes(output)
|
return _compress_internal_silence(bytes(output))
|
||||||
|
|
||||||
|
|
||||||
def _wav(pcm: bytes) -> bytes:
|
def _wav(pcm: bytes) -> bytes:
|
||||||
|
|||||||
Reference in New Issue
Block a user