mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-03 17:15:19 -07:00
fix(kokoro): trim edges only, keep inter-segment gaps
Registering Kokoro with needs_trim=True routed its output through the generic trim_tts_output, whose 1s internal-silence cut was tuned for Chatterbox hallucinations. KPipeline synthesizes newline- and token-limit-separated segments independently, and the ~0.3s lead plus ~0.7s tail pads at each boundary add up to 1.2s of silence, so with af_sarah a four-paragraph script came back as its first paragraph only (10.9s -> 1.8s) and a 51s text lost half its segments. Drop the needs_trim flag, keep the in-backend trim, and give trim_tts_output a max_internal_silence_ms=None mode that only trims the leading and trailing pads. Tests updated for the new behaviour, with a two-segment fake pipeline and an explicit internal-gap case.
This commit is contained in:
committed by
capy-ai-staging[bot]
parent
615aeaeb35
commit
ae300c5316
+14
-13
@@ -153,7 +153,7 @@ def trim_tts_output(
|
||||
frame_ms: int = 20,
|
||||
silence_threshold_db: float = -40.0,
|
||||
min_silence_ms: int = 200,
|
||||
max_internal_silence_ms: int = 1000,
|
||||
max_internal_silence_ms: int | None = 1000,
|
||||
fade_ms: int = 30,
|
||||
) -> np.ndarray:
|
||||
"""
|
||||
@@ -170,7 +170,8 @@ def trim_tts_output(
|
||||
frame_ms: Frame size for RMS energy calculation
|
||||
silence_threshold_db: dB threshold below which a frame is silence
|
||||
min_silence_ms: Minimum trailing silence to keep
|
||||
max_internal_silence_ms: Cut after any silence gap longer than this
|
||||
max_internal_silence_ms: Cut after any silence gap longer than this.
|
||||
``None`` disables the internal cut and only trims the edges.
|
||||
fade_ms: Cosine fade-out duration in ms
|
||||
|
||||
Returns:
|
||||
@@ -200,18 +201,18 @@ def trim_tts_output(
|
||||
break
|
||||
|
||||
# Walk forward from first speech; cut at long internal silence gaps
|
||||
max_silence_frames = int(max_internal_silence_ms / frame_ms)
|
||||
consecutive_silence = 0
|
||||
cut_frame = n_frames
|
||||
|
||||
for i in range(first_speech, n_frames):
|
||||
if is_speech[i]:
|
||||
consecutive_silence = 0
|
||||
else:
|
||||
consecutive_silence += 1
|
||||
if consecutive_silence >= max_silence_frames:
|
||||
cut_frame = i - consecutive_silence + 1
|
||||
break
|
||||
if max_internal_silence_ms is not None:
|
||||
max_silence_frames = int(max_internal_silence_ms / frame_ms)
|
||||
consecutive_silence = 0
|
||||
for i in range(first_speech, n_frames):
|
||||
if is_speech[i]:
|
||||
consecutive_silence = 0
|
||||
else:
|
||||
consecutive_silence += 1
|
||||
if consecutive_silence >= max_silence_frames:
|
||||
cut_frame = i - consecutive_silence + 1
|
||||
break
|
||||
|
||||
# Trim trailing silence from the cut point
|
||||
min_silence_frames = int(min_silence_ms / frame_ms)
|
||||
|
||||
Reference in New Issue
Block a user