From ae300c531611ecfee961bdccee395bdd25f0afb6 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:44:57 +0000 Subject: [PATCH] fix(kokoro): trim edges only, keep inter-segment gaps Registering Kokoro with needs_trim=True routed its output through the generic trim_tts_output, whose 1s internal-silence cut was tuned for Chatterbox hallucinations. KPipeline synthesizes newline- and token-limit-separated segments independently, and the ~0.3s lead plus ~0.7s tail pads at each boundary add up to 1.2s of silence, so with af_sarah a four-paragraph script came back as its first paragraph only (10.9s -> 1.8s) and a 51s text lost half its segments. Drop the needs_trim flag, keep the in-backend trim, and give trim_tts_output a max_internal_silence_ms=None mode that only trims the leading and trailing pads. Tests updated for the new behaviour, with a two-segment fake pipeline and an explicit internal-gap case. --- backend/backends/__init__.py | 1 - backend/backends/kokoro_backend.py | 7 +++++- backend/tests/test_kokoro_trim.py | 36 +++++++++++++++++++++++------- backend/utils/audio.py | 27 +++++++++++----------- 4 files changed, 48 insertions(+), 23 deletions(-) diff --git a/backend/backends/__init__.py b/backend/backends/__init__.py index 8f08c03f..142c430c 100644 --- a/backend/backends/__init__.py +++ b/backend/backends/__init__.py @@ -369,7 +369,6 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]: engine="kokoro", hf_repo_id="hexgrad/Kokoro-82M", size_mb=350, - needs_trim=True, languages=["en", "es", "fr", "hi", "it", "pt", "ja", "zh"], ), ] diff --git a/backend/backends/kokoro_backend.py b/backend/backends/kokoro_backend.py index 4481cad7..1466016a 100644 --- a/backend/backends/kokoro_backend.py +++ b/backend/backends/kokoro_backend.py @@ -291,7 +291,12 @@ class KokoroTTSBackend: audio = np.concatenate(audio_chunks).astype(np.float32) from ..utils.audio import trim_tts_output - audio = trim_tts_output(audio, sample_rate=KOKORO_SAMPLE_RATE) + # Edge-only trim: Kokoro pads ~0.3s before and ~0.7s after speech. + # The internal-gap cut is disabled because KPipeline synthesizes + # newline/token-limit segments independently and the pads at each + # segment boundary add up to >1s of silence; with the default cut + # everything after the first segment would be dropped. + audio = trim_tts_output(audio, sample_rate=KOKORO_SAMPLE_RATE, max_internal_silence_ms=None) return audio, KOKORO_SAMPLE_RATE return await asyncio.to_thread(_generate_sync) diff --git a/backend/tests/test_kokoro_trim.py b/backend/tests/test_kokoro_trim.py index 550b561d..779b31cf 100644 --- a/backend/tests/test_kokoro_trim.py +++ b/backend/tests/test_kokoro_trim.py @@ -7,12 +7,30 @@ from backend.backends import engine_needs_trim, get_model_config from backend.utils.audio import trim_tts_output -def test_kokoro_engine_needs_trim_enabled(): - """Verify Kokoro engine is registered with needs_trim=True in model config.""" - assert engine_needs_trim("kokoro") is True +def test_kokoro_engine_does_not_use_generic_trim(): + """Kokoro trims inside the backend; the generic needs_trim path would cut + multi-segment output at the first >1s inter-segment gap.""" + assert engine_needs_trim("kokoro") is False config = get_model_config("kokoro") assert config is not None - assert config.needs_trim is True + assert config.needs_trim is False + + +def test_trim_tts_output_edge_only_keeps_internal_gaps(): + """With max_internal_silence_ms=None only the edges are trimmed.""" + sr = 24000 + speech = np.full(int(sr * 1.0), 0.2, dtype=np.float32) + gap = np.zeros(int(sr * 1.5), dtype=np.float32) # longer than the 1s default cut + pad = np.zeros(int(sr * 0.5), dtype=np.float32) + raw_audio = np.concatenate([pad, speech, gap, speech, pad]) + + default_trim = trim_tts_output(raw_audio, sample_rate=sr) + edge_trim = trim_tts_output(raw_audio, sample_rate=sr, max_internal_silence_ms=None) + + # Default behaviour cuts at the internal gap and drops the second utterance. + assert len(default_trim) / sr == pytest.approx(1.0, abs=0.25) + # Edge-only trim keeps both utterances and the gap between them. + assert len(edge_trim) / sr == pytest.approx(1.0 + 1.5 + 1.0 + 0.2, abs=0.05) def test_kokoro_trim_tts_output_removes_trailing_dead_space(): @@ -54,13 +72,15 @@ async def test_kokoro_backend_generate_applies_trimming(monkeypatch): class FakePipeline: def __call__(self, text, voice, speed=1.0): yield FakeResult(fake_audio) + yield FakeResult(fake_audio) monkeypatch.setattr(backend, "_get_pipeline", lambda lang: FakePipeline()) audio, sample_rate = await backend.generate("Read it back to me.", voice_prompt={}) assert sample_rate == sr - # Original fake audio was 2.0s (1s speech + 1s silence). - # Trimmed cuts trailing silence down to 1.0s speech boundary. - assert len(audio) / sr == pytest.approx(1.0, abs=0.05) - assert len(audio) < len(fake_audio) + # Two segments of (1s speech + 1s silence) = 4.0s raw. Only the trailing + # pad is trimmed (down to a 0.2s tail); the 1s gap between the segments + # must survive, otherwise the second segment would be dropped. + assert len(audio) / sr == pytest.approx(1.0 + 1.0 + 1.0 + 0.2, abs=0.05) + assert len(audio) < 2 * len(fake_audio) diff --git a/backend/utils/audio.py b/backend/utils/audio.py index bccab2c8..90ec315b 100644 --- a/backend/utils/audio.py +++ b/backend/utils/audio.py @@ -153,7 +153,7 @@ def trim_tts_output( frame_ms: int = 20, silence_threshold_db: float = -40.0, min_silence_ms: int = 200, - max_internal_silence_ms: int = 1000, + max_internal_silence_ms: int | None = 1000, fade_ms: int = 30, ) -> np.ndarray: """ @@ -170,7 +170,8 @@ def trim_tts_output( frame_ms: Frame size for RMS energy calculation silence_threshold_db: dB threshold below which a frame is silence min_silence_ms: Minimum trailing silence to keep - max_internal_silence_ms: Cut after any silence gap longer than this + max_internal_silence_ms: Cut after any silence gap longer than this. + ``None`` disables the internal cut and only trims the edges. fade_ms: Cosine fade-out duration in ms Returns: @@ -200,18 +201,18 @@ def trim_tts_output( break # Walk forward from first speech; cut at long internal silence gaps - max_silence_frames = int(max_internal_silence_ms / frame_ms) - consecutive_silence = 0 cut_frame = n_frames - - for i in range(first_speech, n_frames): - if is_speech[i]: - consecutive_silence = 0 - else: - consecutive_silence += 1 - if consecutive_silence >= max_silence_frames: - cut_frame = i - consecutive_silence + 1 - break + if max_internal_silence_ms is not None: + max_silence_frames = int(max_internal_silence_ms / frame_ms) + consecutive_silence = 0 + for i in range(first_speech, n_frames): + if is_speech[i]: + consecutive_silence = 0 + else: + consecutive_silence += 1 + if consecutive_silence >= max_silence_frames: + cut_frame = i - consecutive_silence + 1 + break # Trim trailing silence from the cut point min_silence_frames = int(min_silence_ms / frame_ms)