fix(kokoro): trim edges only, keep inter-segment gaps

Registering Kokoro with needs_trim=True routed its output through the
generic trim_tts_output, whose 1s internal-silence cut was tuned for
Chatterbox hallucinations. KPipeline synthesizes newline- and
token-limit-separated segments independently, and the ~0.3s lead plus
~0.7s tail pads at each boundary add up to 1.2s of silence, so with
af_sarah a four-paragraph script came back as its first paragraph only
(10.9s -> 1.8s) and a 51s text lost half its segments.

Drop the needs_trim flag, keep the in-backend trim, and give
trim_tts_output a max_internal_silence_ms=None mode that only trims the
leading and trailing pads. Tests updated for the new behaviour, with a
two-segment fake pipeline and an explicit internal-gap case.
This commit is contained in:
jamiepine
2026-10-04 00:01:24 +00:00
committed by capy-ai-staging[bot]
parent 615aeaeb35
commit ae300c5316
4 changed files with 48 additions and 23 deletions
-1
View File
@@ -369,7 +369,6 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]:
engine="kokoro",
hf_repo_id="hexgrad/Kokoro-82M",
size_mb=350,
needs_trim=True,
languages=["en", "es", "fr", "hi", "it", "pt", "ja", "zh"],
),
]
+6 -1
View File
@@ -291,7 +291,12 @@ class KokoroTTSBackend:
audio = np.concatenate(audio_chunks).astype(np.float32)
from ..utils.audio import trim_tts_output
audio = trim_tts_output(audio, sample_rate=KOKORO_SAMPLE_RATE)
# Edge-only trim: Kokoro pads ~0.3s before and ~0.7s after speech.
# The internal-gap cut is disabled because KPipeline synthesizes
# newline/token-limit segments independently and the pads at each
# segment boundary add up to >1s of silence; with the default cut
# everything after the first segment would be dropped.
audio = trim_tts_output(audio, sample_rate=KOKORO_SAMPLE_RATE, max_internal_silence_ms=None)
return audio, KOKORO_SAMPLE_RATE
return await asyncio.to_thread(_generate_sync)
+28 -8
View File
@@ -7,12 +7,30 @@ from backend.backends import engine_needs_trim, get_model_config
from backend.utils.audio import trim_tts_output
def test_kokoro_engine_needs_trim_enabled():
"""Verify Kokoro engine is registered with needs_trim=True in model config."""
assert engine_needs_trim("kokoro") is True
def test_kokoro_engine_does_not_use_generic_trim():
"""Kokoro trims inside the backend; the generic needs_trim path would cut
multi-segment output at the first >1s inter-segment gap."""
assert engine_needs_trim("kokoro") is False
config = get_model_config("kokoro")
assert config is not None
assert config.needs_trim is True
assert config.needs_trim is False
def test_trim_tts_output_edge_only_keeps_internal_gaps():
"""With max_internal_silence_ms=None only the edges are trimmed."""
sr = 24000
speech = np.full(int(sr * 1.0), 0.2, dtype=np.float32)
gap = np.zeros(int(sr * 1.5), dtype=np.float32) # longer than the 1s default cut
pad = np.zeros(int(sr * 0.5), dtype=np.float32)
raw_audio = np.concatenate([pad, speech, gap, speech, pad])
default_trim = trim_tts_output(raw_audio, sample_rate=sr)
edge_trim = trim_tts_output(raw_audio, sample_rate=sr, max_internal_silence_ms=None)
# Default behaviour cuts at the internal gap and drops the second utterance.
assert len(default_trim) / sr == pytest.approx(1.0, abs=0.25)
# Edge-only trim keeps both utterances and the gap between them.
assert len(edge_trim) / sr == pytest.approx(1.0 + 1.5 + 1.0 + 0.2, abs=0.05)
def test_kokoro_trim_tts_output_removes_trailing_dead_space():
@@ -54,13 +72,15 @@ async def test_kokoro_backend_generate_applies_trimming(monkeypatch):
class FakePipeline:
def __call__(self, text, voice, speed=1.0):
yield FakeResult(fake_audio)
yield FakeResult(fake_audio)
monkeypatch.setattr(backend, "_get_pipeline", lambda lang: FakePipeline())
audio, sample_rate = await backend.generate("Read it back to me.", voice_prompt={})
assert sample_rate == sr
# Original fake audio was 2.0s (1s speech + 1s silence).
# Trimmed cuts trailing silence down to 1.0s speech boundary.
assert len(audio) / sr == pytest.approx(1.0, abs=0.05)
assert len(audio) < len(fake_audio)
# Two segments of (1s speech + 1s silence) = 4.0s raw. Only the trailing
# pad is trimmed (down to a 0.2s tail); the 1s gap between the segments
# must survive, otherwise the second segment would be dropped.
assert len(audio) / sr == pytest.approx(1.0 + 1.0 + 1.0 + 0.2, abs=0.05)
assert len(audio) < 2 * len(fake_audio)
+5 -4
View File
@@ -153,7 +153,7 @@ def trim_tts_output(
frame_ms: int = 20,
silence_threshold_db: float = -40.0,
min_silence_ms: int = 200,
max_internal_silence_ms: int = 1000,
max_internal_silence_ms: int | None = 1000,
fade_ms: int = 30,
) -> np.ndarray:
"""
@@ -170,7 +170,8 @@ def trim_tts_output(
frame_ms: Frame size for RMS energy calculation
silence_threshold_db: dB threshold below which a frame is silence
min_silence_ms: Minimum trailing silence to keep
max_internal_silence_ms: Cut after any silence gap longer than this
max_internal_silence_ms: Cut after any silence gap longer than this.
``None`` disables the internal cut and only trims the edges.
fade_ms: Cosine fade-out duration in ms
Returns:
@@ -200,10 +201,10 @@ def trim_tts_output(
break
# Walk forward from first speech; cut at long internal silence gaps
cut_frame = n_frames
if max_internal_silence_ms is not None:
max_silence_frames = int(max_internal_silence_ms / frame_ms)
consecutive_silence = 0
cut_frame = n_frames
for i in range(first_speech, n_frames):
if is_speech[i]:
consecutive_silence = 0