From 502123a9437e9dfa20c1587975fbc9f01fdf09e6 Mon Sep 17 00:00:00 2001 From: Charles Hasse Date: Tue, 11 Aug 2026 20:08:19 -0300 Subject: [PATCH] fix(backend): retry runaway generations on the Chatterbox MLX backend The qwen configs already set retries_runaway on MLX, with the reason in a comment: mlx-audio can continue past an EOS miss and emit silence followed by codec noise. The Chatterbox MLX backend added in this PR hits the same failure and did not have the guard wired. Observed on an M4 Max with a cloned pt-BR profile: a 145 character sentence took 18.0s and returned 5.3s of audio for text worth about 8s, and a short reply came back as an endless hiss. With retries_runaway enabled the same sentence takes 3.4s and returns the full 7.4s of speech. --- backend/backends/__init__.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/backend/backends/__init__.py b/backend/backends/__init__.py index 9c539da0..fd8759cf 100644 --- a/backend/backends/__init__.py +++ b/backend/backends/__init__.py @@ -294,7 +294,8 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]: # Chatterbox multilingual follows the same backend-aware split as Qwen: the MLX # backend loads pre-converted weights, so the download must match the backend that # will consume it. - if get_backend_type() == "mlx": + on_mlx = get_backend_type() == "mlx" + if on_mlx: chatterbox_repo = "mlx-community/chatterbox-multilingual-v3" chatterbox_size_mb = 2600 else: @@ -317,6 +318,11 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]: hf_repo_id=chatterbox_repo, size_mb=chatterbox_size_mb, needs_trim=True, + # Same EOS miss the qwen configs guard against: on mlx-audio the decoder can run past + # the end of the sentence and emit silence followed by codec noise, which reaches the + # listener as an endless hiss. Retrying the affected text as smaller chunks is the + # existing remedy; it just was not wired for this engine. + retries_runaway=on_mlx, languages=[ "zh", "en",