feat(backend): run Chatterbox multilingual on MLX for Apple Silicon

The Chatterbox backend is pinned to the CPU on macOS, so voice cloning runs at
roughly 4x realtime there. This adds an MLX/Metal backend for the same engine and
selects it on Apple Silicon, mirroring the split the qwen engine already makes
between mlx_backend and pytorch_backend.

Measured on an M4 Max (36 GB) with a cloned pt-BR profile, same API, same profile,
model already loaded:

  short sentence (1.8s of audio):  7.4-9.2s  ->  1.2s
  longer sentence (5.0s of audio): 23.7s     ->  3.1s

The model config for chatterbox-tts is now backend aware, same as the qwen configs,
so the download matches the backend that will consume it.

Nothing changes off Apple Silicon: the PyTorch backend is still selected there, and
the CPU pinning it relies on is untouched.
This commit is contained in:
Charles Hasse
2026-10-04 00:25:58 +00:00
committed by capy-ai-staging[bot]
parent 63899fd865
commit ad5d64c0a3
2 changed files with 200 additions and 7 deletions
+21 -7
View File
@@ -290,10 +290,17 @@ def _get_qwen_custom_voice_configs() -> list[ModelConfig]:
def _get_non_qwen_tts_configs() -> list[ModelConfig]:
"""Return model configs for non-Qwen TTS engines.
"""Return model configs for non-Qwen TTS engines."""
# Chatterbox multilingual follows the same backend-aware split as Qwen: the MLX
# backend loads pre-converted weights, so the download must match the backend that
# will consume it.
if get_backend_type() == "mlx":
chatterbox_repo = "mlx-community/chatterbox-multilingual-v3"
chatterbox_size_mb = 2600
else:
chatterbox_repo = "ResembleAI/chatterbox"
chatterbox_size_mb = 3200
These are static — no backend-type branching needed.
"""
return [
ModelConfig(
model_name="luxtts",
@@ -307,8 +314,8 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]:
model_name="chatterbox-tts",
display_name="Chatterbox TTS (Multilingual)",
engine="chatterbox",
hf_repo_id="ResembleAI/chatterbox",
size_mb=3200,
hf_repo_id=chatterbox_repo,
size_mb=chatterbox_size_mb,
needs_trim=True,
languages=[
"zh",
@@ -707,9 +714,16 @@ def get_tts_backend_for_engine(engine: str) -> TTSBackend:
backend = LuxTTSBackend()
elif engine == "chatterbox":
from .chatterbox_backend import ChatterboxTTSBackend
# Same split the qwen engine already makes: on Apple Silicon the MLX/Metal
# port renders 7-9x faster than the CPU-pinned PyTorch path.
if get_backend_type() == "mlx":
from .chatterbox_mlx_backend import ChatterboxMLXTTSBackend
backend = ChatterboxTTSBackend()
backend = ChatterboxMLXTTSBackend()
else:
from .chatterbox_backend import ChatterboxTTSBackend
backend = ChatterboxTTSBackend()
elif engine == "chatterbox_turbo":
from .chatterbox_turbo_backend import ChatterboxTurboTTSBackend