mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-04 09:35:16 -07:00
feat(backend): run Chatterbox multilingual on MLX for Apple Silicon
The Chatterbox backend is pinned to the CPU on macOS, so voice cloning runs at roughly 4x realtime there. This adds an MLX/Metal backend for the same engine and selects it on Apple Silicon, mirroring the split the qwen engine already makes between mlx_backend and pytorch_backend. Measured on an M4 Max (36 GB) with a cloned pt-BR profile, same API, same profile, model already loaded: short sentence (1.8s of audio): 7.4-9.2s -> 1.2s longer sentence (5.0s of audio): 23.7s -> 3.1s The model config for chatterbox-tts is now backend aware, same as the qwen configs, so the download matches the backend that will consume it. Nothing changes off Apple Silicon: the PyTorch backend is still selected there, and the CPU pinning it relies on is untouched.
This commit is contained in:
committed by
capy-ai-staging[bot]
parent
63899fd865
commit
ad5d64c0a3
@@ -290,10 +290,17 @@ def _get_qwen_custom_voice_configs() -> list[ModelConfig]:
|
||||
|
||||
|
||||
def _get_non_qwen_tts_configs() -> list[ModelConfig]:
|
||||
"""Return model configs for non-Qwen TTS engines.
|
||||
"""Return model configs for non-Qwen TTS engines."""
|
||||
# Chatterbox multilingual follows the same backend-aware split as Qwen: the MLX
|
||||
# backend loads pre-converted weights, so the download must match the backend that
|
||||
# will consume it.
|
||||
if get_backend_type() == "mlx":
|
||||
chatterbox_repo = "mlx-community/chatterbox-multilingual-v3"
|
||||
chatterbox_size_mb = 2600
|
||||
else:
|
||||
chatterbox_repo = "ResembleAI/chatterbox"
|
||||
chatterbox_size_mb = 3200
|
||||
|
||||
These are static — no backend-type branching needed.
|
||||
"""
|
||||
return [
|
||||
ModelConfig(
|
||||
model_name="luxtts",
|
||||
@@ -307,8 +314,8 @@ def _get_non_qwen_tts_configs() -> list[ModelConfig]:
|
||||
model_name="chatterbox-tts",
|
||||
display_name="Chatterbox TTS (Multilingual)",
|
||||
engine="chatterbox",
|
||||
hf_repo_id="ResembleAI/chatterbox",
|
||||
size_mb=3200,
|
||||
hf_repo_id=chatterbox_repo,
|
||||
size_mb=chatterbox_size_mb,
|
||||
needs_trim=True,
|
||||
languages=[
|
||||
"zh",
|
||||
@@ -707,9 +714,16 @@ def get_tts_backend_for_engine(engine: str) -> TTSBackend:
|
||||
|
||||
backend = LuxTTSBackend()
|
||||
elif engine == "chatterbox":
|
||||
from .chatterbox_backend import ChatterboxTTSBackend
|
||||
# Same split the qwen engine already makes: on Apple Silicon the MLX/Metal
|
||||
# port renders 7-9x faster than the CPU-pinned PyTorch path.
|
||||
if get_backend_type() == "mlx":
|
||||
from .chatterbox_mlx_backend import ChatterboxMLXTTSBackend
|
||||
|
||||
backend = ChatterboxTTSBackend()
|
||||
backend = ChatterboxMLXTTSBackend()
|
||||
else:
|
||||
from .chatterbox_backend import ChatterboxTTSBackend
|
||||
|
||||
backend = ChatterboxTTSBackend()
|
||||
elif engine == "chatterbox_turbo":
|
||||
from .chatterbox_turbo_backend import ChatterboxTurboTTSBackend
|
||||
|
||||
|
||||
Reference in New Issue
Block a user