fix(mlx): serialize accelerator lifecycle and inference

Route MLX load, inference, unload, reset, cache cleanup, and shutdown through a single worker. Add affinity and concurrent-unload regression coverage.\n\nVerified: 17 related backend tests; frontend CI; cargo check.
This commit is contained in:
Jamie Pine
2026-07-19 17:45:52 -07:00
parent f2cf2a729d
commit 0070c04bcf
10 changed files with 286 additions and 66 deletions
+7 -8
View File
@@ -2,28 +2,27 @@
TTS inference module - delegates to backend abstraction layer.
"""
from typing import Optional
import numpy as np
import io
import numpy as np
import soundfile as sf
from ..backends import get_tts_backend, TTSBackend
from ..backends import TTSBackend, get_tts_backend, unload_backend
def get_tts_model() -> TTSBackend:
"""
Get TTS backend instance (MLX or PyTorch based on platform).
Returns:
TTS backend instance
"""
return get_tts_backend()
def unload_tts_model():
"""Unload TTS model to free memory."""
backend = get_tts_backend()
backend.unload_model()
async def unload_tts_model():
"""Unload TTS model to free memory, serialized onto the MLX worker."""
await unload_backend(get_tts_backend())
def audio_to_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes: