mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-03 17:15:19 -07:00
fix: extend MLX thread-affinity fix to the Qwen3 LLM backend
The dedicated-MLX-thread fix (#886) covers the TTS and STT backends, but MLXQwenLLMBackend still dispatches load/generate through asyncio.to_thread(), so /llm/generate (personality compose/rewrite, dictation refinement) crashes with the same 'There is no Stream(gpu, 0) in current thread.' — raised from mlx_lm.generate's wired_limit on exit — whenever load and generate land on different pool threads. Route the MLX LLM backend's unload/load/generate through the same _run_on_mlx_thread helper so every MLX call in the process shares one worker thread. Verified on Apple M3 Pro (macOS 26.5, mlx 0.32.0, mlx-lm 0.31.1, mlx-audio 0.4.1): /llm/generate failed 100% before, succeeds after; TTS + STT unaffected. Co-Authored-By: Claude Fable 5 <[email protected]>
This commit is contained in:
committed by
capy-ai-staging[bot]
co-authored by
Claude Fable 5
parent
57ef6636bc
commit
2693c983f2
@@ -212,10 +212,15 @@ class MLXQwenLLMBackend:
|
||||
if self.model is not None and self._current_model_size == model_size:
|
||||
return
|
||||
|
||||
if self.model is not None and self._current_model_size != model_size:
|
||||
self.unload_model()
|
||||
# Routed through the same dedicated MLX thread as TTS/STT — MLX's
|
||||
# Metal stream is thread-local, so load and generate must run on
|
||||
# one thread (see mlx_backend._run_on_mlx_thread / issue #699).
|
||||
from .mlx_backend import _run_on_mlx_thread
|
||||
|
||||
await asyncio.to_thread(self._load_model_sync, model_size)
|
||||
if self.model is not None and self._current_model_size != model_size:
|
||||
await _run_on_mlx_thread(self.unload_model)
|
||||
|
||||
await _run_on_mlx_thread(self._load_model_sync, model_size)
|
||||
|
||||
def _load_model_sync(self, model_size: str) -> None:
|
||||
from mlx_lm import load as mlx_load
|
||||
@@ -258,7 +263,9 @@ class MLXQwenLLMBackend:
|
||||
examples: Optional[list[tuple[str, str]]] = None,
|
||||
) -> str:
|
||||
await self.load_model(model_size)
|
||||
return await asyncio.to_thread(
|
||||
from .mlx_backend import _run_on_mlx_thread
|
||||
|
||||
return await _run_on_mlx_thread(
|
||||
self._generate_sync, prompt, system, max_tokens, temperature, examples
|
||||
)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user