From 2693c983f2c04bc05e4e2ade4fea2669deba1426 Mon Sep 17 00:00:00 2001 From: Seunghun Lee Date: Sat, 18 Jul 2026 21:53:40 +0700 Subject: [PATCH] fix: extend MLX thread-affinity fix to the Qwen3 LLM backend MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dedicated-MLX-thread fix (#886) covers the TTS and STT backends, but MLXQwenLLMBackend still dispatches load/generate through asyncio.to_thread(), so /llm/generate (personality compose/rewrite, dictation refinement) crashes with the same 'There is no Stream(gpu, 0) in current thread.' — raised from mlx_lm.generate's wired_limit on exit — whenever load and generate land on different pool threads. Route the MLX LLM backend's unload/load/generate through the same _run_on_mlx_thread helper so every MLX call in the process shares one worker thread. Verified on Apple M3 Pro (macOS 26.5, mlx 0.32.0, mlx-lm 0.31.1, mlx-audio 0.4.1): /llm/generate failed 100% before, succeeds after; TTS + STT unaffected. Co-Authored-By: Claude Fable 5 --- backend/backends/qwen_llm_backend.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/backend/backends/qwen_llm_backend.py b/backend/backends/qwen_llm_backend.py index 5d29d892..c2ec70c3 100644 --- a/backend/backends/qwen_llm_backend.py +++ b/backend/backends/qwen_llm_backend.py @@ -212,10 +212,15 @@ class MLXQwenLLMBackend: if self.model is not None and self._current_model_size == model_size: return - if self.model is not None and self._current_model_size != model_size: - self.unload_model() + # Routed through the same dedicated MLX thread as TTS/STT — MLX's + # Metal stream is thread-local, so load and generate must run on + # one thread (see mlx_backend._run_on_mlx_thread / issue #699). + from .mlx_backend import _run_on_mlx_thread - await asyncio.to_thread(self._load_model_sync, model_size) + if self.model is not None and self._current_model_size != model_size: + await _run_on_mlx_thread(self.unload_model) + + await _run_on_mlx_thread(self._load_model_sync, model_size) def _load_model_sync(self, model_size: str) -> None: from mlx_lm import load as mlx_load @@ -258,7 +263,9 @@ class MLXQwenLLMBackend: examples: Optional[list[tuple[str, str]]] = None, ) -> str: await self.load_model(model_size) - return await asyncio.to_thread( + from .mlx_backend import _run_on_mlx_thread + + return await _run_on_mlx_thread( self._generate_sync, prompt, system, max_tokens, temperature, examples )