mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-04 01:25:18 -07:00
fix(mlx): drain the pool in a finally so a failing in-flight op releases it too; fix the mid-generation test's call count
Review follow-ups: the post-op drain now runs in a finally block for the TTS, STT and LLM folded callables, so an exception escaping the generation (e.g. the voice-clone fallback failing) still releases the buffers the worker held. The mid-generation test asserted one drain call but the path legitimately produces two (unload_model's own and the post-op one); a second test covers the error path.
This commit is contained in:
committed by
capy-ai-staging[bot]
parent
cffe24ffd2
commit
fb67ad436c
@@ -282,12 +282,13 @@ class MLXQwenLLMBackend:
|
||||
# unload can swap or null out self.model / self.tokenizer.
|
||||
if self.model is None or self._current_model_size != resolved_size:
|
||||
self._reload_sync(resolved_size)
|
||||
result = self._generate_sync(prompt, system, max_tokens, temperature, examples)
|
||||
# Drain the MLX pool if an unload landed mid-generation (see
|
||||
# MLXTTSBackend._reload_and_generate_sync).
|
||||
if self.model is None:
|
||||
empty_mlx_cache()
|
||||
return result
|
||||
try:
|
||||
return self._generate_sync(prompt, system, max_tokens, temperature, examples)
|
||||
finally:
|
||||
# Drain the MLX pool if an unload landed mid-generation (see
|
||||
# MLXTTSBackend._reload_and_generate_sync).
|
||||
if self.model is None:
|
||||
empty_mlx_cache()
|
||||
|
||||
async with self._op_lock:
|
||||
return await _run_on_mlx_thread(_reload_and_generate_sync)
|
||||
|
||||
Reference in New Issue
Block a user