fix(mlx): release the model binding inside the sync bodies so a failing op's traceback cannot pin it

Review follow-up: the post-op drain ran in a finally, but on the error
path the exception's traceback kept the _generate_sync/_transcribe_sync
frame (and its model local) alive, so mx.clear_cache() had nothing to
return. Each sync body now drops its model (and tokenizer) binding in its
own finally before the exception propagates.
This commit is contained in:
jamiepine
2026-10-04 00:25:53 +00:00
committed by capy-ai-staging[bot]
parent fb67ad436c
commit 63899fd865
2 changed files with 108 additions and 89 deletions
+23 -17
View File
@@ -305,21 +305,27 @@ class MLXQwenLLMBackend:
from mlx_lm.sample_utils import make_sampler
model, tokenizer = self.model, self.tokenizer
messages = _build_messages(prompt, system, examples)
chat_prompt = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=False,
)
try:
messages = _build_messages(prompt, system, examples)
chat_prompt = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=False,
)
sampler = make_sampler(temp=temperature, top_p=0.9) if temperature > 0 else None
text = mlx_generate(
model,
tokenizer,
prompt=chat_prompt,
max_tokens=max_tokens,
sampler=sampler,
verbose=False,
)
return text.strip()
sampler = make_sampler(temp=temperature, top_p=0.9) if temperature > 0 else None
text = mlx_generate(
model,
tokenizer,
prompt=chat_prompt,
max_tokens=max_tokens,
sampler=sampler,
verbose=False,
)
return text.strip()
finally:
# Drop the local binding here, inside the frame a propagating
# traceback would keep alive, so the caller's drain can
# actually return the model's buffers to MLX.
del model, tokenizer