fix(backend): drop the prompt cache before the allocator-empty; note clear_cache is thread-agnostic

Review follow-ups: clear_voice_prompt_memory_cache() now runs before
backend.unload_model() at all three call sites so the device-resident
prompt tensors are already released when empty_device_cache() /
empty_mlx_cache() runs, instead of going back into the caching allocator
afterwards. empty_mlx_cache's docstring states why it may run off the
MLX worker thread.
This commit is contained in:
jamiepine
2026-10-04 00:25:53 +00:00
committed by capy-ai-staging[bot]
parent 1a803aa05f
commit b1323ed0a8
3 changed files with 9 additions and 3 deletions
+2 -2
View File
@@ -595,16 +595,16 @@ def unload_model_by_config(config: ModelConfig) -> bool:
backend = get_tts_backend_for_engine(config.engine)
loaded_size = getattr(backend, "_current_model_size", None) or getattr(backend, "model_size", None)
if backend.is_loaded() and loaded_size == config.model_size:
backend.unload_model()
clear_voice_prompt_memory_cache()
backend.unload_model()
return True
return False
# All other TTS engines
backend = get_tts_backend_for_engine(config.engine)
if backend.is_loaded():
backend.unload_model()
clear_voice_prompt_memory_cache()
backend.unload_model()
return True
return False
+6
View File
@@ -232,6 +232,12 @@ def empty_mlx_cache() -> None:
returning them to the OS. Backends must call this after unloading an
MLX model, or the process's memory footprint never shrinks even though
the model object itself was dropped.
Safe from any thread: ``mx.clear_cache`` only drains the global
allocator pool and never touches the per-thread stream registry, so
unlike load/generate it does not have to run on the MLX worker thread
(verified from the FastAPI event loop with a generation in flight on
the worker).
"""
import mlx.core as mx