fix(backend): drop the prompt cache before the allocator-empty; note clear_cache is thread-agnostic

Review follow-ups: clear_voice_prompt_memory_cache() now runs before
backend.unload_model() at all three call sites so the device-resident
prompt tensors are already released when empty_device_cache() /
empty_mlx_cache() runs, instead of going back into the caching allocator
afterwards. empty_mlx_cache's docstring states why it may run off the
MLX worker thread.
This commit is contained in:
jamiepine
2026-10-04 00:25:53 +00:00
committed by capy-ai-staging[bot]
parent 1a803aa05f
commit b1323ed0a8
3 changed files with 9 additions and 3 deletions
+2 -2
View File
@@ -595,16 +595,16 @@ def unload_model_by_config(config: ModelConfig) -> bool:
backend = get_tts_backend_for_engine(config.engine)
loaded_size = getattr(backend, "_current_model_size", None) or getattr(backend, "model_size", None)
if backend.is_loaded() and loaded_size == config.model_size:
backend.unload_model()
clear_voice_prompt_memory_cache()
backend.unload_model()
return True
return False
# All other TTS engines
backend = get_tts_backend_for_engine(config.engine)
if backend.is_loaded():
backend.unload_model()
clear_voice_prompt_memory_cache()
backend.unload_model()
return True
return False