mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-03 17:15:19 -07:00
fix(offline-guard): make every cache check cover the files the forced-offline load reads
The load body now runs with HF_HUB_OFFLINE forced when _is_model_cached() reports True, so a snapshot that holds the weights but not the small files the loader also opens would fail hard instead of fetching them. Verified against chatterbox-tts 0.1.7 (mtl_tts.py allow_patterns, tts_turbo.py from_local) and the TADA loader's unsloth/Llama-3.2-1B tokenizer download; list those files in the required_files checks. Also note in the changelog why the 0.4.5 removal of this guard no longer applies.
This commit is contained in:
@@ -35,6 +35,9 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
# HuggingFace repos
|
||||
TADA_CODEC_REPO = "HumeAI/tada-codec"
|
||||
# TADA hardcodes the gated meta-llama/Llama-3.2-1B tokenizer; we load it from
|
||||
# this ungated mirror instead (see load_model).
|
||||
TADA_TOKENIZER_REPO = "unsloth/Llama-3.2-1B"
|
||||
TADA_1B_REPO = "HumeAI/tada-1b"
|
||||
TADA_3B_ML_REPO = "HumeAI/tada-3b-ml"
|
||||
|
||||
@@ -52,6 +55,12 @@ _TADA_CODEC_WEIGHT_FILES = [
|
||||
"encoder/model.safetensors",
|
||||
]
|
||||
|
||||
_TADA_TOKENIZER_FILES = [
|
||||
"tokenizer.json",
|
||||
"tokenizer_config.json",
|
||||
"special_tokens_map.json",
|
||||
]
|
||||
|
||||
|
||||
class HumeTadaBackend:
|
||||
"""HumeAI TADA TTS backend for high-quality voice cloning."""
|
||||
@@ -80,7 +89,8 @@ class HumeTadaBackend:
|
||||
repo = TADA_MODEL_REPOS.get(model_size, TADA_1B_REPO)
|
||||
model_cached = is_model_cached(repo, required_files=_TADA_MODEL_WEIGHT_FILES)
|
||||
codec_cached = is_model_cached(TADA_CODEC_REPO, required_files=_TADA_CODEC_WEIGHT_FILES)
|
||||
return model_cached and codec_cached
|
||||
tokenizer_cached = is_model_cached(TADA_TOKENIZER_REPO, required_files=_TADA_TOKENIZER_FILES)
|
||||
return model_cached and codec_cached and tokenizer_cached
|
||||
|
||||
async def load_model(self, model_size: str = "1B") -> None:
|
||||
"""Load the TADA model and encoder."""
|
||||
@@ -140,7 +150,7 @@ class HumeTadaBackend:
|
||||
# local cache path so we can point TADA at it directly.
|
||||
logger.info("Downloading Llama tokenizer (ungated mirror)...")
|
||||
tokenizer_path = snapshot_download(
|
||||
repo_id="unsloth/Llama-3.2-1B",
|
||||
repo_id=TADA_TOKENIZER_REPO,
|
||||
token=None,
|
||||
allow_patterns=["tokenizer*", "special_tokens*"],
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user