mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-03 09:05:17 -07:00
fix torchcodec error by using soundfile instead of torchaudio.load
torchaudio 2.10+ switched its default audio loading backend to torchcodec, which isn't installed. Replace torchaudio.load() with soundfile.read() in create_voice_prompt(). TADA's internal use of torchaudio.functional.resample() is unaffected (pure PyTorch math, no torchcodec dependency).
This commit is contained in:
@@ -231,12 +231,17 @@ class HumeTadaBackend:
|
|||||||
|
|
||||||
def _encode_sync():
|
def _encode_sync():
|
||||||
import torch
|
import torch
|
||||||
import torchaudio
|
import soundfile as sf
|
||||||
|
|
||||||
device = self._device
|
device = self._device
|
||||||
|
|
||||||
# Load and prepare audio
|
# Load audio with soundfile (torchaudio 2.10+ requires torchcodec)
|
||||||
audio, sr = torchaudio.load(str(audio_path))
|
audio_np, sr = sf.read(str(audio_path), dtype="float32")
|
||||||
|
audio = torch.from_numpy(audio_np).float()
|
||||||
|
if audio.ndim == 1:
|
||||||
|
audio = audio.unsqueeze(0) # (samples,) -> (1, samples)
|
||||||
|
else:
|
||||||
|
audio = audio.T # (samples, channels) -> (channels, samples)
|
||||||
audio = audio.to(device)
|
audio = audio.to(device)
|
||||||
|
|
||||||
# Encode with forced alignment
|
# Encode with forced alignment
|
||||||
|
|||||||
Reference in New Issue
Block a user