mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 15:15:27 -07:00
Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
10247851d4 | ||
|
|
79b70b8970 | ||
|
|
de15d8fdc6 | ||
|
|
f3ed312cf2 |
@@ -26,45 +26,12 @@ jobs:
|
|||||||
args: ""
|
args: ""
|
||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
backend: "pytorch"
|
backend: "pytorch"
|
||||||
- platform: "ubuntu-22.04"
|
|
||||||
# --config override disables updater-artifact generation on Linux.
|
|
||||||
# tauri.conf.json has createUpdaterArtifacts: "v1Compatible" which
|
|
||||||
# on Linux wants to synthesize a .AppImage.tar.gz by downloading
|
|
||||||
# linuxdeploy at build time — this is what silently hangs CI
|
|
||||||
# (see v0.4.2 round 2, 25 min of no output after rpm bundling).
|
|
||||||
# We ship deb+rpm only; Linux users update via apt/dnf, not the
|
|
||||||
# Tauri in-app updater.
|
|
||||||
args: '--target x86_64-unknown-linux-gnu --bundles deb,rpm --verbose --config {"bundle":{"createUpdaterArtifacts":false}}'
|
|
||||||
python-version: "3.12"
|
|
||||||
backend: "pytorch"
|
|
||||||
|
|
||||||
runs-on: ${{ matrix.platform }}
|
runs-on: ${{ matrix.platform }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
# Ubuntu runners ship with ~14 GB free; pip + PyInstaller + torch can
|
|
||||||
# peak well above that during the build. Reclaim ~25 GB by pruning
|
|
||||||
# preinstalled toolchains we don't use. This is what likely tripped
|
|
||||||
# the March 2026 Linux release attempts (see commit 103e98b
|
|
||||||
# "github runners suck") — not a code issue, a disk-pressure one.
|
|
||||||
- name: Free up disk space (ubuntu)
|
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
|
||||||
# Pinned to v1.3.1 (SHA) — this job runs with contents: write and
|
|
||||||
# handles signing secrets later, so we don't want a floating ref.
|
|
||||||
uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be
|
|
||||||
with:
|
|
||||||
tool-cache: false
|
|
||||||
android: true
|
|
||||||
dotnet: true
|
|
||||||
haskell: true
|
|
||||||
# large-packages: true would `apt-get remove '^llvm-.*'`, which
|
|
||||||
# cascade-removes reverse deps that won't be pulled back in by the
|
|
||||||
# `llvm-dev` install below. The other flags already free ~20 GB,
|
|
||||||
# enough for the Python + torch + PyInstaller build.
|
|
||||||
large-packages: false
|
|
||||||
swap-storage: true
|
|
||||||
|
|
||||||
- name: Install dependencies (ubuntu only)
|
- name: Install dependencies (ubuntu only)
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
||||||
run: |
|
run: |
|
||||||
@@ -106,10 +73,8 @@ jobs:
|
|||||||
# fine on transformers 4.57.x in practice (verified in dev), so install
|
# fine on transformers 4.57.x in practice (verified in dev), so install
|
||||||
# them --no-deps. mlx-audio's other runtime deps (huggingface_hub,
|
# them --no-deps. mlx-audio's other runtime deps (huggingface_hub,
|
||||||
# librosa, numpy, numba, pyloudnorm) are already in requirements.txt;
|
# librosa, numpy, numba, pyloudnorm) are already in requirements.txt;
|
||||||
# miniaudio is in requirements-mlx.txt (needed by mlx_audio.stt,
|
# the rest (sounddevice, miniaudio, protobuf, sentencepiece, pyyaml,
|
||||||
# not transitively pulled by anything else — see issue #505); the
|
# jinja2) are pulled in by other engines.
|
||||||
# rest (sounddevice, protobuf, sentencepiece, pyyaml, jinja2) are
|
|
||||||
# pulled in by other engines.
|
|
||||||
pip install --no-deps mlx-lm==0.31.1
|
pip install --no-deps mlx-lm==0.31.1
|
||||||
pip install --no-deps mlx-audio==0.4.1
|
pip install --no-deps mlx-audio==0.4.1
|
||||||
|
|
||||||
@@ -168,21 +133,6 @@ jobs:
|
|||||||
p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }}
|
p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }}
|
||||||
p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
|
p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
|
||||||
|
|
||||||
- name: Disk / environment snapshot (pre-bundle debug)
|
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
|
||||||
run: |
|
|
||||||
echo "=== df -h ==="
|
|
||||||
df -h
|
|
||||||
echo "=== free -h ==="
|
|
||||||
free -h
|
|
||||||
echo "=== Rust / Cargo ==="
|
|
||||||
rustc --version
|
|
||||||
cargo --version
|
|
||||||
echo "=== Bun ==="
|
|
||||||
bun --version
|
|
||||||
echo "=== Tauri CLI ==="
|
|
||||||
cd tauri && bun run tauri --version
|
|
||||||
|
|
||||||
- name: Extract release notes from CHANGELOG.md
|
- name: Extract release notes from CHANGELOG.md
|
||||||
id: changelog
|
id: changelog
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -206,13 +156,7 @@ jobs:
|
|||||||
echo "CHANGELOG_EOF"
|
echo "CHANGELOG_EOF"
|
||||||
} >> "$GITHUB_OUTPUT"
|
} >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
# Linux hang watchdog: previous releases silently wedged inside tauri
|
|
||||||
# bundling (possibly linuxdeploy/AppImage download, possibly cargo link).
|
|
||||||
# Cap the step at 30 min so we get logs instead of waiting out the 6hr
|
|
||||||
# job timeout. Other platforms historically complete in ~25 min, so 45
|
|
||||||
# is comfortable.
|
|
||||||
- uses: tauri-apps/[email protected]
|
- uses: tauri-apps/[email protected]
|
||||||
timeout-minutes: ${{ (contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')) && 30 || 45 }}
|
|
||||||
env:
|
env:
|
||||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
|
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
|
||||||
@@ -224,9 +168,6 @@ jobs:
|
|||||||
APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }}
|
APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }}
|
||||||
APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }}
|
APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }}
|
||||||
APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }}
|
APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }}
|
||||||
# Stream subprocess stdout/stderr so the hang is visible in logs.
|
|
||||||
CARGO_TERM_VERBOSE: "true"
|
|
||||||
RUST_BACKTRACE: "1"
|
|
||||||
with:
|
with:
|
||||||
projectPath: tauri
|
projectPath: tauri
|
||||||
tagName: v__VERSION__
|
tagName: v__VERSION__
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/app",
|
"name": "@voicebox/app",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -177,7 +177,7 @@ def _get_qwen_model_configs() -> list[ModelConfig]:
|
|||||||
backend_type = get_backend_type()
|
backend_type = get_backend_type()
|
||||||
if backend_type == "mlx":
|
if backend_type == "mlx":
|
||||||
repo_1_7b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
|
repo_1_7b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
|
||||||
repo_0_6b = "mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16"
|
repo_0_6b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16" # 0.6B not available in MLX, falls back
|
||||||
else:
|
else:
|
||||||
repo_1_7b = "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
|
repo_1_7b = "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
|
||||||
repo_0_6b = "Qwen/Qwen3-TTS-12Hz-0.6B-Base"
|
repo_0_6b = "Qwen/Qwen3-TTS-12Hz-0.6B-Base"
|
||||||
|
|||||||
@@ -45,9 +45,11 @@ class MLXTTSBackend:
|
|||||||
Returns:
|
Returns:
|
||||||
HuggingFace Hub model ID for MLX
|
HuggingFace Hub model ID for MLX
|
||||||
"""
|
"""
|
||||||
|
# MLX model mapping
|
||||||
mlx_model_map = {
|
mlx_model_map = {
|
||||||
"1.7B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16",
|
"1.7B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16",
|
||||||
"0.6B": "mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16",
|
# 0.6B not yet converted to MLX format
|
||||||
|
"0.6B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16", # Fallback to 1.7B
|
||||||
}
|
}
|
||||||
|
|
||||||
if model_size not in mlx_model_map:
|
if model_size not in mlx_model_map:
|
||||||
|
|||||||
@@ -3,12 +3,6 @@
|
|||||||
|
|
||||||
mlx>=0.30.0
|
mlx>=0.30.0
|
||||||
|
|
||||||
# miniaudio is a runtime dep of mlx-audio's STT path (mlx_audio.stt).
|
|
||||||
# mlx-audio itself is installed --no-deps (see comment below), so we
|
|
||||||
# must list miniaudio explicitly here or transcription fails on fresh
|
|
||||||
# M1 installs with `ModuleNotFoundError: miniaudio` (issue #505).
|
|
||||||
miniaudio>=1.59
|
|
||||||
|
|
||||||
# NOTE: mlx-audio is intentionally not listed here. From 0.3.1 onward it
|
# NOTE: mlx-audio is intentionally not listed here. From 0.3.1 onward it
|
||||||
# declares `transformers==5.0.0rc3` / `>=5.0.0`, which conflicts with the
|
# declares `transformers==5.0.0rc3` / `>=5.0.0`, which conflicts with the
|
||||||
# `transformers<=4.57.6` cap in requirements.txt and breaks CI's clean
|
# `transformers<=4.57.6` cap in requirements.txt and breaks CI's clean
|
||||||
@@ -16,7 +10,6 @@ miniaudio>=1.59
|
|||||||
# mlx_audio.stt.load) works fine on transformers 4.57.x in practice.
|
# mlx_audio.stt.load) works fine on transformers 4.57.x in practice.
|
||||||
#
|
#
|
||||||
# Install it via `pip install --no-deps mlx-audio==0.4.1` after this file
|
# Install it via `pip install --no-deps mlx-audio==0.4.1` after this file
|
||||||
# (see .github/workflows/release.yml). Most other mlx-audio runtime deps
|
# (see .github/workflows/release.yml). All other mlx-audio runtime deps
|
||||||
# (huggingface_hub, librosa, mlx-lm, numba, numpy, protobuf, pyloudnorm,
|
# (huggingface_hub, librosa, miniaudio, mlx-lm, numba, numpy, protobuf,
|
||||||
# sounddevice, tqdm) are already in requirements.txt or pulled in by
|
# pyloudnorm, sounddevice, tqdm) are already in requirements.txt.
|
||||||
# other engines.
|
|
||||||
|
|||||||
@@ -1,112 +0,0 @@
|
|||||||
"""
|
|
||||||
Unit tests for reference-audio preprocessing.
|
|
||||||
|
|
||||||
Covers :func:`backend.utils.audio.preprocess_reference_audio` and
|
|
||||||
:func:`backend.utils.audio.validate_and_load_reference_audio`.
|
|
||||||
"""
|
|
||||||
|
|
||||||
import sys
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pytest
|
|
||||||
import soundfile as sf
|
|
||||||
|
|
||||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
|
||||||
|
|
||||||
from utils.audio import ( # noqa: E402
|
|
||||||
preprocess_reference_audio,
|
|
||||||
validate_and_load_reference_audio,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
SR = 24000
|
|
||||||
|
|
||||||
|
|
||||||
def _tone(duration_s: float, amp: float = 0.3, freq: float = 220.0) -> np.ndarray:
|
|
||||||
n = int(duration_s * SR)
|
|
||||||
t = np.arange(n, dtype=np.float32) / SR
|
|
||||||
return (amp * np.sin(2 * np.pi * freq * t)).astype(np.float32)
|
|
||||||
|
|
||||||
|
|
||||||
def test_peak_cap_scales_hot_input():
|
|
||||||
audio = _tone(3.0, amp=0.99)
|
|
||||||
out = preprocess_reference_audio(audio, SR)
|
|
||||||
assert np.abs(out).max() <= 0.951
|
|
||||||
|
|
||||||
|
|
||||||
def test_peak_cap_leaves_moderate_input_untouched():
|
|
||||||
audio = _tone(3.0, amp=0.5)
|
|
||||||
out = preprocess_reference_audio(audio, SR)
|
|
||||||
assert np.isclose(np.abs(out).max(), 0.5, atol=1e-3)
|
|
||||||
|
|
||||||
|
|
||||||
def test_dc_offset_removed():
|
|
||||||
audio = _tone(3.0, amp=0.3) + 0.1
|
|
||||||
out = preprocess_reference_audio(audio, SR)
|
|
||||||
assert abs(float(np.mean(out))) < 1e-3
|
|
||||||
|
|
||||||
|
|
||||||
def test_silence_is_trimmed_with_padding_kept():
|
|
||||||
silence = np.zeros(int(SR * 1.0), dtype=np.float32)
|
|
||||||
speech = _tone(3.0, amp=0.3)
|
|
||||||
audio = np.concatenate([silence, speech, silence])
|
|
||||||
out = preprocess_reference_audio(audio, SR)
|
|
||||||
# Most of the 2s of leading/trailing silence should be gone, but the
|
|
||||||
# 3s of speech plus ~200ms of padding should remain.
|
|
||||||
assert len(audio) - len(out) >= SR, "expected >=1s of silence trimmed"
|
|
||||||
assert len(out) >= int(3.0 * SR), "speech body should be preserved"
|
|
||||||
|
|
||||||
|
|
||||||
def test_clean_audio_is_not_padded_past_original_length():
|
|
||||||
# Well-recorded audio with no edge silence shouldn't get longer after
|
|
||||||
# preprocessing — otherwise a 29.9 s upload could be pushed past the
|
|
||||||
# 30 s max_duration ceiling downstream.
|
|
||||||
audio = _tone(3.0, amp=0.3)
|
|
||||||
out = preprocess_reference_audio(audio, SR)
|
|
||||||
assert len(out) <= len(audio)
|
|
||||||
|
|
||||||
|
|
||||||
def test_empty_input_returns_empty():
|
|
||||||
out = preprocess_reference_audio(np.zeros(0, dtype=np.float32), SR)
|
|
||||||
assert out.size == 0
|
|
||||||
|
|
||||||
|
|
||||||
def test_validate_accepts_previously_rejected_hot_file(tmp_path):
|
|
||||||
audio = _tone(3.0, amp=0.995)
|
|
||||||
path = tmp_path / "hot.wav"
|
|
||||||
sf.write(str(path), audio, SR)
|
|
||||||
|
|
||||||
ok, err, out_audio, out_sr = validate_and_load_reference_audio(str(path))
|
|
||||||
|
|
||||||
assert ok, f"expected pass, got error: {err}"
|
|
||||||
assert out_audio is not None
|
|
||||||
assert out_sr == SR
|
|
||||||
assert np.abs(out_audio).max() <= 0.951
|
|
||||||
|
|
||||||
|
|
||||||
def test_validate_still_rejects_silent_input(tmp_path):
|
|
||||||
audio = np.zeros(int(SR * 3.0), dtype=np.float32)
|
|
||||||
path = tmp_path / "silent.wav"
|
|
||||||
sf.write(str(path), audio, SR)
|
|
||||||
|
|
||||||
ok, err, _, _ = validate_and_load_reference_audio(str(path))
|
|
||||||
|
|
||||||
assert not ok
|
|
||||||
assert err is not None
|
|
||||||
assert "too short" in err.lower() or "quiet" in err.lower()
|
|
||||||
|
|
||||||
|
|
||||||
def test_validate_rejects_too_short(tmp_path):
|
|
||||||
audio = _tone(0.5, amp=0.3)
|
|
||||||
path = tmp_path / "short.wav"
|
|
||||||
sf.write(str(path), audio, SR)
|
|
||||||
|
|
||||||
ok, err, _, _ = validate_and_load_reference_audio(str(path))
|
|
||||||
|
|
||||||
assert not ok
|
|
||||||
assert "too short" in (err or "").lower()
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
pytest.main([__file__, "-v"])
|
|
||||||
@@ -175,7 +175,7 @@ async def main():
|
|||||||
print(" ✅ Server is running")
|
print(" ✅ Server is running")
|
||||||
|
|
||||||
# Test model
|
# Test model
|
||||||
model_name = "qwen-tts-0.6B"
|
model_name = "qwen-tts-0.6B" # Note: 0.6B currently maps to 1.7B on MLX
|
||||||
|
|
||||||
# Check current status
|
# Check current status
|
||||||
print(f"\n📊 Checking status of {model_name}...")
|
print(f"\n📊 Checking status of {model_name}...")
|
||||||
|
|||||||
+3
-65
@@ -199,66 +199,6 @@ def trim_tts_output(
|
|||||||
return trimmed
|
return trimmed
|
||||||
|
|
||||||
|
|
||||||
def preprocess_reference_audio(
|
|
||||||
audio: np.ndarray,
|
|
||||||
sample_rate: int,
|
|
||||||
peak_target: float = 0.95,
|
|
||||||
trim_top_db: float = 40.0,
|
|
||||||
edge_padding_ms: int = 100,
|
|
||||||
) -> np.ndarray:
|
|
||||||
"""
|
|
||||||
Clean up a reference-audio sample before validation/storage.
|
|
||||||
|
|
||||||
Removes DC offset, trims leading/trailing silence, and caps the peak so a
|
|
||||||
slightly-hot recording doesn't get rejected downstream as "clipping". The
|
|
||||||
goal is to accept reasonable real-world recordings — not to repair badly
|
|
||||||
distorted ones. True clipping artifacts inside the waveform can't be
|
|
||||||
recovered by peak scaling and will still sound bad.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
audio: Mono audio array.
|
|
||||||
sample_rate: Sample rate of ``audio`` in Hz.
|
|
||||||
peak_target: Peak amplitude cap in [0, 1]. Applied only if the input
|
|
||||||
peak exceeds this value.
|
|
||||||
trim_top_db: Silence threshold for edge trimming, in dB below peak.
|
|
||||||
40 dB sits below normal speech dynamic range (≈30 dB) so soft
|
|
||||||
trailing syllables are preserved, while still catching obvious
|
|
||||||
leading/trailing silence. Lower values are more aggressive;
|
|
||||||
librosa's own default is 60.
|
|
||||||
edge_padding_ms: Milliseconds of padding to add back at each edge
|
|
||||||
*only if* trimming shortened the waveform, so TTS engines have a
|
|
||||||
brief silence to anchor on without ever making the output longer
|
|
||||||
than the input.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Preprocessed audio array (float32).
|
|
||||||
"""
|
|
||||||
audio = audio.astype(np.float32, copy=False)
|
|
||||||
|
|
||||||
if audio.size == 0:
|
|
||||||
return audio
|
|
||||||
|
|
||||||
audio = audio - float(np.mean(audio))
|
|
||||||
|
|
||||||
trimmed, _ = librosa.effects.trim(audio, top_db=trim_top_db)
|
|
||||||
if 0 < trimmed.size < audio.size:
|
|
||||||
pad_each = int(sample_rate * edge_padding_ms / 1000)
|
|
||||||
# Never pad past the original length — for near-max-duration uploads
|
|
||||||
# an unconditional pad would push them over the 30 s ceiling and
|
|
||||||
# trigger a spurious "too long" rejection.
|
|
||||||
headroom = (audio.size - trimmed.size) // 2
|
|
||||||
pad = min(pad_each, max(headroom, 0))
|
|
||||||
if pad > 0:
|
|
||||||
trimmed = np.pad(trimmed, (pad, pad), mode="constant")
|
|
||||||
audio = trimmed
|
|
||||||
|
|
||||||
peak = float(np.abs(audio).max())
|
|
||||||
if peak > peak_target and peak > 0:
|
|
||||||
audio = audio * (peak_target / peak)
|
|
||||||
|
|
||||||
return audio
|
|
||||||
|
|
||||||
|
|
||||||
def validate_reference_audio(
|
def validate_reference_audio(
|
||||||
audio_path: str,
|
audio_path: str,
|
||||||
min_duration: float = 2.0,
|
min_duration: float = 2.0,
|
||||||
@@ -292,16 +232,11 @@ def validate_and_load_reference_audio(
|
|||||||
"""
|
"""
|
||||||
Validate and load reference audio in a single pass.
|
Validate and load reference audio in a single pass.
|
||||||
|
|
||||||
Applies :func:`preprocess_reference_audio` before checks so that
|
|
||||||
slightly-hot recordings aren't rejected as clipping. Duration and RMS
|
|
||||||
checks run on the preprocessed waveform.
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Tuple of (is_valid, error_message, audio_array, sample_rate)
|
Tuple of (is_valid, error_message, audio_array, sample_rate)
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
audio, sr = load_audio(audio_path)
|
audio, sr = load_audio(audio_path)
|
||||||
audio = preprocess_reference_audio(audio, sr)
|
|
||||||
duration = len(audio) / sr
|
duration = len(audio) / sr
|
||||||
|
|
||||||
if duration < min_duration:
|
if duration < min_duration:
|
||||||
@@ -313,6 +248,9 @@ def validate_and_load_reference_audio(
|
|||||||
if rms < min_rms:
|
if rms < min_rms:
|
||||||
return False, "Audio is too quiet or silent", None, None
|
return False, "Audio is too quiet or silent", None, None
|
||||||
|
|
||||||
|
if np.abs(audio).max() > 0.99:
|
||||||
|
return False, "Audio is clipping (reduce input gain)", None, None
|
||||||
|
|
||||||
return True, None, audio, sr
|
return True, None, audio, sr
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
return False, f"Error validating audio: {str(e)}", None, None
|
return False, f"Error validating audio: {str(e)}", None, None
|
||||||
|
|||||||
@@ -309,7 +309,11 @@ Still reported. Users get stuck downloads, can't resume, offline mode edge cases
|
|||||||
|
|
||||||
**Fix path:** PR #443 addresses infinite offline retry. CustomVoice-specific download failures (#475, #445) need triage — likely related to frozen-binary import fixes in PR #438. TADA cluster (#336, #348) and macOS ARM import regressions (#287, #275, #304) need a dedicated triage pass.
|
**Fix path:** PR #443 addresses infinite offline retry. CustomVoice-specific download failures (#475, #445) need triage — likely related to frozen-binary import fixes in PR #438. TADA cluster (#336, #348) and macOS ARM import regressions (#287, #275, #304) need a dedicated triage pass.
|
||||||
|
|
||||||
**Qwen 0.6B-downloads-1.7B reports:** **#485** (2026-04-19), **#423** (macOS M1), **#329**. Originally a stale-fallback bug: `mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16` wasn't published when MLX support shipped, so the 0.6B slot was aliased to the 1.7B repo. The 0.6B bf16 conversion is live now and both `backend/backends/mlx_backend.py` and `backend/backends/__init__.py` point at their correct repos. Qwen CustomVoice is unaffected — it runs via PyTorch on all platforms, both sizes always have dedicated repos.
|
**Qwen 0.6B-downloads-1.7B reports:** **#485** (2026-04-19), **#423** (macOS M1), **#329**. Platform-dependent:
|
||||||
|
|
||||||
|
- **On MLX (Apple Silicon) — not a bug.** `mlx-community` only publishes 1.7B-Base-bf16 weights, so the 0.6B Base option intentionally resolves to the same repo (`backend/backends/__init__.py:180` — `# 0.6B not available in MLX, falls back`). UX gap: the selector offers a size that doesn't exist on the active backend. Fix: (a) hide the 0.6B option on MLX, or (b) label it "0.6B (uses 1.7B on Apple Silicon)".
|
||||||
|
- **On PyTorch (Windows/Linux/CUDA/ROCm/XPU/CPU) — real bug if reported.** Both 0.6B and 1.7B have distinct repos (`Qwen/Qwen3-TTS-12Hz-0.6B-Base` vs `-1.7B-Base`). Triage each report by platform before merging into the MLX cluster.
|
||||||
|
- **Qwen CustomVoice (either platform)** — no fallback, both sizes always have dedicated repos.
|
||||||
|
|
||||||
### Language Requests (ongoing)
|
### Language Requests (ongoing)
|
||||||
|
|
||||||
@@ -389,7 +393,7 @@ Notable:
|
|||||||
| **#306** ("voice model"), **#389** ("New model"), **#473** ("New functionality") | Title-only issues, no content. Request details or close. |
|
| **#306** ("voice model"), **#389** ("New model"), **#473** ("New functionality") | Title-only issues, no content. Request details or close. |
|
||||||
| **#309** | Uninstall/cleanup question. Answer and close. |
|
| **#309** | Uninstall/cleanup question. Answer and close. |
|
||||||
| **#241** | "How to use in Colab" — support question, not a bug. |
|
| **#241** | "How to use in Colab" — support question, not a bug. |
|
||||||
| **#423** / **#485** / **#329** | Stale MLX fallback to 1.7B repo — fixed; 0.6B bf16 conversion now live on `mlx-community`, registry points at correct repo on both backends. |
|
| **#423** / **#485** / **#329** | Platform-dependent. On MLX: not a bug (0.6B weights don't exist upstream, fallback is intentional — fix UX). On PyTorch: real bug if reproducible. Classify each by reporter's platform before deduping. |
|
||||||
| **#336** / **#348** | TADA download/registration cluster — triage together. |
|
| **#336** / **#348** | TADA download/registration cluster — triage together. |
|
||||||
| **#287** / **#275** / **#304** | macOS ARM import regressions on new version — likely one root cause. |
|
| **#287** / **#275** / **#304** | macOS ARM import regressions on new version — likely one root cause. |
|
||||||
| **#292**, **#349** | Possibly already fixed by merged PRs (#321/#412 and #345). Verify + close. |
|
| **#292**, **#349** | Possibly already fixed by merged PRs (#321/#412 and #345). Verify + close. |
|
||||||
|
|||||||
@@ -446,6 +446,20 @@ Restart the app to create a fresh database.
|
|||||||
huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base
|
huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Qwen 0.6B Downloads the Same Files as 1.7B on Apple Silicon
|
||||||
|
|
||||||
|
**Symptoms:**
|
||||||
|
- You select Qwen 0.6B on an Apple Silicon Mac and the download is the same size as 1.7B
|
||||||
|
- Generation speed and VRAM usage match 1.7B, not the expected smaller model
|
||||||
|
|
||||||
|
**Explanation:**
|
||||||
|
This is intentional, not a bug. The MLX community only publishes `mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16` — there is no 0.6B MLX build. Voicebox's model registry falls back to the 1.7B weights when 0.6B is selected on MLX (see `backend/backends/__init__.py`).
|
||||||
|
|
||||||
|
**Solution:**
|
||||||
|
- On Apple Silicon, both size options use the 1.7B model — pick either.
|
||||||
|
- If you specifically need a smaller model, switch to **Kokoro 82M** (~350 MB) or **LuxTTS** (~300 MB) — both CPU-realtime.
|
||||||
|
- On Windows/Linux with PyTorch, 0.6B and 1.7B are distinct repos and behave differently.
|
||||||
|
|
||||||
### Wrong Model Version
|
### Wrong Model Version
|
||||||
|
|
||||||
**Symptoms:**
|
**Symptoms:**
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/landing",
|
"name": "@voicebox/landing",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"description": "Landing page for voicebox.sh",
|
"description": "Landing page for voicebox.sh",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "next dev --turbo",
|
"dev": "next dev --turbo",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "voicebox",
|
"name": "voicebox",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"private": true,
|
"private": true,
|
||||||
"workspaces": [
|
"workspaces": [
|
||||||
"app",
|
"app",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/tauri",
|
"name": "@voicebox/tauri",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Generated
+1
-1
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.4.2"
|
version = "0.4.1"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"core-foundation-sys",
|
"core-foundation-sys",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.4.2"
|
version = "0.4.1"
|
||||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||||
authors = ["you"]
|
authors = ["you"]
|
||||||
license = ""
|
license = ""
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://schema.tauri.app/config/2",
|
"$schema": "https://schema.tauri.app/config/2",
|
||||||
"productName": "Voicebox",
|
"productName": "Voicebox",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"identifier": "sh.voicebox.app",
|
"identifier": "sh.voicebox.app",
|
||||||
"build": {
|
"build": {
|
||||||
"beforeDevCommand": "bun run dev",
|
"beforeDevCommand": "bun run dev",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/web",
|
"name": "@voicebox/web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Reference in New Issue
Block a user