Compare commits

..
Author SHA1 Message Date
James PineandClaude Opus 4.7 010c3d5219 fix(audio): prevent WKWebView audio session teardown after backgrounding (#41)
Keep a silent looping <audio> element mounted at the app root so macOS
never tears down the CoreAudio session. Without this, backgrounding the
app long enough leaves WaveSurfer's AudioContext in a state where play()
resolves and timeupdate fires, but no audio reaches the output — and not
even cmd+R (full JS reload) restores it, only a full app relaunch.

Uses a zero-PCM WAV blob at full volume rather than a muted element,
since WebKit can optimize muted media away and defeat the purpose.

Co-Authored-By: Claude Opus 4.7 (1M context) <[email protected]>
2026-04-18 22:02:45 -07:00
32 changed files with 519 additions and 1146 deletions
+2 -61
View File
@@ -26,45 +26,12 @@ jobs:
args: "" args: ""
python-version: "3.12" python-version: "3.12"
backend: "pytorch" backend: "pytorch"
- platform: "ubuntu-22.04"
# --config override disables updater-artifact generation on Linux.
# tauri.conf.json has createUpdaterArtifacts: "v1Compatible" which
# on Linux wants to synthesize a .AppImage.tar.gz by downloading
# linuxdeploy at build time — this is what silently hangs CI
# (see v0.4.2 round 2, 25 min of no output after rpm bundling).
# We ship deb+rpm only; Linux users update via apt/dnf, not the
# Tauri in-app updater.
args: '--target x86_64-unknown-linux-gnu --bundles deb,rpm --verbose --config {"bundle":{"createUpdaterArtifacts":false}}'
python-version: "3.12"
backend: "pytorch"
runs-on: ${{ matrix.platform }} runs-on: ${{ matrix.platform }}
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
# Ubuntu runners ship with ~14 GB free; pip + PyInstaller + torch can
# peak well above that during the build. Reclaim ~25 GB by pruning
# preinstalled toolchains we don't use. This is what likely tripped
# the March 2026 Linux release attempts (see commit 103e98b
# "github runners suck") — not a code issue, a disk-pressure one.
- name: Free up disk space (ubuntu)
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
# Pinned to v1.3.1 (SHA) — this job runs with contents: write and
# handles signing secrets later, so we don't want a floating ref.
uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be
with:
tool-cache: false
android: true
dotnet: true
haskell: true
# large-packages: true would `apt-get remove '^llvm-.*'`, which
# cascade-removes reverse deps that won't be pulled back in by the
# `llvm-dev` install below. The other flags already free ~20 GB,
# enough for the Python + torch + PyInstaller build.
large-packages: false
swap-storage: true
- name: Install dependencies (ubuntu only) - name: Install dependencies (ubuntu only)
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace') if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
run: | run: |
@@ -106,10 +73,8 @@ jobs:
# fine on transformers 4.57.x in practice (verified in dev), so install # fine on transformers 4.57.x in practice (verified in dev), so install
# them --no-deps. mlx-audio's other runtime deps (huggingface_hub, # them --no-deps. mlx-audio's other runtime deps (huggingface_hub,
# librosa, numpy, numba, pyloudnorm) are already in requirements.txt; # librosa, numpy, numba, pyloudnorm) are already in requirements.txt;
# miniaudio is in requirements-mlx.txt (needed by mlx_audio.stt, # the rest (sounddevice, miniaudio, protobuf, sentencepiece, pyyaml,
# not transitively pulled by anything else — see issue #505); the # jinja2) are pulled in by other engines.
# rest (sounddevice, protobuf, sentencepiece, pyyaml, jinja2) are
# pulled in by other engines.
pip install --no-deps mlx-lm==0.31.1 pip install --no-deps mlx-lm==0.31.1
pip install --no-deps mlx-audio==0.4.1 pip install --no-deps mlx-audio==0.4.1
@@ -168,21 +133,6 @@ jobs:
p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }} p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }}
p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }} p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
- name: Disk / environment snapshot (pre-bundle debug)
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
run: |
echo "=== df -h ==="
df -h
echo "=== free -h ==="
free -h
echo "=== Rust / Cargo ==="
rustc --version
cargo --version
echo "=== Bun ==="
bun --version
echo "=== Tauri CLI ==="
cd tauri && bun run tauri --version
- name: Extract release notes from CHANGELOG.md - name: Extract release notes from CHANGELOG.md
id: changelog id: changelog
shell: bash shell: bash
@@ -206,13 +156,7 @@ jobs:
echo "CHANGELOG_EOF" echo "CHANGELOG_EOF"
} >> "$GITHUB_OUTPUT" } >> "$GITHUB_OUTPUT"
# Linux hang watchdog: previous releases silently wedged inside tauri
# bundling (possibly linuxdeploy/AppImage download, possibly cargo link).
# Cap the step at 30 min so we get logs instead of waiting out the 6hr
# job timeout. Other platforms historically complete in ~25 min, so 45
# is comfortable.
- uses: tauri-apps/[email protected] - uses: tauri-apps/[email protected]
timeout-minutes: ${{ (contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')) && 30 || 45 }}
env: env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }} TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
@@ -224,9 +168,6 @@ jobs:
APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }} APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }}
APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }} APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }}
APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }} APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }}
# Stream subprocess stdout/stderr so the hang is visible in logs.
CARGO_TERM_VERBOSE: "true"
RUST_BACKTRACE: "1"
with: with:
projectPath: tauri projectPath: tauri
tagName: v__VERSION__ tagName: v__VERSION__
+2 -2
View File
@@ -359,7 +359,7 @@ Releases are managed by maintainers:
## Troubleshooting ## Troubleshooting
See [docs/content/docs/overview/troubleshooting.mdx](docs/content/docs/overview/troubleshooting.mdx) for common issues and solutions. See [docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md) for common issues and solutions.
**Quick fixes:** **Quick fixes:**
@@ -372,7 +372,7 @@ See [docs/content/docs/overview/troubleshooting.mdx](docs/content/docs/overview/
- Open an issue for bugs or feature requests - Open an issue for bugs or feature requests
- Check existing issues and discussions - Check existing issues and discussions
- Review the codebase to understand patterns - Review the codebase to understand patterns
- See [docs/content/docs/overview/troubleshooting.mdx](docs/content/docs/overview/troubleshooting.mdx) for common issues - See [docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md) for common issues
## Additional Resources ## Additional Resources
+1 -4
View File
@@ -33,8 +33,7 @@
<a href="https://docs.voicebox.sh">Docs</a> • <a href="https://docs.voicebox.sh">Docs</a> •
<a href="#download">Download</a> • <a href="#download">Download</a> •
<a href="#features">Features</a> • <a href="#features">Features</a> •
<a href="#api">API</a> • <a href="#api">API</a>
<a href="docs/content/docs/overview/troubleshooting.mdx">Troubleshooting</a>
</p> </p>
<br/> <br/>
@@ -92,8 +91,6 @@ Voicebox is a **local-first voice cloning studio** — a free and open-source al
> **Linux** — Pre-built binaries are not yet available. See [voicebox.sh/linux-install](https://voicebox.sh/linux-install) for build-from-source instructions. > **Linux** — Pre-built binaries are not yet available. See [voicebox.sh/linux-install](https://voicebox.sh/linux-install) for build-from-source instructions.
> **Having trouble?** See the [Troubleshooting Guide](docs/content/docs/overview/troubleshooting.mdx) for common install, generation, model-download, and GPU issues.
--- ---
## Features ## Features
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@voicebox/app", "name": "@voicebox/app",
"version": "0.4.2", "version": "0.4.1",
"private": true, "private": true,
"type": "module", "type": "module",
"scripts": { "scripts": {
+1 -1
View File
@@ -177,7 +177,7 @@ def _get_qwen_model_configs() -> list[ModelConfig]:
backend_type = get_backend_type() backend_type = get_backend_type()
if backend_type == "mlx": if backend_type == "mlx":
repo_1_7b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16" repo_1_7b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
repo_0_6b = "mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16" repo_0_6b = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16" # 0.6B not available in MLX, falls back
else: else:
repo_1_7b = "Qwen/Qwen3-TTS-12Hz-1.7B-Base" repo_1_7b = "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
repo_0_6b = "Qwen/Qwen3-TTS-12Hz-0.6B-Base" repo_0_6b = "Qwen/Qwen3-TTS-12Hz-0.6B-Base"
+26 -36
View File
@@ -45,9 +45,11 @@ class MLXTTSBackend:
Returns: Returns:
HuggingFace Hub model ID for MLX HuggingFace Hub model ID for MLX
""" """
# MLX model mapping
mlx_model_map = { mlx_model_map = {
"1.7B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16", "1.7B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16",
"0.6B": "mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16", # 0.6B not yet converted to MLX format
"0.6B": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16", # Fallback to 1.7B
} }
if model_size not in mlx_model_map: if model_size not in mlx_model_map:
@@ -193,8 +195,6 @@ class MLXTTSBackend:
logger.info("Generating audio for text: %s", text) logger.info("Generating audio for text: %s", text)
model_name = f"qwen-tts-{self._current_model_size}"
def _generate_sync(): def _generate_sync():
"""Run synchronous generation in thread pool.""" """Run synchronous generation in thread pool."""
# MLX generate() returns a generator yielding GenerationResult objects # MLX generate() returns a generator yielding GenerationResult objects
@@ -220,40 +220,36 @@ class MLXTTSBackend:
logger.warning("Regenerating without voice prompt.") logger.warning("Regenerating without voice prompt.")
ref_audio = None ref_audio = None
# Model is loaded → weights are on disk. Force offline so # Check if model supports voice cloning via generate method
# lazy tokenizer/config lookups inside mlx_audio don't hang # MLX API may support ref_audio parameter directly
# when the user is disconnected (issue #462). try:
with force_offline_if_cached(True, model_name): # Try with voice cloning parameters if supported
# Check if model supports voice cloning via generate method if ref_audio:
# MLX API may support ref_audio parameter directly # Check if generate accepts ref_audio parameter
try: import inspect
# Try with voice cloning parameters if supported
if ref_audio:
# Check if generate accepts ref_audio parameter
import inspect
sig = inspect.signature(self.model.generate) sig = inspect.signature(self.model.generate)
if "ref_audio" in sig.parameters: if "ref_audio" in sig.parameters:
# Generate with voice cloning # Generate with voice cloning
for result in self.model.generate(text, ref_audio=ref_audio, ref_text=ref_text, lang_code=lang): for result in self.model.generate(text, ref_audio=ref_audio, ref_text=ref_text, lang_code=lang):
audio_chunks.append(np.array(result.audio)) audio_chunks.append(np.array(result.audio))
sample_rate = result.sample_rate sample_rate = result.sample_rate
else:
# Fallback: generate without voice cloning
for result in self.model.generate(text, lang_code=lang):
audio_chunks.append(np.array(result.audio))
sample_rate = result.sample_rate
else: else:
# No voice prompt, generate normally # Fallback: generate without voice cloning
for result in self.model.generate(text, lang_code=lang): for result in self.model.generate(text, lang_code=lang):
audio_chunks.append(np.array(result.audio)) audio_chunks.append(np.array(result.audio))
sample_rate = result.sample_rate sample_rate = result.sample_rate
except Exception as e: else:
# If voice cloning fails, try without it # No voice prompt, generate normally
logger.warning("Voice cloning failed, generating without voice prompt: %s", e)
for result in self.model.generate(text, lang_code=lang): for result in self.model.generate(text, lang_code=lang):
audio_chunks.append(np.array(result.audio)) audio_chunks.append(np.array(result.audio))
sample_rate = result.sample_rate sample_rate = result.sample_rate
except Exception as e:
# If voice cloning fails, try without it
logger.warning("Voice cloning failed, generating without voice prompt: %s", e)
for result in self.model.generate(text, lang_code=lang):
audio_chunks.append(np.array(result.audio))
sample_rate = result.sample_rate
# Concatenate all chunks # Concatenate all chunks
if audio_chunks: if audio_chunks:
@@ -347,8 +343,6 @@ class MLXSTTBackend:
""" """
await self.load_model_async(model_size) await self.load_model_async(model_size)
progress_model_name = f"whisper-{self.model_size}"
def _transcribe_sync(): def _transcribe_sync():
"""Run synchronous transcription in thread pool.""" """Run synchronous transcription in thread pool."""
# MLX Whisper transcription using generate method # MLX Whisper transcription using generate method
@@ -357,11 +351,7 @@ class MLXSTTBackend:
if language: if language:
decode_options["language"] = language decode_options["language"] = language
# Model is loaded → weights are on disk. Force offline so result = self.model.generate(str(audio_path), **decode_options)
# lazy tokenizer/config lookups don't hang when the user is
# disconnected (issue #462).
with force_offline_if_cached(True, progress_model_name):
result = self.model.generate(str(audio_path), **decode_options)
# Extract text from result # Extract text from result
if isinstance(result, str): if isinstance(result, str):
+39 -56
View File
@@ -172,19 +172,13 @@ class PyTorchTTSBackend:
# This shouldn't happen in practice, but handle it # This shouldn't happen in practice, but handle it
return {"prompt": cached_prompt}, True return {"prompt": cached_prompt}, True
model_name = f"qwen-tts-{self._current_model_size}"
def _create_prompt_sync(): def _create_prompt_sync():
"""Run synchronous voice prompt creation in thread pool.""" """Run synchronous voice prompt creation in thread pool."""
# Model is loaded → weights are on disk. Force offline so return self.model.create_voice_clone_prompt(
# lazy tokenizer/config lookups inside qwen_tts don't hang ref_audio=str(audio_path),
# when the user is disconnected (issue #462). ref_text=reference_text,
with force_offline_if_cached(True, model_name): x_vector_only_mode=False,
return self.model.create_voice_clone_prompt( )
ref_audio=str(audio_path),
ref_text=reference_text,
x_vector_only_mode=False,
)
# Run blocking operation in thread pool # Run blocking operation in thread pool
voice_prompt_items = await asyncio.to_thread(_create_prompt_sync) voice_prompt_items = await asyncio.to_thread(_create_prompt_sync)
@@ -227,24 +221,19 @@ class PyTorchTTSBackend:
# Load model # Load model
await self.load_model_async(None) await self.load_model_async(None)
model_name = f"qwen-tts-{self._current_model_size}"
def _generate_sync(): def _generate_sync():
"""Run synchronous generation in thread pool.""" """Run synchronous generation in thread pool."""
# Set seed if provided # Set seed if provided
if seed is not None: if seed is not None:
manual_seed(seed, self.device) manual_seed(seed, self.device)
# Model is loaded → weights are on disk. Force offline so # Generate audio - this is the blocking operation
# lazy tokenizer/config lookups inside qwen_tts don't hang wavs, sample_rate = self.model.generate_voice_clone(
# when the user is disconnected (issue #462). text=text,
with force_offline_if_cached(True, model_name): voice_clone_prompt=voice_prompt,
wavs, sample_rate = self.model.generate_voice_clone( language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
text=text, instruct=instruct,
voice_clone_prompt=voice_prompt, )
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
instruct=instruct,
)
return wavs[0], sample_rate return wavs[0], sample_rate
# Run blocking inference in thread pool to avoid blocking event loop # Run blocking inference in thread pool to avoid blocking event loop
@@ -342,46 +331,40 @@ class PyTorchSTTBackend:
""" """
await self.load_model_async(model_size) await self.load_model_async(model_size)
progress_model_name = f"whisper-{self.model_size}"
def _transcribe_sync(): def _transcribe_sync():
"""Run synchronous transcription in thread pool.""" """Run synchronous transcription in thread pool."""
# Load audio # Load audio
audio, _sr = load_audio(audio_path, sample_rate=16000) audio, sr = load_audio(audio_path, sample_rate=16000)
# Model is loaded → weights are on disk. Force offline so # Process audio
# `get_decoder_prompt_ids` and any lazy tokenizer lookups inputs = self.processor(
# don't hang when the user is disconnected (issue #462). audio,
with force_offline_if_cached(True, progress_model_name): sampling_rate=16000,
# Process audio return_tensors="pt",
inputs = self.processor( )
audio, inputs = inputs.to(self.device)
sampling_rate=16000,
return_tensors="pt", # Generate transcription
# If language is provided, force it; otherwise let Whisper auto-detect
generate_kwargs = {}
if language:
forced_decoder_ids = self.processor.get_decoder_prompt_ids(
language=language,
task="transcribe",
) )
inputs = inputs.to(self.device) generate_kwargs["forced_decoder_ids"] = forced_decoder_ids
# Generate transcription with torch.no_grad():
# If language is provided, force it; otherwise let Whisper auto-detect predicted_ids = self.model.generate(
generate_kwargs = {} inputs["input_features"],
if language: **generate_kwargs,
forced_decoder_ids = self.processor.get_decoder_prompt_ids( )
language=language,
task="transcribe",
)
generate_kwargs["forced_decoder_ids"] = forced_decoder_ids
with torch.no_grad(): # Decode
predicted_ids = self.model.generate( transcription = self.processor.batch_decode(
inputs["input_features"], predicted_ids,
**generate_kwargs, skip_special_tokens=True,
) )[0]
# Decode
transcription = self.processor.batch_decode(
predicted_ids,
skip_special_tokens=True,
)[0]
return transcription.strip() return transcription.strip()
+13 -20
View File
@@ -28,7 +28,6 @@ from .base import (
combine_voice_prompts as _combine_voice_prompts, combine_voice_prompts as _combine_voice_prompts,
model_load_progress, model_load_progress,
) )
from ..utils.hf_offline_patch import force_offline_if_cached
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -105,19 +104,18 @@ class QwenCustomVoiceBackend:
model_path = self._get_model_path(model_size) model_path = self._get_model_path(model_size)
logger.info("Loading Qwen CustomVoice %s on %s...", model_size, self.device) logger.info("Loading Qwen CustomVoice %s on %s...", model_size, self.device)
with force_offline_if_cached(is_cached, model_name): if self.device == "cpu":
if self.device == "cpu": self.model = Qwen3TTSModel.from_pretrained(
self.model = Qwen3TTSModel.from_pretrained( model_path,
model_path, torch_dtype=torch.float32,
torch_dtype=torch.float32, low_cpu_mem_usage=False,
low_cpu_mem_usage=False, )
) else:
else: self.model = Qwen3TTSModel.from_pretrained(
self.model = Qwen3TTSModel.from_pretrained( model_path,
model_path, device_map=self.device,
device_map=self.device, torch_dtype=torch.bfloat16,
torch_dtype=torch.bfloat16, )
)
self._current_model_size = model_size self._current_model_size = model_size
self.model_size = model_size self.model_size = model_size
@@ -186,7 +184,6 @@ class QwenCustomVoiceBackend:
await self.load_model_async(None) await self.load_model_async(None)
speaker = voice_prompt.get("preset_voice_id") or QWEN_CV_DEFAULT_SPEAKER speaker = voice_prompt.get("preset_voice_id") or QWEN_CV_DEFAULT_SPEAKER
model_name = f"qwen-custom-voice-{self._current_model_size}"
def _generate_sync(): def _generate_sync():
if seed is not None: if seed is not None:
@@ -206,11 +203,7 @@ class QwenCustomVoiceBackend:
if instruct: if instruct:
kwargs["instruct"] = instruct kwargs["instruct"] = instruct
# Model is loaded → weights are on disk. Force offline so wavs, sample_rate = self.model.generate_custom_voice(**kwargs)
# lazy tokenizer/config lookups inside qwen_tts don't hang
# when the user is disconnected (issue #462).
with force_offline_if_cached(True, model_name):
wavs, sample_rate = self.model.generate_custom_voice(**kwargs)
return wavs[0], sample_rate return wavs[0], sample_rate
audio, sample_rate = await asyncio.to_thread(_generate_sync) audio, sample_rate = await asyncio.to_thread(_generate_sync)
+3 -10
View File
@@ -3,12 +3,6 @@
mlx>=0.30.0 mlx>=0.30.0
# miniaudio is a runtime dep of mlx-audio's STT path (mlx_audio.stt).
# mlx-audio itself is installed --no-deps (see comment below), so we
# must list miniaudio explicitly here or transcription fails on fresh
# M1 installs with `ModuleNotFoundError: miniaudio` (issue #505).
miniaudio>=1.59
# NOTE: mlx-audio is intentionally not listed here. From 0.3.1 onward it # NOTE: mlx-audio is intentionally not listed here. From 0.3.1 onward it
# declares `transformers==5.0.0rc3` / `>=5.0.0`, which conflicts with the # declares `transformers==5.0.0rc3` / `>=5.0.0`, which conflicts with the
# `transformers<=4.57.6` cap in requirements.txt and breaks CI's clean # `transformers<=4.57.6` cap in requirements.txt and breaks CI's clean
@@ -16,7 +10,6 @@ miniaudio>=1.59
# mlx_audio.stt.load) works fine on transformers 4.57.x in practice. # mlx_audio.stt.load) works fine on transformers 4.57.x in practice.
# #
# Install it via `pip install --no-deps mlx-audio==0.4.1` after this file # Install it via `pip install --no-deps mlx-audio==0.4.1` after this file
# (see .github/workflows/release.yml). Most other mlx-audio runtime deps # (see .github/workflows/release.yml). All other mlx-audio runtime deps
# (huggingface_hub, librosa, mlx-lm, numba, numpy, protobuf, pyloudnorm, # (huggingface_hub, librosa, miniaudio, mlx-lm, numba, numpy, protobuf,
# sounddevice, tqdm) are already in requirements.txt or pulled in by # pyloudnorm, sounddevice, tqdm) are already in requirements.txt.
# other engines.
-112
View File
@@ -1,112 +0,0 @@
"""
Unit tests for reference-audio preprocessing.
Covers :func:`backend.utils.audio.preprocess_reference_audio` and
:func:`backend.utils.audio.validate_and_load_reference_audio`.
"""
import sys
from pathlib import Path
import numpy as np
import pytest
import soundfile as sf
sys.path.insert(0, str(Path(__file__).parent.parent))
from utils.audio import ( # noqa: E402
preprocess_reference_audio,
validate_and_load_reference_audio,
)
SR = 24000
def _tone(duration_s: float, amp: float = 0.3, freq: float = 220.0) -> np.ndarray:
n = int(duration_s * SR)
t = np.arange(n, dtype=np.float32) / SR
return (amp * np.sin(2 * np.pi * freq * t)).astype(np.float32)
def test_peak_cap_scales_hot_input():
audio = _tone(3.0, amp=0.99)
out = preprocess_reference_audio(audio, SR)
assert np.abs(out).max() <= 0.951
def test_peak_cap_leaves_moderate_input_untouched():
audio = _tone(3.0, amp=0.5)
out = preprocess_reference_audio(audio, SR)
assert np.isclose(np.abs(out).max(), 0.5, atol=1e-3)
def test_dc_offset_removed():
audio = _tone(3.0, amp=0.3) + 0.1
out = preprocess_reference_audio(audio, SR)
assert abs(float(np.mean(out))) < 1e-3
def test_silence_is_trimmed_with_padding_kept():
silence = np.zeros(int(SR * 1.0), dtype=np.float32)
speech = _tone(3.0, amp=0.3)
audio = np.concatenate([silence, speech, silence])
out = preprocess_reference_audio(audio, SR)
# Most of the 2s of leading/trailing silence should be gone, but the
# 3s of speech plus ~200ms of padding should remain.
assert len(audio) - len(out) >= SR, "expected >=1s of silence trimmed"
assert len(out) >= int(3.0 * SR), "speech body should be preserved"
def test_clean_audio_is_not_padded_past_original_length():
# Well-recorded audio with no edge silence shouldn't get longer after
# preprocessing — otherwise a 29.9 s upload could be pushed past the
# 30 s max_duration ceiling downstream.
audio = _tone(3.0, amp=0.3)
out = preprocess_reference_audio(audio, SR)
assert len(out) <= len(audio)
def test_empty_input_returns_empty():
out = preprocess_reference_audio(np.zeros(0, dtype=np.float32), SR)
assert out.size == 0
def test_validate_accepts_previously_rejected_hot_file(tmp_path):
audio = _tone(3.0, amp=0.995)
path = tmp_path / "hot.wav"
sf.write(str(path), audio, SR)
ok, err, out_audio, out_sr = validate_and_load_reference_audio(str(path))
assert ok, f"expected pass, got error: {err}"
assert out_audio is not None
assert out_sr == SR
assert np.abs(out_audio).max() <= 0.951
def test_validate_still_rejects_silent_input(tmp_path):
audio = np.zeros(int(SR * 3.0), dtype=np.float32)
path = tmp_path / "silent.wav"
sf.write(str(path), audio, SR)
ok, err, _, _ = validate_and_load_reference_audio(str(path))
assert not ok
assert err is not None
assert "too short" in err.lower() or "quiet" in err.lower()
def test_validate_rejects_too_short(tmp_path):
audio = _tone(0.5, amp=0.3)
path = tmp_path / "short.wav"
sf.write(str(path), audio, SR)
ok, err, _, _ = validate_and_load_reference_audio(str(path))
assert not ok
assert "too short" in (err or "").lower()
if __name__ == "__main__":
pytest.main([__file__, "-v"])
-118
View File
@@ -1,118 +0,0 @@
"""
Unit tests for the ``force_offline_if_cached`` helper.
Verifies that the helper mutates the cached module constants in
``huggingface_hub.constants`` and ``transformers.utils.hub`` — not just
``os.environ`` — and that concurrent users are refcount-coordinated so
one thread's exit can't strip another thread's offline protection.
NOTE: These tests mutate process-global state in ``huggingface_hub.constants``
and ``transformers.utils.hub``. They are not safe under cross-process
parallelism (e.g. ``pytest-xdist`` with ``--dist=loadfile``/``loadscope``);
run this file serially.
"""
import os
import sys
import threading
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).parent.parent))
from utils.hf_offline_patch import force_offline_if_cached # noqa: E402
def _hf_const():
import huggingface_hub.constants as hf_const
return hf_const
def _tf_hub():
import transformers.utils.hub as tf_hub
return tf_hub
def test_mutates_cached_huggingface_hub_constant():
original = _hf_const().HF_HUB_OFFLINE
with force_offline_if_cached(True, "t"):
assert _hf_const().HF_HUB_OFFLINE is True
assert original == _hf_const().HF_HUB_OFFLINE
def test_mutates_cached_transformers_constant():
original = _tf_hub()._is_offline_mode
with force_offline_if_cached(True, "t"):
assert _tf_hub()._is_offline_mode is True
assert original == _tf_hub()._is_offline_mode
def test_sets_env_variable():
original = os.environ.get("HF_HUB_OFFLINE")
with force_offline_if_cached(True, "t"):
assert "1" == os.environ.get("HF_HUB_OFFLINE")
assert original == os.environ.get("HF_HUB_OFFLINE")
def test_noop_when_not_cached():
before = _hf_const().HF_HUB_OFFLINE
with force_offline_if_cached(False, "t"):
assert before == _hf_const().HF_HUB_OFFLINE
def test_nested_contexts_respect_refcount():
original = _hf_const().HF_HUB_OFFLINE
with force_offline_if_cached(True, "outer"):
assert _hf_const().HF_HUB_OFFLINE is True
with force_offline_if_cached(True, "inner"):
assert _hf_const().HF_HUB_OFFLINE is True
# inner exit must not restore while outer is still active
assert _hf_const().HF_HUB_OFFLINE is True
assert original == _hf_const().HF_HUB_OFFLINE
def test_concurrent_threads_share_offline_window():
"""A slow thread must keep seeing offline mode even if a peer exits first."""
original = _hf_const().HF_HUB_OFFLINE
observations: list[bool] = []
errors: list[Exception] = []
barrier = threading.Barrier(2)
fast_exited = threading.Event()
def slow():
try:
with force_offline_if_cached(True, "slow"):
barrier.wait(timeout=5)
assert fast_exited.wait(timeout=5), "fast thread did not exit"
observations.append(_hf_const().HF_HUB_OFFLINE)
except Exception as exc: # noqa: BLE001
errors.append(exc)
def fast():
try:
with force_offline_if_cached(True, "fast"):
barrier.wait(timeout=5)
except Exception as exc: # noqa: BLE001
errors.append(exc)
finally:
fast_exited.set()
t_slow = threading.Thread(target=slow)
t_fast = threading.Thread(target=fast)
t_slow.start()
t_fast.start()
t_slow.join(timeout=5)
t_fast.join(timeout=5)
assert not t_slow.is_alive(), "slow thread did not finish"
assert not t_fast.is_alive(), "fast thread did not finish"
assert not errors, errors
assert observations == [True], "slow thread lost offline protection"
assert original == _hf_const().HF_HUB_OFFLINE
if __name__ == "__main__":
pytest.main([__file__, "-v"])
+1 -1
View File
@@ -175,7 +175,7 @@ async def main():
print(" ✅ Server is running") print(" ✅ Server is running")
# Test model # Test model
model_name = "qwen-tts-0.6B" model_name = "qwen-tts-0.6B" # Note: 0.6B currently maps to 1.7B on MLX
# Check current status # Check current status
print(f"\n📊 Checking status of {model_name}...") print(f"\n📊 Checking status of {model_name}...")
+3 -65
View File
@@ -199,66 +199,6 @@ def trim_tts_output(
return trimmed return trimmed
def preprocess_reference_audio(
audio: np.ndarray,
sample_rate: int,
peak_target: float = 0.95,
trim_top_db: float = 40.0,
edge_padding_ms: int = 100,
) -> np.ndarray:
"""
Clean up a reference-audio sample before validation/storage.
Removes DC offset, trims leading/trailing silence, and caps the peak so a
slightly-hot recording doesn't get rejected downstream as "clipping". The
goal is to accept reasonable real-world recordings — not to repair badly
distorted ones. True clipping artifacts inside the waveform can't be
recovered by peak scaling and will still sound bad.
Args:
audio: Mono audio array.
sample_rate: Sample rate of ``audio`` in Hz.
peak_target: Peak amplitude cap in [0, 1]. Applied only if the input
peak exceeds this value.
trim_top_db: Silence threshold for edge trimming, in dB below peak.
40 dB sits below normal speech dynamic range (≈30 dB) so soft
trailing syllables are preserved, while still catching obvious
leading/trailing silence. Lower values are more aggressive;
librosa's own default is 60.
edge_padding_ms: Milliseconds of padding to add back at each edge
*only if* trimming shortened the waveform, so TTS engines have a
brief silence to anchor on without ever making the output longer
than the input.
Returns:
Preprocessed audio array (float32).
"""
audio = audio.astype(np.float32, copy=False)
if audio.size == 0:
return audio
audio = audio - float(np.mean(audio))
trimmed, _ = librosa.effects.trim(audio, top_db=trim_top_db)
if 0 < trimmed.size < audio.size:
pad_each = int(sample_rate * edge_padding_ms / 1000)
# Never pad past the original length — for near-max-duration uploads
# an unconditional pad would push them over the 30 s ceiling and
# trigger a spurious "too long" rejection.
headroom = (audio.size - trimmed.size) // 2
pad = min(pad_each, max(headroom, 0))
if pad > 0:
trimmed = np.pad(trimmed, (pad, pad), mode="constant")
audio = trimmed
peak = float(np.abs(audio).max())
if peak > peak_target and peak > 0:
audio = audio * (peak_target / peak)
return audio
def validate_reference_audio( def validate_reference_audio(
audio_path: str, audio_path: str,
min_duration: float = 2.0, min_duration: float = 2.0,
@@ -292,16 +232,11 @@ def validate_and_load_reference_audio(
""" """
Validate and load reference audio in a single pass. Validate and load reference audio in a single pass.
Applies :func:`preprocess_reference_audio` before checks so that
slightly-hot recordings aren't rejected as clipping. Duration and RMS
checks run on the preprocessed waveform.
Returns: Returns:
Tuple of (is_valid, error_message, audio_array, sample_rate) Tuple of (is_valid, error_message, audio_array, sample_rate)
""" """
try: try:
audio, sr = load_audio(audio_path) audio, sr = load_audio(audio_path)
audio = preprocess_reference_audio(audio, sr)
duration = len(audio) / sr duration = len(audio) / sr
if duration < min_duration: if duration < min_duration:
@@ -313,6 +248,9 @@ def validate_and_load_reference_audio(
if rms < min_rms: if rms < min_rms:
return False, "Audio is too quiet or silent", None, None return False, "Audio is too quiet or silent", None, None
if np.abs(audio).max() > 0.99:
return False, "Audio is clipping (reduce input gain)", None, None
return True, None, audio, sr return True, None, audio, sr
except Exception as e: except Exception as e:
return False, f"Error validating audio: {str(e)}", None, None return False, f"Error validating audio: {str(e)}", None, None
+27 -110
View File
@@ -6,7 +6,6 @@ are already downloaded. Must be imported BEFORE mlx_audio.
import logging import logging
import os import os
import threading
from contextlib import contextmanager from contextlib import contextmanager
from pathlib import Path from pathlib import Path
from typing import Optional, Union from typing import Optional, Union
@@ -14,33 +13,13 @@ from typing import Optional, Union
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
# huggingface_hub reads ``HF_HUB_OFFLINE`` once at import time into
# ``huggingface_hub.constants.HF_HUB_OFFLINE``; transformers mirrors that into
# ``transformers.utils.hub._is_offline_mode`` at *its* import time. Toggling
# ``os.environ`` after either module is imported does not flip those cached
# bools, and the hot paths (``_http._default_backend_factory``,
# ``transformers.utils.hub.is_offline_mode``) read the bools — not the env.
# We mutate the cached constants directly, guarded by a refcount so
# concurrent inference threads share a single offline window safely.
_offline_lock = threading.RLock()
_offline_refcount = 0
_saved_env: Optional[str] = None
_saved_hf_const: Optional[bool] = None
_saved_transformers_const: Optional[bool] = None
@contextmanager @contextmanager
def force_offline_if_cached(is_cached: bool, model_label: str = ""): def force_offline_if_cached(is_cached: bool, model_label: str = ""):
"""Force offline mode for the duration of a cached-model operation. """Context manager that sets ``HF_HUB_OFFLINE=1`` while loading a cached model.
Flips ``HF_HUB_OFFLINE`` in the process env **and** in the cached bools
inside ``huggingface_hub.constants`` / ``transformers.utils.hub`` so HTTP
adapters and offline-mode checks actually see the change. Uses a refcount
so multiple concurrent inference threads share a single offline window
and the last one to exit restores state.
If *is_cached* is ``False`` the block runs normally (network allowed). If *is_cached* is ``False`` the block runs normally (network allowed).
If the offline load raises an error containing "offline" we automatically
retry with network access so a partially-cached model still works.
Args: Args:
is_cached: Whether the model weights are already on disk. is_cached: Whether the model weights are already on disk.
@@ -50,96 +29,34 @@ def force_offline_if_cached(is_cached: bool, model_label: str = ""):
yield yield
return return
global _offline_refcount, _saved_env, _saved_hf_const, _saved_transformers_const original_value = os.environ.get("HF_HUB_OFFLINE")
os.environ["HF_HUB_OFFLINE"] = "1"
with _offline_lock: logger.info(
if _offline_refcount == 0: "[offline-guard] %s is cached — forcing HF_HUB_OFFLINE=1",
# Snapshot prior state, apply new state, roll back on *any* model_label or "model",
# failure. Catching only ImportError here would let a partially )
# broken install (RuntimeError, AttributeError from a half-init
# module, etc.) leave the cached HF constants mutated without
# bumping the refcount — a persistent offline leak that outlives
# the process and is miserable to debug.
prev_env = os.environ.get("HF_HUB_OFFLINE")
prev_hf: Optional[bool] = None
prev_tf: Optional[bool] = None
try:
try:
import huggingface_hub.constants as hf_const
prev_hf = hf_const.HF_HUB_OFFLINE
hf_const.HF_HUB_OFFLINE = True
except ImportError:
prev_hf = None
try:
import transformers.utils.hub as tf_hub
prev_tf = tf_hub._is_offline_mode
tf_hub._is_offline_mode = True
except ImportError:
prev_tf = None
os.environ["HF_HUB_OFFLINE"] = "1"
except BaseException:
# Roll back whatever we already changed, then re-raise so
# the caller sees the real failure.
if prev_hf is not None:
try:
import huggingface_hub.constants as hf_const
hf_const.HF_HUB_OFFLINE = prev_hf
except ImportError:
pass
if prev_tf is not None:
try:
import transformers.utils.hub as tf_hub
tf_hub._is_offline_mode = prev_tf
except ImportError:
pass
if prev_env is not None:
os.environ["HF_HUB_OFFLINE"] = prev_env
else:
os.environ.pop("HF_HUB_OFFLINE", None)
raise
_saved_env = prev_env
_saved_hf_const = prev_hf
_saved_transformers_const = prev_tf
logger.info(
"[offline-guard] %s is cached — forcing offline mode",
model_label or "model",
)
_offline_refcount += 1
try: try:
yield yield
except Exception as exc:
if "offline" in str(exc).lower():
logger.warning(
"[offline-guard] Offline load failed for %s, retrying with network: %s",
model_label or "model",
exc,
)
# Restore original env and retry — caller must wrap the load
# inside force_offline_if_cached so retrying here isn't possible.
# Instead, propagate a flag via the exception so the caller can
# decide. For simplicity we just let it fall through to the
# finally block and re-raise.
raise
raise
finally: finally:
with _offline_lock: if original_value is not None:
_offline_refcount -= 1 os.environ["HF_HUB_OFFLINE"] = original_value
if _offline_refcount == 0: else:
if _saved_env is not None: os.environ.pop("HF_HUB_OFFLINE", None)
os.environ["HF_HUB_OFFLINE"] = _saved_env
else:
os.environ.pop("HF_HUB_OFFLINE", None)
if _saved_hf_const is not None:
try:
import huggingface_hub.constants as hf_const
hf_const.HF_HUB_OFFLINE = _saved_hf_const
except ImportError:
pass
if _saved_transformers_const is not None:
try:
import transformers.utils.hub as tf_hub
tf_hub._is_offline_mode = _saved_transformers_const
except ImportError:
pass
_saved_env = None
_saved_hf_const = None
_saved_transformers_const = None
def patch_huggingface_hub_offline(): def patch_huggingface_hub_offline():
+4 -7
View File
@@ -17,7 +17,7 @@
}, },
"app": { "app": {
"name": "@voicebox/app", "name": "@voicebox/app",
"version": "0.4.1", "version": "0.2.0",
"dependencies": { "dependencies": {
"@dnd-kit/core": "^6.3.1", "@dnd-kit/core": "^6.3.1",
"@dnd-kit/sortable": "^10.0.0", "@dnd-kit/sortable": "^10.0.0",
@@ -72,10 +72,9 @@
}, },
"landing": { "landing": {
"name": "@voicebox/landing", "name": "@voicebox/landing",
"version": "0.4.1", "version": "0.2.0",
"dependencies": { "dependencies": {
"@fontsource/space-grotesk": "^5.2.10", "@fontsource/space-grotesk": "^5.2.10",
"@icons-pack/react-simple-icons": "^13.13.0",
"@radix-ui/react-separator": "^1.1.8", "@radix-ui/react-separator": "^1.1.8",
"@radix-ui/react-slot": "^1.2.4", "@radix-ui/react-slot": "^1.2.4",
"autoprefixer": "^10.4.17", "autoprefixer": "^10.4.17",
@@ -101,7 +100,7 @@
}, },
"tauri": { "tauri": {
"name": "@voicebox/tauri", "name": "@voicebox/tauri",
"version": "0.4.1", "version": "0.2.0",
"dependencies": { "dependencies": {
"@tauri-apps/api": "^2.0.0", "@tauri-apps/api": "^2.0.0",
"@tauri-apps/plugin-dialog": "^2.0.0", "@tauri-apps/plugin-dialog": "^2.0.0",
@@ -124,7 +123,7 @@
}, },
"web": { "web": {
"name": "@voicebox/web", "name": "@voicebox/web",
"version": "0.4.1", "version": "0.2.0",
"dependencies": { "dependencies": {
"@tanstack/react-query": "^5.0.0", "@tanstack/react-query": "^5.0.0",
"react": "^18.3.0", "react": "^18.3.0",
@@ -288,8 +287,6 @@
"@humanwhocodes/object-schema": ["@humanwhocodes/[email protected]", "", {}, "sha512-93zYdMES/c1D69yZiKDBj0V24vqNzB/koF26KPaagAfd3P/4gUlh3Dys5ogAK+Exi9QyzlD8x/08Zt7wIKcDcA=="], "@humanwhocodes/object-schema": ["@humanwhocodes/[email protected]", "", {}, "sha512-93zYdMES/c1D69yZiKDBj0V24vqNzB/koF26KPaagAfd3P/4gUlh3Dys5ogAK+Exi9QyzlD8x/08Zt7wIKcDcA=="],
"@icons-pack/react-simple-icons": ["@icons-pack/[email protected]", "", { "peerDependencies": { "react": "^16.13 || ^17 || ^18 || ^19" } }, "sha512-B5HhQMIpcSH4z8IZ8HFhD59CboHceKYMpPC9kAwGyKntvPdyJJv26DLu4Z1wAjcCLyrJhf11tMhiQGom9Rxb9g=="],
"@img/colour": ["@img/[email protected]", "", {}, "sha512-A5P/LfWGFSl6nsckYtjw9da+19jB8hkJ6ACTGcDfEJ0aE+l2n2El7dsVM7UVHZQ9s2lmYMWlrS21YLy2IR1LUw=="], "@img/colour": ["@img/[email protected]", "", {}, "sha512-A5P/LfWGFSl6nsckYtjw9da+19jB8hkJ6ACTGcDfEJ0aE+l2n2El7dsVM7UVHZQ9s2lmYMWlrS21YLy2IR1LUw=="],
"@img/sharp-darwin-arm64": ["@img/[email protected]", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="], "@img/sharp-darwin-arm64": ["@img/[email protected]", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="],
+10 -40
View File
@@ -1,6 +1,6 @@
# Voicebox Project Status & Roadmap # Voicebox Project Status & Roadmap
> Last updated: 2026-04-18 | Current version: **v0.4.1** | 232 open issues | 12 open PRs > Last updated: 2026-04-18 | Current version: **v0.4.1** | ~155 open issues | 12 open PRs
--- ---
@@ -224,8 +224,6 @@ POST /generate
- **Blackwell (RTX 50-series) CUDA**: cu128 + sm_120 kernel support shipped (PR #401, #316), but users still report `cudaErrorNoKernelImageForDevice` (#417, #400, #396, #395, #390, #362) — likely a stale CUDA binary on upgraded installs. Needs a follow-up diagnostic / forced re-download path. - **Blackwell (RTX 50-series) CUDA**: cu128 + sm_120 kernel support shipped (PR #401, #316), but users still report `cudaErrorNoKernelImageForDevice` (#417, #400, #396, #395, #390, #362) — likely a stale CUDA binary on upgraded installs. Needs a follow-up diagnostic / forced re-download path.
- **Long text 50k character limit** (#464, #365, #354): Still hit on GPU despite chunking (PR #266). Chunking reliability needs another pass. - **Long text 50k character limit** (#464, #365, #354): Still hit on GPU despite chunking (PR #266). Chunking reliability needs another pass.
- **ROCm on RDNA 3/4** (#469): `HSA_OVERRIDE_GFX_VERSION` is hardcoded and harms newer cards. - **ROCm on RDNA 3/4** (#469): `HSA_OVERRIDE_GFX_VERSION` is hardcoded and harms newer cards.
- **`flash-attn is not installed` warning on every platform (cosmetic, common user complaint)**: Our transformer-based engines (Chatterbox / Qwen) emit `Warning: flash-attn is not installed. Will only run the manual PyTorch version. Please install flash-attn for faster inference.` on every startup, on every platform — we don't pin `flash-attn` in requirements because installing it is fragile and version-sensitive. Fallback is PyTorch SDPA, which is near-FA2 throughput on Ampere+ and is what actually runs. **Per-platform reality:** (a) **macOS/Apple Silicon** — FlashAttention is CUDA-only, irrelevant here; MLX has its own attention kernels. (b) **Linux** — `pip install flash-attn --no-build-isolation` works but takes 20+ min to compile. (c) **Windows** — no official support (Dao-AILab README still says only "Might work"; source builds routinely fail on recent CUDA/MSVC, issues #1715, #1828, #2395). Windows users can install community prebuilt wheels from `kingbri1/flash-attention` or `bdashore3/flash-attention` (latest v2.8.3, Aug 2025; `win_amd64` wheels for CUDA 12.4/12.8, Torch 2.6–2.9, Python 3.10–3.13) matching their exact CUDA/Torch/Python, or use WSL2. **Native-Windows alternatives worth considering as a build-time swap:** SageAttention (thu-ml, Apache 2.0, claims 2–5× over FA2) and xformers (official Windows wheels). **Action for us:** troubleshooting doc now covers it (see `docs/content/docs/overview/troubleshooting.mdx`), and we should optionally suppress the warning via `logging.getLogger(...).setLevel(ERROR)` at backend import since the fallback is functionally fine.
- **WebAudio playback dies after audio-session interruption** (#41, plus an internal repro where the app is backgrounded long enough): WaveSurfer's `AudioContext` gets suspended by macOS — either because another app grabs the audio output, or because the WKWebView throttles when backgrounded. `play()` resolves and `timeupdate` can still fire, but no audio reaches the output. Only app restart fixes it. **Things already tried that didn't work:** (a) swapping WaveSurfer backend away from WebAudio — introduced more bugs, not an option; (b) remount hook on the player — doesn't help because a freshly-created `AudioContext` is born suspended and only resumes on a user gesture. PR #293 was a prior partial fix that doesn't cover this path. **Next thing to try** (not yet attempted — confirmed via grep of `AudioPlayer.tsx`): call `wavesurfer.getMediaElement().getGainNode().context.resume()` on the play button click (the click itself is a valid user gesture), plus a `visibilitychange` + `statechange` listener as belt-and-suspenders. The `ctx.resume()` pattern already exists in the codebase at `useStoryPlayback.ts:52` — just not wired into the main player.
--- ---
@@ -305,11 +303,9 @@ POST /generate
Still reported. Users get stuck downloads, can't resume, offline mode edge cases. Still reported. Users get stuck downloads, can't resume, offline mode edge cases.
**Key issues:** #475 (MAC CustomVoice install error), #449 (infinite loading macOS), #445 (can't download CustomVoice), #462 (Qwen requires internet even when loaded — regression from #150), #434 (infinite retry loop offline — PR #443 open), #432 (storage location change hangs when empty — partly fixed by PR #439/#433), #348 (TADA 3B Multilingual download fails), #336 (TADA model not listed in app), #275 (`No module named 'chatterbox'` on download), #304 (whisper-base feature extractor load error), #287 (macOS ARM `check_model_inputs` ImportError on new version), #181, #180. **Key issues:** #475 (MAC CustomVoice install error), #449 (infinite loading macOS), #445 (can't download CustomVoice), #462 (Qwen requires internet even when loaded — regression from #150), #434 (infinite retry loop offline — PR #443 open), #432 (storage location change hangs when empty — partly fixed by PR #439/#433), #181, #180.
**Fix path:** PR #443 addresses infinite offline retry. CustomVoice-specific download failures (#475, #445) need triage — likely related to frozen-binary import fixes in PR #438. TADA cluster (#336, #348) and macOS ARM import regressions (#287, #275, #304) need a dedicated triage pass. **Fix path:** PR #443 addresses infinite offline retry. CustomVoice-specific download failures (#475, #445) need triage — likely related to frozen-binary import fixes in PR #438.
**Qwen 0.6B-downloads-1.7B reports:** **#485** (2026-04-19), **#423** (macOS M1), **#329**. Originally a stale-fallback bug: `mlx-community/Qwen3-TTS-12Hz-0.6B-Base-bf16` wasn't published when MLX support shipped, so the 0.6B slot was aliased to the 1.7B repo. The 0.6B bf16 conversion is live now and both `backend/backends/mlx_backend.py` and `backend/backends/__init__.py` point at their correct repos. Qwen CustomVoice is unaffected — it runs via PyTorch on all platforms, both sizes always have dedicated repos.
### Language Requests (ongoing) ### Language Requests (ongoing)
@@ -368,9 +364,6 @@ Notable:
- **#383** — Concatenate partial reference audio into generated audio - **#383** — Concatenate partial reference audio into generated audio
- **#382** — Lightning.ai support - **#382** — Lightning.ai support
- **#376** — Remote mode - **#376** — Remote mode
- **#353** — Audio transcoding
- **#317** — Voice pitch control
- **#189** — "Auto" language option
- **#173** — Vocal intonation/inflection control - **#173** — Vocal intonation/inflection control
- **#165, #270** — Audiobook mode (PR #154 open) - **#165, #270** — Audiobook mode (PR #154 open)
- **#242** — Seed value pinning - **#242** — Seed value pinning
@@ -378,40 +371,17 @@ Notable:
- **#235** — Finetuned Qwen3-TTS tokenizer (PR #253 open) - **#235** — Finetuned Qwen3-TTS tokenizer (PR #253 open)
- **#144** — Copy text to clipboard - **#144** — Copy text to clipboard
### Housekeeping / Triage Needed
| Issue | Reason |
|-------|--------|
| **#431**, **#408** | Spam — Chinese "free Claude API" promos. Close. |
| **#398** ("Excelente") | Non-issue. Close. |
| **#357** | Informational — project featured in Awesome MLX. Close after acknowledgement. |
| **#374**, **#377** | Version-release questions, no bug. Close. |
| **#306** ("voice model"), **#389** ("New model"), **#473** ("New functionality") | Title-only issues, no content. Request details or close. |
| **#309** | Uninstall/cleanup question. Answer and close. |
| **#241** | "How to use in Colab" — support question, not a bug. |
| **#423** / **#485** / **#329** | Stale MLX fallback to 1.7B repo — fixed; 0.6B bf16 conversion now live on `mlx-community`, registry points at correct repo on both backends. |
| **#336** / **#348** | TADA download/registration cluster — triage together. |
| **#287** / **#275** / **#304** | macOS ARM import regressions on new version — likely one root cause. |
| **#292**, **#349** | Possibly already fixed by merged PRs (#321/#412 and #345). Verify + close. |
**~70 older issues (pre-#170) not individually categorized above.** Most are long-tail support questions or duplicates of problems now addressed by the multi-engine / model-registry work. A dedicated backlog-sweep pass is overdue.
### Bugs (ongoing) ### Bugs (ongoing)
| Category | Issues | | Category | Issues |
|----------|--------| |----------|--------|
| Generation failures | #476, #467, #452, #459 (voice clone fetch error), #468 (tada-1b marked error), #437, #300, #301, #282 | | Generation failures | #476, #467, #452, #459 (voice clone fetch error), #468 (tada-1b marked error), #437, #282 |
| Audio quality | #456 (clipping errors v0.4.0), #436 (emotion labels), #333 (pitch/echo), #307 (by-model breakdown), #340 (all generations say "www...") | | Audio quality | #456 (clipping errors v0.4.0), #436 (emotion labels), #333 (pitch/echo), #307 (by-model breakdown) |
| Transcription | #371 (fails every time), #291 (extract transcription from generated audio) | | File ops | #477 (spacy_pkuseg dict missing on frozen Windows build), #472 (storage location change) |
| Effects / presets | #349 ("Failed to save" when creating effects presets — possibly fixed by merged #345) | | Windows | #466 (install problem), #273 (port 8000 conflict) |
| File ops | #477 (spacy_pkuseg dict missing on frozen Windows build), #472 (storage location change), #283 (allow longer files for voice creation + in-app trim), #350 (failed to add sample) | | Linux | #471 (thread-safe PULSE_SOURCE), #413 (Arch build), #409 (Kubuntu build), #341 |
| History | #292 (can't delete failed generations — possibly fixed by merged #321/#412) | | macOS | #441 (older macOS), #369 (malware flag), #171 (ARM64 binary won't open) |
| Windows | #466 (install problem), #375 (WinError 5 access denied), #273 (port 8000 conflict), #201 (model doesn't stay loaded) | | Profile/UI | #360 (Kokoro profile hides others — partly addressed by auto-switch), #299 (drag-drop on Win11), #329 (size selector state bug) |
| Linux | #471 (thread-safe PULSE_SOURCE), #413 (Arch build), #409 (Kubuntu build), #351, #341 |
| macOS | #441 (older macOS), #369 (malware flag), #334 (microphone permission), #287 (`check_model_inputs` ImportError — regression), #171 (ARM64 binary won't open) |
| Profile/UI | #360 (Kokoro profile hides others — partly addressed by auto-switch), #299 (drag-drop on Win11), #329 (size selector state bug), #393 (stuck loading screen after reinstall to new dir) |
| Integrations | #397 (SAMMI-bot 422 Unprocessable Entity) |
| Audio playback / session | **#41** (macOS: Voicebox goes silent after another app takes audio output; restart restores it) — see deep-dive below |
| Database | #174 (sqlite3 IntegrityError) | | Database | #174 (sqlite3 IntegrityError) |
--- ---
+311
View File
@@ -0,0 +1,311 @@
---
title: "Troubleshooting Guide"
description: "Common issues and solutions for Voicebox"
---
Common issues and solutions for Voicebox.
## Installation Issues
### macOS: "Voicebox cannot be opened because it is from an unidentified developer"
**Solution:**
1. Right-click the `.dmg` file
2. Select "Open"
3. Click "Open" in the security dialog
4. Alternatively, go to System Settings → Privacy & Security → Allow Voicebox
### Windows: "Windows protected your PC"
**Solution:**
1. Click "More info"
2. Click "Run anyway"
3. Windows Defender may flag new software; this is normal for unsigned apps
### Linux: AppImage won't run
**Solution:**
```bash
chmod +x voicebox-*.AppImage
./voicebox-*.AppImage
```
## Runtime Issues
### Server won't start
**Symptoms:** App opens but shows "Server not connected"
**Solutions:**
1. **Check Python installation**
```bash
python --version # Should be 3.11+
```
2. **Check server binary exists**
- Look in `tauri/src-tauri/binaries/` for your platform
- Binary should match your system architecture
3. **Check permissions**
```bash
# macOS/Linux
chmod +x tauri/src-tauri/binaries/voicebox-server-*
```
4. **Check logs**
- macOS: Open Console.app and search for "voicebox"
- Linux: Check `~/.local/share/voicebox/` for logs
- Windows: Check Event Viewer
### "Model download failed"
**Symptoms:** First generation fails with download error
**Solutions:**
1. **Check internet connection**
- Models download from HuggingFace Hub (~2-4GB)
- First download may take several minutes
2. **Check disk space**
- Models are cached in `~/.cache/huggingface/`
- Ensure at least 5GB free space
3. **Manual download** (if automatic fails)
```bash
pip install huggingface_hub
huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base
```
### "Out of memory" errors
**Symptoms:** Generation fails with CUDA/VRAM errors
**Solutions:**
1. **Use smaller model**
- Switch to 0.6B model instead of 1.7B
- Settings → Model Management → Load 0.6B
2. **Close other applications**
- Free up GPU memory
- Close browser tabs, other ML apps
3. **Use CPU mode**
- Slower but works without GPU
- Backend automatically falls back to CPU
### MLX "Failed to load the default metallib" error (Apple Silicon)
**Symptoms:** Generation fails with "library not found" or "metallib" errors
**Solutions:**
1. **Rebuild server binary**
```bash
just build-server
```
The build script automatically includes MLX Metal shader libraries on Apple Silicon.
2. **Check MLX installation**
```bash
pip install -r backend/requirements-mlx.txt
```
3. **Verify backend detection**
- Check server logs for "Backend: MLX"
- If showing "Backend: PYTORCH", MLX may not be installed correctly
### Audio playback issues
**Symptoms:** Generated audio won't play
**Solutions:**
1. **Check audio format**
- Audio is saved as WAV files
- Ensure your system supports WAV playback
2. **Try downloading audio**
- Right-click → Download
- Play in external player
3. **Check browser permissions** (web version)
- Allow audio autoplay in browser settings
### Slow generation
**Symptoms:** Generation takes >30 seconds
**Solutions:**
1. **Check backend type** (Apple Silicon)
- Check Settings → Server Status
- Should show "Backend: MLX" on Apple Silicon
- If showing "Backend: PYTORCH", install MLX: `pip install -r backend/requirements-mlx.txt`
- MLX provides 4-5x faster inference on Apple Silicon
2. **Use GPU** (if available)
- Check Settings → Server Status
- Should show "GPU available: true"
- Apple Silicon: Should show "Metal (Apple Silicon via MLX)"
- Windows/Linux: Should show "CUDA" if GPU available
3. **Enable caching**
- Voice prompts are cached automatically
- Second generation with same voice should be faster
4. **Use smaller model**
- 0.6B model is faster than 1.7B
- Quality difference is minimal for most voices
5. **Check system resources**
- Close other CPU/GPU intensive apps
- Ensure adequate RAM (8GB+ recommended)
## API Issues
### "Connection refused" when using API
**Solutions:**
1. **Check server is running**
```bash
curl http://localhost:17493/health
```
2. **Check remote mode**
- If connecting remotely, ensure server is started with `--host 0.0.0.0`
- Check firewall settings
3. **Check port availability**
- The current local app and dev workflow uses port 17493 by default
- Ensure no other service is using it
### CORS errors in browser
**Solutions:**
1. **Use desktop app** (recommended)
- Desktop app doesn't have CORS restrictions
2. **Configure CORS** (for web deployment)
- Update `backend/main.py` CORS settings
- Add your domain to allowed origins
## Update Issues
### "Update check failed"
**Solutions:**
1. **Check internet connection**
- Updates are fetched from GitHub releases
2. **Check GitHub access**
- Ensure `github.com` is accessible
- Check firewall/proxy settings
3. **Manual update**
- Download latest release from GitHub
- Install manually
### "Invalid signature" error
**Solutions:**
1. **Re-download installer**
- Signature may be corrupted
- Download fresh copy from GitHub
2. **Check release integrity**
- Verify `.sig` file matches installer
- Report issue if signature is invalid
## Data Issues
### Profiles disappeared
**Solutions:**
1. **Check data directory**
- macOS: `~/Library/Application Support/sh.voicebox.app/`
- Windows: `%APPDATA%/sh.voicebox.app/`
- Linux: `~/.config/sh.voicebox.app/`
2. **Check database**
- Database: `data/voicebox.db`
- Ensure file exists and is readable
3. **Restore from backup**
- Profiles can be exported/imported
- Check for backup files
### "Database locked" error
**Solutions:**
1. **Close other instances**
- Ensure only one Voicebox instance is running
2. **Restart app**
- Close and reopen Voicebox
3. **Check file permissions**
- Ensure database file is writable
- Check directory permissions
## Development Issues
### Build fails
**Solutions:**
1. **Check Rust installation**
```bash
rustc --version
rustup update
```
2. **Check Tauri dependencies**
```bash
cd tauri
bun install
```
3. **Clean build**
```bash
cd tauri/src-tauri
cargo clean
cd ../..
just build
```
### API client generation fails
**Solutions:**
1. **Start backend server**
```bash
just dev-backend
```
2. **Check OpenAPI endpoint**
```bash
curl http://localhost:17493/openapi.json
```
3. **Regenerate client**
```bash
just generate-api
```
## Still Having Issues?
1. **Check existing issues**
- Search GitHub issues for similar problems
- Check closed issues for solutions
2. **Create new issue**
- Include:
- OS and version
- Voicebox version
- Steps to reproduce
- Error messages/logs
- Screenshots (if applicable)
3. **Get help**
- Check documentation in `docs/`
- Review `backend/README.md` for API details
- See `CONTRIBUTING.md` for development help
---
For more help, open an issue on [GitHub](https://github.com/jamiepine/voicebox/issues).
+3 -113
View File
@@ -29,14 +29,6 @@ Windows SmartScreen may warn that the app is unrecognized.
This is expected for unsigned applications. We're working on code signing for future releases. This is expected for unsigned applications. We're working on code signing for future releases.
</Callout> </Callout>
### Linux: AppImage Won't Run
**Solution:**
```bash
chmod +x voicebox-*.AppImage
./voicebox-*.AppImage
```
## Server Issues ## Server Issues
### Backend Server Won't Start ### Backend Server Won't Start
@@ -93,52 +85,6 @@ chmod +x voicebox-*.AppImage
</Accordion> </Accordion>
</AccordionGroup> </AccordionGroup>
### `flash-attn is not installed` Warning in Server Logs
**Symptoms:**
```
Warning: flash-attn is not installed. Will only run the manual PyTorch version.
Please install flash-attn for faster inference.
```
**This is harmless.** The warning is emitted by our transformer-based engines (Chatterbox / Qwen) on every startup. FlashAttention is an optional acceleration library — when it's not present, PyTorch's built-in scaled-dot-product attention (SDPA) runs instead, which is near-FA2 throughput on modern GPUs. Generation works normally.
**Why it shows up on every platform:**
- **Windows:** `flash-attn` has no official Windows support. The upstream project (Dao-AILab/flash-attention) still only says it *might* work, and source builds typically fail on recent CUDA/MSVC combinations.
- **macOS (Apple Silicon):** FlashAttention is CUDA-only and doesn't apply here at all. MLX has its own optimized attention kernels.
- **Linux:** It's not pinned in our requirements because installing it is fragile and version-sensitive; users who want it install it themselves.
**Solutions (all optional):**
<AccordionGroup>
<Accordion title="Ignore it (recommended)">
PyTorch SDPA is what actually runs the model, and on Ampere/Ada/Hopper GPUs it's within a few percent of FA2 for our workloads. You won't notice a meaningful speed difference.
</Accordion>
<Accordion title="Install flash-attn on Linux">
```bash
pip install flash-attn --no-build-isolation
```
Requires a matching CUDA toolkit. Build can take 20+ minutes.
</Accordion>
<Accordion title="Install flash-attn on Windows (community wheels)">
Official builds don't exist, but community maintainers publish prebuilt wheels:
- [kingbri1/flash-attention releases](https://github.com/kingbri1/flash-attention/releases)
- [bdashore3/flash-attention releases](https://github.com/bdashore3/flash-attention/releases)
Pick the wheel matching your exact CUDA + PyTorch + Python combination. Example:
```bash
pip install https://github.com/kingbri1/flash-attention/releases/download/v2.8.3/flash_attn-2.8.3+cu128torch2.8.0cxx11abiFALSE-cp312-cp312-win_amd64.whl
```
Alternatively, run Voicebox's backend inside WSL2 and use the standard Linux wheels.
</Accordion>
</AccordionGroup>
### Connection Timeout ### Connection Timeout
**Symptoms:** **Symptoms:**
@@ -228,34 +174,6 @@ This is expected behavior. The first generation downloads the selected TTS engin
</Accordion> </Accordion>
</AccordionGroup> </AccordionGroup>
### MLX "Failed to load the default metallib" (Apple Silicon)
**Symptoms:**
- Generation fails with "library not found" or "metallib" errors
- Server logs reference missing Metal shader libraries
**Solutions:**
<AccordionGroup>
<Accordion title="Rebuild the Server Binary">
```bash
just build-server
```
The build script bundles MLX Metal shader libraries on Apple Silicon automatically.
</Accordion>
<Accordion title="Reinstall MLX Dependencies">
```bash
pip install -r backend/requirements-mlx.txt
```
</Accordion>
<Accordion title="Verify Backend Detection">
Check Settings → Server Status. Should show **Backend: MLX** on Apple Silicon. If it shows **Backend: PYTORCH**, MLX isn't installed correctly.
</Accordion>
</AccordionGroup>
## Audio Issues ## Audio Issues
### No Audio Playback ### No Audio Playback
@@ -439,12 +357,7 @@ Restart the app to create a fresh database.
- Check your internet connection - Check your internet connection
- Check HuggingFace Hub status - Check HuggingFace Hub status
- Try using a VPN if HuggingFace is blocked in your region - Try using a VPN if HuggingFace is blocked in your region
- Manually download via the HuggingFace CLI and place in the cache directory: - Manually download and place in cache directory
```bash
pip install huggingface_hub
huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base
```
### Wrong Model Version ### Wrong Model Version
@@ -491,16 +404,6 @@ rmdir /s %USERPROFILE%\.cache\huggingface\hub\models--Qwen*
<Accordion title="Update GPU Drivers"> <Accordion title="Update GPU Drivers">
Outdated drivers can cause performance issues. Update to the latest NVIDIA drivers. Outdated drivers can cause performance issues. Update to the latest NVIDIA drivers.
</Accordion> </Accordion>
<Accordion title="Apple Silicon: Confirm MLX Backend">
Check Settings → Server Status. Should show **Backend: MLX** on Apple Silicon — MLX is 4–5× faster than PyTorch here. If it shows **Backend: PYTORCH**, reinstall MLX:
```bash
pip install -r backend/requirements-mlx.txt
```
GPU availability should read "Metal (Apple Silicon via MLX)".
</Accordion>
</AccordionGroup> </AccordionGroup>
### High Memory Usage ### High Memory Usage
@@ -514,21 +417,6 @@ rmdir /s %USERPROFILE%\.cache\huggingface\hub\models--Qwen*
- Clear generation history - Clear generation history
- Restart the app periodically - Restart the app periodically
## Update Issues
### "Update Check Failed"
**Solutions:**
- Confirm your internet connection — updates are fetched from GitHub releases.
- Ensure `github.com` is accessible and not blocked by a firewall or proxy.
- As a fallback, download the latest release from GitHub and install manually.
### "Invalid Signature" Error
**Solutions:**
- Re-download the installer — the signature may have been corrupted in transit.
- Verify the `.sig` file matches the installer; if it doesn't, file an issue.
## Remote Mode Issues ## Remote Mode Issues
### Can't Connect to Remote Server ### Can't Connect to Remote Server
@@ -594,3 +482,5 @@ python --version
# GPU info (if generation issues) # GPU info (if generation issues)
nvidia-smi # NVIDIA GPUs nvidia-smi # NVIDIA GPUs
``` ```
For more detailed troubleshooting, see the [TROUBLESHOOTING.md](https://github.com/jamiepine/voicebox/blob/main/docs/TROUBLESHOOTING.md) file in the repository.
+2 -3
View File
@@ -1,16 +1,15 @@
{ {
"name": "@voicebox/landing", "name": "@voicebox/landing",
"version": "0.4.2", "version": "0.4.1",
"description": "Landing page for voicebox.sh", "description": "Landing page for voicebox.sh",
"scripts": { "scripts": {
"dev": "next dev --turbo", "dev": "bun --bun next dev --turbo",
"build": "bun --bun next build", "build": "bun --bun next build",
"start": "bun --bun next start", "start": "bun --bun next start",
"lint": "next lint" "lint": "next lint"
}, },
"dependencies": { "dependencies": {
"@fontsource/space-grotesk": "^5.2.10", "@fontsource/space-grotesk": "^5.2.10",
"@icons-pack/react-simple-icons": "^13.13.0",
"@radix-ui/react-separator": "^1.1.8", "@radix-ui/react-separator": "^1.1.8",
"@radix-ui/react-slot": "^1.2.4", "@radix-ui/react-slot": "^1.2.4",
"autoprefixer": "^10.4.17", "autoprefixer": "^10.4.17",
+27 -31
View File
@@ -1,46 +1,42 @@
import { type NextRequest, NextResponse } from 'next/server'; import { type NextRequest, NextResponse } from 'next/server';
import { getLatestRelease } from '@/lib/releases';
export const dynamic = 'force-dynamic'; export const dynamic = 'force-dynamic';
// Pretty URLs from README / docs (e.g. /download/mac-arm) are kept for const PLATFORM_MAP: Record<
// compatibility, but we now always route through the /download page so users string,
// see context + a donate prompt + resources while the download kicks off. keyof Awaited<ReturnType<typeof getLatestRelease>>['downloadLinks']
// The page handles the actual file trigger itself — no more silent redirects > = {
// to GitHub or direct asset URLs.
const PLATFORM_ALIAS: Record<string, string> = {
'mac-arm': 'macArm', 'mac-arm': 'macArm',
macArm: 'macArm',
'mac-intel': 'macIntel', 'mac-intel': 'macIntel',
macIntel: 'macIntel',
windows: 'windows', windows: 'windows',
linux: 'linux',
}; };
function getPublicOrigin(request: NextRequest): string {
const forwardedHost = request.headers.get('x-forwarded-host');
const forwardedProto = request.headers.get('x-forwarded-proto');
if (forwardedHost && forwardedProto) {
// Behind reverse proxies/CDNs, request.url can be an internal origin
// (for example localhost:8080). Prefer forwarded headers so redirects
// keep users on the public domain.
return `${forwardedProto}://${forwardedHost}`;
}
return new URL(request.url).origin;
}
export async function GET( export async function GET(
request: NextRequest, _request: NextRequest,
{ params }: { params: Promise<{ platform: string }> }, { params }: { params: Promise<{ platform: string }> },
) { ) {
const origin = getPublicOrigin(request);
const { platform } = await params; const { platform } = await params;
// No prebuilt Linux binary yet — send straight to the build-from-source page. const key = PLATFORM_MAP[platform];
if (platform === 'linux') {
return NextResponse.redirect(new URL('/linux-install', origin), 307); if (!key) {
return NextResponse.json(
{ error: `Unknown platform: ${platform}. Use: ${Object.keys(PLATFORM_MAP).join(', ')}` },
{ status: 404 },
);
}
try {
const release = await getLatestRelease();
const url = release.downloadLinks[key];
if (!url) {
return NextResponse.json({ error: `No download available for ${platform}` }, { status: 404 });
}
return NextResponse.redirect(url);
} catch {
return NextResponse.redirect(`https://github.com/jamiepine/voicebox/releases/latest`);
} }
const normalized = PLATFORM_ALIAS[platform];
const target = new URL('/download', origin);
if (normalized) target.searchParams.set('platform', normalized);
return NextResponse.redirect(target, 307);
} }
-313
View File
@@ -1,313 +0,0 @@
'use client';
import {
ArrowLeft,
Bot,
Coffee,
Download as DownloadIcon,
FileText,
Github,
} from 'lucide-react';
import Image from 'next/image';
import Link from 'next/link';
import { useEffect, useMemo, useState } from 'react';
import { AppleIcon, LinuxIcon, WindowsIcon } from '@/components/PlatformIcons';
import { Button } from '@/components/ui/button';
import { DONATE_URL, GITHUB_RELEASES_PAGE, GITHUB_REPO } from '@/lib/constants';
import type { DownloadLinks } from '@/lib/releases';
type Platform = keyof DownloadLinks;
type PlatformMeta = {
key: Platform;
label: string;
description: string;
icon: React.ComponentType<{ className?: string }>;
};
const PLATFORMS: PlatformMeta[] = [
{ key: 'macArm', label: 'macOS', description: 'Apple Silicon', icon: AppleIcon },
{ key: 'macIntel', label: 'macOS', description: 'Intel (x64)', icon: AppleIcon },
{ key: 'windows', label: 'Windows', description: '64-bit (MSI)', icon: WindowsIcon },
{ key: 'linux', label: 'Linux', description: 'Build from source', icon: LinuxIcon },
];
function detectPlatform(): Platform | null {
if (typeof navigator === 'undefined') return null;
const ua = navigator.userAgent;
if (/Windows/i.test(ua)) return 'windows';
if (/Linux/i.test(ua) && !/Android/i.test(ua)) return 'linux';
if (/Mac/i.test(ua)) {
// Apple Silicon Safari reports "Intel" for compat; default to ARM since
// M-series is the majority. Users can click the Intel button if needed.
return 'macArm';
}
return null;
}
function parseQueryPlatform(search: string): Platform | null {
const params = new URLSearchParams(search);
const raw = params.get('platform');
if (!raw) return null;
// Accept both camelCase and hyphenated forms (/download/mac-arm → ?platform=mac-arm).
const normalized = raw
.toLowerCase()
.replace(/[-_\s]/g, '')
.replace('macarm', 'macArm')
.replace('macintel', 'macIntel');
const valid: Platform[] = ['macArm', 'macIntel', 'windows', 'linux'];
return (valid as string[]).includes(normalized) ? (normalized as Platform) : null;
}
export default function DownloadPage() {
const [links, setLinks] = useState<DownloadLinks | null>(null);
const [linksError, setLinksError] = useState(false);
const [platform, setPlatform] = useState<Platform | null>(null);
const [triggered, setTriggered] = useState(false);
useEffect(() => {
const fromQuery = parseQueryPlatform(window.location.search);
const resolved = fromQuery ?? detectPlatform();
// No prebuilt Linux binary yet — send Linux users to the build-from-source
// instructions instead of sitting on /download trying to trigger a
// download that doesn't exist.
if (resolved === 'linux') {
window.location.replace('/linux-install');
return;
}
setPlatform(resolved);
}, []);
useEffect(() => {
let cancelled = false;
fetch('/api/releases')
.then((r) => {
if (!r.ok) throw new Error(`releases ${r.status}`);
return r.json();
})
.then((data) => {
if (cancelled) return;
if (data.downloadLinks) setLinks(data.downloadLinks as DownloadLinks);
})
.catch(() => {
if (!cancelled) setLinksError(true);
});
return () => {
cancelled = true;
};
}, []);
useEffect(() => {
if (triggered || !links || !platform) return;
const url = links[platform];
if (!url) return;
const a = document.createElement('a');
a.href = url;
a.rel = 'noopener';
a.style.display = 'none';
document.body.appendChild(a);
a.click();
document.body.removeChild(a);
setTriggered(true);
}, [triggered, links, platform]);
const activeMeta = useMemo(
() => PLATFORMS.find((p) => p.key === platform) ?? null,
[platform],
);
return (
<div className="min-h-screen bg-background">
{/* Minimal branded header */}
<header className="border-b border-border/50">
<div className="mx-auto flex max-w-5xl items-center justify-between px-6 py-4">
<Link href="/" className="flex items-center gap-2.5">
<Image
src="/voicebox-logo-app.webp"
alt="Voicebox"
width={28}
height={28}
className="h-7 w-7"
/>
<span className="text-[15px] font-semibold text-foreground">Voicebox</span>
</Link>
<Link
href="/"
className="flex items-center gap-1.5 text-sm text-muted-foreground hover:text-foreground transition-colors"
>
<ArrowLeft className="h-3.5 w-3.5" />
Back to voicebox.sh
</Link>
</div>
</header>
<main className="mx-auto max-w-5xl px-6 py-16 md:py-24">
{/* Hero */}
<div className="flex flex-col md:flex-row md:items-center gap-10 md:gap-14">
<Image
src="/voicebox-logo-app.webp"
alt="Voicebox"
width={200}
height={200}
priority
className="h-32 w-32 md:h-44 md:w-44 shrink-0 drop-shadow-2xl"
/>
<div className="flex-1 min-w-0 text-center md:text-left">
{triggered ? (
<>
<h1 className="text-4xl md:text-5xl font-semibold tracking-tight text-foreground mb-4">
Your download has started.
</h1>
<p className="text-lg text-muted-foreground">
{activeMeta
? `Downloading Voicebox for ${activeMeta.label} (${activeMeta.description}). Check your downloads folder.`
: 'Check your downloads folder for Voicebox.'}
</p>
</>
) : (
<>
<h1 className="text-4xl md:text-5xl font-semibold tracking-tight text-foreground mb-4">
{linksError ? "We couldn't load the latest release." : 'Download Voicebox'}
</h1>
<p className="text-lg text-muted-foreground">
{linksError
? 'Our release server is temporarily unreachable. Please try again in a moment.'
: 'Pick your platform to get started.'}
</p>
</>
)}
</div>
</div>
{/* Platform buttons — always visible as a fallback */}
{linksError ? (
<div className="mt-12 rounded-xl border border-border bg-card/60 backdrop-blur-sm p-6 text-center">
<p className="text-sm text-muted-foreground mb-4">
If this keeps happening, you can{' '}
<a
href={`${GITHUB_RELEASES_PAGE}/latest`}
target="_blank"
rel="noopener noreferrer"
className="text-accent underline underline-offset-2 hover:text-accent/80"
>
browse releases on GitHub
</a>
{' '}and grab the build for your platform manually.
</p>
</div>
) : (
<div className="mt-12 rounded-xl border border-border bg-card/60 backdrop-blur-sm p-6">
<h2 className="text-sm font-medium text-foreground mb-4">
{triggered ? 'Download not working?' : 'Choose your platform'}
</h2>
<div className="grid grid-cols-1 sm:grid-cols-2 gap-3">
{PLATFORMS.map((meta) => {
const isLinux = meta.key === 'linux';
const url = isLinux ? '/linux-install' : links?.[meta.key];
const isActive = meta.key === platform;
const disabled = !isLinux && !url;
return (
<a
key={meta.key}
href={url ?? '#'}
{...(isLinux ? {} : { download: true })}
aria-disabled={disabled}
onClick={(e) => {
if (disabled) e.preventDefault();
}}
className={`flex items-center rounded-xl border px-5 py-4 transition-all group ${
isActive
? 'border-accent/40 bg-accent/5 hover:border-accent/60'
: 'border-border bg-card/40 hover:border-accent/30 hover:bg-card'
} ${disabled ? 'opacity-50 cursor-not-allowed' : ''}`}
>
<meta.icon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" />
<div className="ml-4 flex-1">
<div className="text-sm font-medium text-foreground">{meta.label}</div>
<div className="text-xs text-muted-foreground">{meta.description}</div>
</div>
<DownloadIcon className="h-4 w-4 text-muted-foreground/60 group-hover:text-accent transition-colors" />
</a>
);
})}
</div>
</div>
)}
{/* Donate — prominent, heartfelt, post-click context */}
<div className="mt-16 rounded-2xl border border-border bg-gradient-to-br from-card via-card/80 to-background backdrop-blur-sm p-8 md:p-10 overflow-hidden relative">
<div className="absolute top-0 right-0 w-64 h-64 bg-[#FFDD00]/5 rounded-full blur-3xl -translate-y-1/2 translate-x-1/2 pointer-events-none" />
<div className="relative">
<div className="inline-flex items-center gap-2 rounded-full border border-[#FFDD00]/30 bg-[#FFDD00]/10 px-3 py-1 mb-4">
<Coffee className="h-3 w-3 text-[#FFDD00]" />
<span className="text-[11px] font-medium uppercase tracking-wider text-[#FFDD00]">
Hi from the maintainer
</span>
</div>
<h2 className="text-2xl md:text-3xl font-semibold tracking-tight text-foreground mb-4">
Jamie here — Voicebox is a side project.
</h2>
<p className="text-muted-foreground leading-relaxed mb-6 max-w-2xl">
I build and maintain Voicebox in my spare time. It's completely
free, open source, runs entirely on your machine — no accounts, no
cloud, no subscriptions, no upsells. If it saves you an ElevenLabs
bill or just made your day, a coffee genuinely helps me keep
shipping updates, adding new models, and fixing bugs. Every little
bit keeps the lights on.
</p>
<Button asChild size="lg" className="bg-[#FFDD00]/10 border-[#FFDD00]/30 text-[#FFDD00] hover:bg-[#FFDD00]/20 hover:border-[#FFDD00]/50">
<a href={DONATE_URL} target="_blank" rel="noopener noreferrer">
<Coffee className="h-4 w-4 mr-2" />
Buy me a coffee
</a>
</Button>
</div>
</div>
{/* Resources */}
<div className="mt-10">
<h2 className="text-sm font-medium text-foreground mb-4">While you wait</h2>
<div className="grid grid-cols-1 md:grid-cols-3 gap-4">
<a
href="https://docs.voicebox.sh"
target="_blank"
rel="noopener noreferrer"
className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-5 hover:border-accent/30 hover:bg-card transition-all group"
>
<FileText className="h-5 w-5 text-accent mb-3" />
<h3 className="text-sm font-medium text-foreground mb-1">Read the docs</h3>
<p className="text-xs text-muted-foreground leading-relaxed">
Get familiar with Voicebox — setup, voice cloning, the REST API.
</p>
</a>
<a
href="https://deepwiki.com/jamiepine/voicebox"
target="_blank"
rel="noopener noreferrer"
className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-5 hover:border-accent/30 hover:bg-card transition-all group"
>
<Bot className="h-5 w-5 text-accent mb-3" />
<h3 className="text-sm font-medium text-foreground mb-1">Got questions? Ask AI.</h3>
<p className="text-xs text-muted-foreground leading-relaxed">
DeepWiki is an AI that knows Voicebox inside-out. Ask anything.
</p>
</a>
<a
href={GITHUB_REPO}
target="_blank"
rel="noopener noreferrer"
className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-5 hover:border-accent/30 hover:bg-card transition-all group"
>
<Github className="h-5 w-5 text-accent mb-3" />
<h3 className="text-sm font-medium text-foreground mb-1">Source on GitHub</h3>
<p className="text-xs text-muted-foreground leading-relaxed">
Star the repo, file issues, or contribute a PR.
</p>
</a>
</div>
</div>
</main>
</div>
);
}
+12 -5
View File
@@ -17,9 +17,12 @@ import {Navbar} from "@/components/Navbar";
import {AppleIcon, LinuxIcon, WindowsIcon} from "@/components/PlatformIcons"; import {AppleIcon, LinuxIcon, WindowsIcon} from "@/components/PlatformIcons";
import {TutorialsSection} from "@/components/TutorialsSection"; import {TutorialsSection} from "@/components/TutorialsSection";
import {VoiceCreator} from "@/components/VoiceCreator"; import {VoiceCreator} from "@/components/VoiceCreator";
import {GITHUB_REPO} from "@/lib/constants"; import {DOWNLOAD_LINKS, GITHUB_REPO} from "@/lib/constants";
import type {DownloadLinks} from "@/lib/releases";
export default function Home() { export default function Home() {
const [downloadLinks, setDownloadLinks] =
useState<DownloadLinks>(DOWNLOAD_LINKS);
const [version, setVersion] = useState<string | null>(null); const [version, setVersion] = useState<string | null>(null);
const [totalDownloads, setTotalDownloads] = useState<number | null>(null); const [totalDownloads, setTotalDownloads] = useState<number | null>(null);
@@ -30,6 +33,7 @@ export default function Home() {
return res.json(); return res.json();
}) })
.then((data) => { .then((data) => {
if (data.downloadLinks) setDownloadLinks(data.downloadLinks);
if (data.version) setVersion(data.version); if (data.version) setVersion(data.version);
if (data.totalDownloads != null) setTotalDownloads(data.totalDownloads); if (data.totalDownloads != null) setTotalDownloads(data.totalDownloads);
}) })
@@ -88,7 +92,7 @@ export default function Home() {
style={{animationDelay: "300ms"}} style={{animationDelay: "300ms"}}
> >
<a <a
href="/download" href="#download"
className="rounded-full bg-accent px-8 py-3.5 text-sm font-semibold uppercase tracking-wider text-white shadow-[0_4px_20px_hsl(43_60%_50%/0.3),inset_0_2px_0_rgba(255,255,255,0.2),inset_0_-2px_0_rgba(0,0,0,0.1)] transition-all hover:bg-accent-faint active:shadow-[0_2px_10px_hsl(43_60%_50%/0.3),inset_0_4px_8px_rgba(0,0,0,0.3)]" className="rounded-full bg-accent px-8 py-3.5 text-sm font-semibold uppercase tracking-wider text-white shadow-[0_4px_20px_hsl(43_60%_50%/0.3),inset_0_2px_0_rgba(255,255,255,0.2),inset_0_-2px_0_rgba(0,0,0,0.1)] transition-all hover:bg-accent-faint active:shadow-[0_2px_10px_hsl(43_60%_50%/0.3),inset_0_4px_8px_rgba(0,0,0,0.3)]"
> >
Download Download
@@ -399,7 +403,8 @@ export default function Home() {
<div className="grid grid-cols-1 sm:grid-cols-2 gap-3 max-w-2xl mx-auto"> <div className="grid grid-cols-1 sm:grid-cols-2 gap-3 max-w-2xl mx-auto">
{/* macOS ARM */} {/* macOS ARM */}
<a <a
href="/download?platform=macArm" href={downloadLinks.macArm}
download
className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group" className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group"
> >
<AppleIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" /> <AppleIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" />
@@ -413,7 +418,8 @@ export default function Home() {
{/* macOS Intel */} {/* macOS Intel */}
<a <a
href="/download?platform=macIntel" href={downloadLinks.macIntel}
download
className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group" className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group"
> >
<AppleIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" /> <AppleIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" />
@@ -425,7 +431,8 @@ export default function Home() {
{/* Windows */} {/* Windows */}
<a <a
href="/download?platform=windows" href={downloadLinks.windows}
download
className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group" className="flex items-center rounded-xl border border-border bg-card/60 backdrop-blur-sm px-5 py-4 transition-all hover:border-accent/30 hover:bg-card group"
> >
<WindowsIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" /> <WindowsIcon className="h-6 w-6 shrink-0 text-muted-foreground group-hover:text-foreground transition-colors" />
+2 -2
View File
@@ -29,8 +29,8 @@ const CURL_SNIPPET = `curl -X POST http://127.0.0.1:17493/generate \\
-H "Content-Type: application/json" \\ -H "Content-Type: application/json" \\
-d '{ -d '{
"text": "Welcome to the game, player one.", "text": "Welcome to the game, player one.",
"profile_id": "b3f1c2d4-5e6f-4a7b-8c9d-0e1f2a3b4c5d", "profile_id": "morgan-freeman",
"engine": "qwen_custom_voice", "engine": "qwen",
"instruct": "warm, slow, cinematic" "instruct": "warm, slow, cinematic"
}' \\ }' \\
--output line.wav`; --output line.wav`;
+1 -1
View File
@@ -45,7 +45,7 @@ export function Footer() {
</a> </a>
</li> </li>
<li> <li>
<a href="/download" className="hover:text-foreground transition-colors"> <a href="#download" className="hover:text-foreground transition-colors">
Download Download
</a> </a>
</li> </li>
+1 -1
View File
@@ -66,7 +66,7 @@ export function Navbar() {
API API
</a> </a>
<a <a
href="/download" href="#download"
className="rounded-md px-3 py-1.5 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground" className="rounded-md px-3 py-1.5 text-sm font-medium text-muted-foreground transition-colors hover:text-foreground"
> >
Download Download
+15 -21
View File
@@ -1,29 +1,23 @@
import { SiApple, SiLinux } from '@icons-pack/react-simple-icons';
// Official brand icons via Simple Icons (apple/linux). Simple Icons drops
// Microsoft's mark due to trademark policy, so the Windows 11 flag is
// inlined from Microsoft's public brand guidance.
export function AppleIcon({ className }: { className?: string }) { export function AppleIcon({ className }: { className?: string }) {
return <SiApple className={className} color="currentColor" />; return (
} <svg className={className} viewBox="0 0 24 24" fill="currentColor">
<path d="M17.05 20.28c-.98.95-2.05.88-3.08.4-1.09-.5-2.08-.48-3.24 0-1.44.62-2.2.44-3.06-.4C2.79 15.25 3.51 7.59 9.05 7.31c1.35.07 2.29.74 3.08.8 1.18-.24 2.31-.93 3.57-.84 1.51.12 2.65.72 3.4 1.8-3.12 1.87-2.38 5.98.48 7.13-.57 1.5-1.31 2.99-2.54 4.09l.01-.01zM12.03 7.25c-.15-2.23 1.66-4.07 3.74-4.25.29 2.58-2.34 4.5-3.74 4.25z" />
export function LinuxIcon({ className }: { className?: string }) { </svg>
return <SiLinux className={className} color="currentColor" />; );
} }
export function WindowsIcon({ className }: { className?: string }) { export function WindowsIcon({ className }: { className?: string }) {
return ( return (
<svg <svg className={className} viewBox="0 0 24 24" fill="currentColor">
className={className} <path d="M3 12V6.75l6-1.32v6.48L3 12zm17-9v8.75l-10 .15V5.21L20 3zM3 13l6 .09v7.81l-6-1.15V13zm17 .25V22l-10-1.8v-7.15l10 .15z" />
viewBox="0 0 24 24" </svg>
fill="currentColor" );
xmlns="http://www.w3.org/2000/svg" }
role="img"
aria-label="Windows" export function LinuxIcon({ className }: { className?: string }) {
> return (
<title>Windows</title> <svg className={className} viewBox="0 0 24 24" fill="currentColor">
<path d="M0 3.449L9.75 2.1v9.451H0m10.949-9.602L24 0v11.4l-13.051.149M0 12.6h9.75v9.451L0 20.699M10.949 12.6H24V24l-12.9-1.801" /> <path d="M12.504 0c-.155 0-.315.008-.48.021-4.226.333-3.105 4.807-3.17 6.298-.076 1.092-.3 1.953-1.05 3.02-.885 1.051-2.127 2.75-2.716 4.521-.278.832-.41 1.684-.287 2.489a.424.424 0 00-.11.135c-.26.26-.195.69-.133 1.001.054.27.112.553.077.784-.12.794-.3 1.593-.3 2.406 0 .599.18 1.193.3 1.791.12.599.3 1.193.3 1.792 0 .812.18 1.611.3 2.405.035.23-.023.514-.077.783-.062.312-.127.742.133 1.002a.424.424 0 00.11.135c-.123.805.01 1.657.287 2.489.589 1.771 1.831 3.47 2.716 4.521.75 1.067 0.974 1.928 1.05 3.02.065 1.491-1.056 5.965 3.17 6.298.165.013.325.021.48.021.155 0 .315-.008.48-.021 4.226-.333 3.105-4.807 3.17-6.298.076-1.092.3-1.953 1.05-3.02.885-1.051 2.127-2.75 2.716-4.521.278-.832.41-1.684.287-2.489a.424.424 0 00.11-.135c.26-.26.195-.69.133-1.001-.054-.27-.112-.553-.077-.784.12-.794.3-1.593.3-2.406 0-.599-.18-1.193-.3-1.791-.12-.599-.3-1.193-.3-1.792 0-.812-.18-1.611-.3-2.405-.035-.23.023-.514.077-.783.062-.312.127-.742-.133-1.002a.424.424 0 00-.11-.135c.123-.805-.01-1.657-.287-2.489-.589-1.771-1.831-3.47-2.716-4.521-.75-1.067-.974-1.928-1.05-3.02-.065-1.491 1.056-5.965-3.17-6.298C12.819.008 12.659 0 12.504 0z" />
</svg> </svg>
); );
} }
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "voicebox", "name": "voicebox",
"version": "0.4.2", "version": "0.4.1",
"private": true, "private": true,
"workspaces": [ "workspaces": [
"app", "app",
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"name": "@voicebox/tauri", "name": "@voicebox/tauri",
"private": true, "private": true,
"version": "0.4.2", "version": "0.4.1",
"type": "module", "type": "module",
"scripts": { "scripts": {
"dev": "vite", "dev": "vite",
+1 -1
View File
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
[[package]] [[package]]
name = "voicebox" name = "voicebox"
version = "0.4.2" version = "0.4.1"
dependencies = [ dependencies = [
"base64 0.22.1", "base64 0.22.1",
"core-foundation-sys", "core-foundation-sys",
+1 -1
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "voicebox" name = "voicebox"
version = "0.4.2" version = "0.4.1"
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation" description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
authors = ["you"] authors = ["you"]
license = "" license = ""
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"$schema": "https://schema.tauri.app/config/2", "$schema": "https://schema.tauri.app/config/2",
"productName": "Voicebox", "productName": "Voicebox",
"version": "0.4.2", "version": "0.4.1",
"identifier": "sh.voicebox.app", "identifier": "sh.voicebox.app",
"build": { "build": {
"beforeDevCommand": "bun run dev", "beforeDevCommand": "bun run dev",
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"name": "@voicebox/web", "name": "@voicebox/web",
"private": true, "private": true,
"version": "0.4.2", "version": "0.4.1",
"type": "module", "type": "module",
"scripts": { "scripts": {
"dev": "vite", "dev": "vite",