mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 15:15:27 -07:00
Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8fcc9b6695 | ||
|
|
b91d5d74cb | ||
|
|
3412dca662 | ||
|
|
ed1915223f |
@@ -26,45 +26,12 @@ jobs:
|
|||||||
args: ""
|
args: ""
|
||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
backend: "pytorch"
|
backend: "pytorch"
|
||||||
- platform: "ubuntu-22.04"
|
|
||||||
# --config override disables updater-artifact generation on Linux.
|
|
||||||
# tauri.conf.json has createUpdaterArtifacts: "v1Compatible" which
|
|
||||||
# on Linux wants to synthesize a .AppImage.tar.gz by downloading
|
|
||||||
# linuxdeploy at build time — this is what silently hangs CI
|
|
||||||
# (see v0.4.2 round 2, 25 min of no output after rpm bundling).
|
|
||||||
# We ship deb+rpm only; Linux users update via apt/dnf, not the
|
|
||||||
# Tauri in-app updater.
|
|
||||||
args: '--target x86_64-unknown-linux-gnu --bundles deb,rpm --verbose --config {"bundle":{"createUpdaterArtifacts":false}}'
|
|
||||||
python-version: "3.12"
|
|
||||||
backend: "pytorch"
|
|
||||||
|
|
||||||
runs-on: ${{ matrix.platform }}
|
runs-on: ${{ matrix.platform }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
# Ubuntu runners ship with ~14 GB free; pip + PyInstaller + torch can
|
|
||||||
# peak well above that during the build. Reclaim ~25 GB by pruning
|
|
||||||
# preinstalled toolchains we don't use. This is what likely tripped
|
|
||||||
# the March 2026 Linux release attempts (see commit 103e98b
|
|
||||||
# "github runners suck") — not a code issue, a disk-pressure one.
|
|
||||||
- name: Free up disk space (ubuntu)
|
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
|
||||||
# Pinned to v1.3.1 (SHA) — this job runs with contents: write and
|
|
||||||
# handles signing secrets later, so we don't want a floating ref.
|
|
||||||
uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be
|
|
||||||
with:
|
|
||||||
tool-cache: false
|
|
||||||
android: true
|
|
||||||
dotnet: true
|
|
||||||
haskell: true
|
|
||||||
# large-packages: true would `apt-get remove '^llvm-.*'`, which
|
|
||||||
# cascade-removes reverse deps that won't be pulled back in by the
|
|
||||||
# `llvm-dev` install below. The other flags already free ~20 GB,
|
|
||||||
# enough for the Python + torch + PyInstaller build.
|
|
||||||
large-packages: false
|
|
||||||
swap-storage: true
|
|
||||||
|
|
||||||
- name: Install dependencies (ubuntu only)
|
- name: Install dependencies (ubuntu only)
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
||||||
run: |
|
run: |
|
||||||
@@ -166,21 +133,6 @@ jobs:
|
|||||||
p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }}
|
p12-file-base64: ${{ secrets.APPLE_CERTIFICATE }}
|
||||||
p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
|
p12-password: ${{ secrets.APPLE_CERTIFICATE_PASSWORD }}
|
||||||
|
|
||||||
- name: Disk / environment snapshot (pre-bundle debug)
|
|
||||||
if: contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')
|
|
||||||
run: |
|
|
||||||
echo "=== df -h ==="
|
|
||||||
df -h
|
|
||||||
echo "=== free -h ==="
|
|
||||||
free -h
|
|
||||||
echo "=== Rust / Cargo ==="
|
|
||||||
rustc --version
|
|
||||||
cargo --version
|
|
||||||
echo "=== Bun ==="
|
|
||||||
bun --version
|
|
||||||
echo "=== Tauri CLI ==="
|
|
||||||
cd tauri && bun run tauri --version
|
|
||||||
|
|
||||||
- name: Extract release notes from CHANGELOG.md
|
- name: Extract release notes from CHANGELOG.md
|
||||||
id: changelog
|
id: changelog
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -204,13 +156,7 @@ jobs:
|
|||||||
echo "CHANGELOG_EOF"
|
echo "CHANGELOG_EOF"
|
||||||
} >> "$GITHUB_OUTPUT"
|
} >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
# Linux hang watchdog: previous releases silently wedged inside tauri
|
|
||||||
# bundling (possibly linuxdeploy/AppImage download, possibly cargo link).
|
|
||||||
# Cap the step at 30 min so we get logs instead of waiting out the 6hr
|
|
||||||
# job timeout. Other platforms historically complete in ~25 min, so 45
|
|
||||||
# is comfortable.
|
|
||||||
- uses: tauri-apps/[email protected]
|
- uses: tauri-apps/[email protected]
|
||||||
timeout-minutes: ${{ (contains(matrix.platform, 'ubuntu') || contains(matrix.platform, 'namespace')) && 30 || 45 }}
|
|
||||||
env:
|
env:
|
||||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
|
TAURI_SIGNING_PRIVATE_KEY: ${{ secrets.TAURI_SIGNING_PRIVATE_KEY }}
|
||||||
@@ -222,9 +168,6 @@ jobs:
|
|||||||
APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }}
|
APPLE_PROVIDER_SHORT_NAME: ${{ secrets.APPLE_PROVIDER_SHORT_NAME }}
|
||||||
APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }}
|
APPLE_API_ISSUER: ${{ secrets.APPLE_API_ISSUER }}
|
||||||
APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }}
|
APPLE_API_KEY: ${{ secrets.APPLE_API_KEY }}
|
||||||
# Stream subprocess stdout/stderr so the hang is visible in logs.
|
|
||||||
CARGO_TERM_VERBOSE: "true"
|
|
||||||
RUST_BACKTRACE: "1"
|
|
||||||
with:
|
with:
|
||||||
projectPath: tauri
|
projectPath: tauri
|
||||||
tagName: v__VERSION__
|
tagName: v__VERSION__
|
||||||
|
|||||||
@@ -62,14 +62,13 @@
|
|||||||
|
|
||||||
## What is Voicebox?
|
## What is Voicebox?
|
||||||
|
|
||||||
Voicebox is a **local-first voice cloning studio** — a free and open-source alternative to ElevenLabs. Clone voices from a few seconds of audio or pick from 50+ preset voices, generate speech in 23 languages across 7 TTS engines, apply post-processing effects, and compose multi-voice projects with a timeline editor.
|
Voicebox is a **local-first voice cloning studio** — a free and open-source alternative to ElevenLabs. Clone voices from a few seconds of audio, generate speech in 23 languages across 5 TTS engines, apply post-processing effects, and compose multi-voice projects with a timeline editor.
|
||||||
|
|
||||||
- **Complete privacy** — models and voice data stay on your machine
|
- **Complete privacy** — models and voice data stay on your machine
|
||||||
- **7 TTS engines** — Qwen3-TTS, Qwen CustomVoice, LuxTTS, Chatterbox Multilingual, Chatterbox Turbo, HumeAI TADA, and Kokoro
|
- **5 TTS engines** — Qwen3-TTS, LuxTTS, Chatterbox Multilingual, Chatterbox Turbo, and HumeAI TADA
|
||||||
- **Cloning and preset voices** — zero-shot cloning from a reference sample, or curated preset voices via Kokoro (50 voices) and Qwen CustomVoice (9 voices)
|
|
||||||
- **23 languages** — from English to Arabic, Japanese, Hindi, Swahili, and more
|
- **23 languages** — from English to Arabic, Japanese, Hindi, Swahili, and more
|
||||||
- **Post-processing effects** — pitch shift, reverb, delay, chorus, compression, and filters
|
- **Post-processing effects** — pitch shift, reverb, delay, chorus, compression, and filters
|
||||||
- **Expressive speech** — paralinguistic tags like `[laugh]`, `[sigh]`, `[gasp]` via Chatterbox Turbo; natural-language delivery control via Qwen CustomVoice
|
- **Expressive speech** — paralinguistic tags like `[laugh]`, `[sigh]`, `[gasp]` via Chatterbox Turbo
|
||||||
- **Unlimited length** — auto-chunking with crossfade for scripts, articles, and chapters
|
- **Unlimited length** — auto-chunking with crossfade for scripts, articles, and chapters
|
||||||
- **Stories editor** — multi-track timeline for conversations, podcasts, and narratives
|
- **Stories editor** — multi-track timeline for conversations, podcasts, and narratives
|
||||||
- **API-first** — REST API for integrating voice synthesis into your own projects
|
- **API-first** — REST API for integrating voice synthesis into your own projects
|
||||||
@@ -97,17 +96,15 @@ Voicebox is a **local-first voice cloning studio** — a free and open-source al
|
|||||||
|
|
||||||
### Multi-Engine Voice Cloning
|
### Multi-Engine Voice Cloning
|
||||||
|
|
||||||
Seven TTS engines with different strengths, switchable per-generation:
|
Five TTS engines with different strengths, switchable per-generation:
|
||||||
|
|
||||||
| Engine | Languages | Strengths |
|
| Engine | Languages | Strengths |
|
||||||
| --------------------------- | --------- | ---------------------------------------------------------------------------------------------------------------------------------------- |
|
| --------------------------- | --------- | ---------------------------------------------------------------------------------------------------------------------------------------- |
|
||||||
| **Qwen3-TTS** (0.6B / 1.7B) | 10 | High-quality multilingual cloning, delivery instructions ("speak slowly", "whisper") |
|
| **Qwen3-TTS** (0.6B / 1.7B) | 10 | High-quality multilingual cloning, delivery instructions ("speak slowly", "whisper") |
|
||||||
| **Qwen CustomVoice** | 10 | 9 curated preset voices with natural-language delivery control — no reference audio required |
|
|
||||||
| **LuxTTS** | English | Lightweight (~1GB VRAM), 48kHz output, 150x realtime on CPU |
|
| **LuxTTS** | English | Lightweight (~1GB VRAM), 48kHz output, 150x realtime on CPU |
|
||||||
| **Chatterbox Multilingual** | 23 | Broadest language coverage — Arabic, Danish, Finnish, Greek, Hebrew, Hindi, Malay, Norwegian, Polish, Swahili, Swedish, Turkish and more |
|
| **Chatterbox Multilingual** | 23 | Broadest language coverage — Arabic, Danish, Finnish, Greek, Hebrew, Hindi, Malay, Norwegian, Polish, Swahili, Swedish, Turkish and more |
|
||||||
| **Chatterbox Turbo** | English | Fast 350M model with paralinguistic emotion/sound tags |
|
| **Chatterbox Turbo** | English | Fast 350M model with paralinguistic emotion/sound tags |
|
||||||
| **TADA** (1B / 3B) | 10 | HumeAI speech-language model — 700s+ coherent audio, text-acoustic dual alignment |
|
| **TADA** (1B / 3B) | 10 | HumeAI speech-language model — 700s+ coherent audio, text-acoustic dual alignment |
|
||||||
| **Kokoro** | 8 | 50 curated preset voices, tiny 82M model, fast CPU inference |
|
|
||||||
|
|
||||||
### Emotions & Paralinguistic Tags
|
### Emotions & Paralinguistic Tags
|
||||||
|
|
||||||
@@ -242,7 +239,7 @@ Full API documentation available at `http://localhost:17493/docs`.
|
|||||||
| Frontend | React, TypeScript, Tailwind CSS |
|
| Frontend | React, TypeScript, Tailwind CSS |
|
||||||
| State | Zustand, React Query |
|
| State | Zustand, React Query |
|
||||||
| Backend | FastAPI (Python) |
|
| Backend | FastAPI (Python) |
|
||||||
| TTS Engines | Qwen3-TTS, Qwen CustomVoice, LuxTTS, Chatterbox, Chatterbox Turbo, TADA, Kokoro |
|
| TTS Engines | Qwen3-TTS, LuxTTS, Chatterbox, Chatterbox Turbo, TADA |
|
||||||
| Effects | Pedalboard (Spotify) |
|
| Effects | Pedalboard (Spotify) |
|
||||||
| Transcription | Whisper / Whisper Turbo (PyTorch or MLX) |
|
| Transcription | Whisper / Whisper Turbo (PyTorch or MLX) |
|
||||||
| Inference | MLX (Apple Silicon) / PyTorch (CUDA/ROCm/XPU/CPU) |
|
| Inference | MLX (Apple Silicon) / PyTorch (CUDA/ROCm/XPU/CPU) |
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/app",
|
"name": "@voicebox/app",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
import { useRouterState } from '@tanstack/react-router';
|
import { useRouterState } from '@tanstack/react-router';
|
||||||
import { TitleBarDragRegion } from '@/components/TitleBarDragRegion';
|
import { TitleBarDragRegion } from '@/components/TitleBarDragRegion';
|
||||||
import { AudioKeepAlive } from '@/components/AudioPlayer/AudioKeepAlive';
|
|
||||||
import { AudioPlayer } from '@/components/AudioPlayer/AudioPlayer';
|
import { AudioPlayer } from '@/components/AudioPlayer/AudioPlayer';
|
||||||
import { StoryTrackEditor } from '@/components/StoriesTab/StoryTrackEditor';
|
import { StoryTrackEditor } from '@/components/StoriesTab/StoryTrackEditor';
|
||||||
import { TOP_SAFE_AREA_PADDING } from '@/lib/constants/ui';
|
import { TOP_SAFE_AREA_PADDING } from '@/lib/constants/ui';
|
||||||
@@ -27,7 +26,6 @@ export function AppFrame({ children }: AppFrameProps) {
|
|||||||
className={cn('h-screen bg-background flex flex-col overflow-hidden', TOP_SAFE_AREA_PADDING)}
|
className={cn('h-screen bg-background flex flex-col overflow-hidden', TOP_SAFE_AREA_PADDING)}
|
||||||
>
|
>
|
||||||
<TitleBarDragRegion />
|
<TitleBarDragRegion />
|
||||||
<AudioKeepAlive />
|
|
||||||
{children}
|
{children}
|
||||||
{showTrackEditor ? (
|
{showTrackEditor ? (
|
||||||
<StoryTrackEditor storyId={story.id} items={story.items} />
|
<StoryTrackEditor storyId={story.id} items={story.items} />
|
||||||
|
|||||||
@@ -1,85 +0,0 @@
|
|||||||
import { useEffect, useRef } from 'react';
|
|
||||||
import { debug } from '@/lib/utils/debug';
|
|
||||||
|
|
||||||
// WKWebView tears down the app's CoreAudio output when idle for long enough,
|
|
||||||
// and a JS-level reload (cmd+R) does NOT restore it — only relaunching the
|
|
||||||
// Tauri app does. Keeping a silent <audio> element looping forever prevents
|
|
||||||
// the OS audio session from ever going dormant.
|
|
||||||
//
|
|
||||||
// Real silence (zero PCM samples) at full volume is preferred over a muted
|
|
||||||
// element: browsers/WebKit can optimize muted media away, which defeats the
|
|
||||||
// purpose of holding the session open.
|
|
||||||
|
|
||||||
function buildSilentWavUrl(seconds = 1, sampleRate = 8000): string {
|
|
||||||
const numSamples = seconds * sampleRate;
|
|
||||||
const bytes = 44 + numSamples * 2;
|
|
||||||
const buffer = new ArrayBuffer(bytes);
|
|
||||||
const view = new DataView(buffer);
|
|
||||||
const write = (offset: number, str: string) => {
|
|
||||||
for (let i = 0; i < str.length; i++) view.setUint8(offset + i, str.charCodeAt(i));
|
|
||||||
};
|
|
||||||
write(0, 'RIFF');
|
|
||||||
view.setUint32(4, bytes - 8, true);
|
|
||||||
write(8, 'WAVE');
|
|
||||||
write(12, 'fmt ');
|
|
||||||
view.setUint32(16, 16, true);
|
|
||||||
view.setUint16(20, 1, true);
|
|
||||||
view.setUint16(22, 1, true);
|
|
||||||
view.setUint32(24, sampleRate, true);
|
|
||||||
view.setUint32(28, sampleRate * 2, true);
|
|
||||||
view.setUint16(32, 2, true);
|
|
||||||
view.setUint16(34, 16, true);
|
|
||||||
write(36, 'data');
|
|
||||||
view.setUint32(40, numSamples * 2, true);
|
|
||||||
return URL.createObjectURL(new Blob([buffer], { type: 'audio/wav' }));
|
|
||||||
}
|
|
||||||
|
|
||||||
export function AudioKeepAlive() {
|
|
||||||
const audioRef = useRef<HTMLAudioElement | null>(null);
|
|
||||||
|
|
||||||
useEffect(() => {
|
|
||||||
const url = buildSilentWavUrl(1, 8000);
|
|
||||||
const el = new Audio(url);
|
|
||||||
el.loop = true;
|
|
||||||
el.volume = 1;
|
|
||||||
el.preload = 'auto';
|
|
||||||
audioRef.current = el;
|
|
||||||
|
|
||||||
const tryPlay = () => {
|
|
||||||
if (!audioRef.current) return;
|
|
||||||
if (!audioRef.current.paused) return;
|
|
||||||
audioRef.current.play().catch((err) => {
|
|
||||||
debug.log('[AudioKeepAlive] play blocked (will retry on next gesture):', err);
|
|
||||||
});
|
|
||||||
};
|
|
||||||
|
|
||||||
tryPlay();
|
|
||||||
|
|
||||||
// Autoplay may be blocked until first user interaction — re-attempt then.
|
|
||||||
const onGesture = () => tryPlay();
|
|
||||||
window.addEventListener('pointerdown', onGesture, { once: false });
|
|
||||||
window.addEventListener('keydown', onGesture, { once: false });
|
|
||||||
|
|
||||||
// If the webview ever pauses the element on background, resume on return.
|
|
||||||
const onWake = () => {
|
|
||||||
if (!document.hidden) tryPlay();
|
|
||||||
};
|
|
||||||
document.addEventListener('visibilitychange', onWake);
|
|
||||||
window.addEventListener('focus', onWake);
|
|
||||||
window.addEventListener('pageshow', onWake);
|
|
||||||
|
|
||||||
return () => {
|
|
||||||
window.removeEventListener('pointerdown', onGesture);
|
|
||||||
window.removeEventListener('keydown', onGesture);
|
|
||||||
document.removeEventListener('visibilitychange', onWake);
|
|
||||||
window.removeEventListener('focus', onWake);
|
|
||||||
window.removeEventListener('pageshow', onWake);
|
|
||||||
el.pause();
|
|
||||||
el.src = '';
|
|
||||||
URL.revokeObjectURL(url);
|
|
||||||
audioRef.current = null;
|
|
||||||
};
|
|
||||||
}, []);
|
|
||||||
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/landing",
|
"name": "@voicebox/landing",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"description": "Landing page for voicebox.sh",
|
"description": "Landing page for voicebox.sh",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "bun --bun next dev --turbo",
|
"dev": "bun --bun next dev --turbo",
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "voicebox",
|
"name": "voicebox",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"private": true,
|
"private": true,
|
||||||
"workspaces": [
|
"workspaces": [
|
||||||
"app",
|
"app",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/tauri",
|
"name": "@voicebox/tauri",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Generated
+1
-1
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.4.2"
|
version = "0.4.0"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"core-foundation-sys",
|
"core-foundation-sys",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.4.2"
|
version = "0.4.1"
|
||||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||||
authors = ["you"]
|
authors = ["you"]
|
||||||
license = ""
|
license = ""
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://schema.tauri.app/config/2",
|
"$schema": "https://schema.tauri.app/config/2",
|
||||||
"productName": "Voicebox",
|
"productName": "Voicebox",
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"identifier": "sh.voicebox.app",
|
"identifier": "sh.voicebox.app",
|
||||||
"build": {
|
"build": {
|
||||||
"beforeDevCommand": "bun run dev",
|
"beforeDevCommand": "bun run dev",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/web",
|
"name": "@voicebox/web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.4.2",
|
"version": "0.4.1",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Reference in New Issue
Block a user