mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 15:15:27 -07:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
38bf96ff20 | ||
|
|
0b14cb1b2c | ||
|
|
4d24e69012 | ||
|
|
90436e428d | ||
|
|
e4bb288904 | ||
|
|
6f8bc7f23b | ||
|
|
baca111d50 | ||
|
|
cc298fe6d8 | ||
|
|
46b8f6b882 | ||
|
|
162cf4fb84 | ||
|
|
68558243d9 | ||
|
|
8d5ad926f9 | ||
|
|
334f037dce | ||
|
|
f6522eea80 | ||
|
|
7615a08f81 | ||
|
|
31ea3c68a5 | ||
|
|
54d72ddfd0 | ||
|
|
d4794f78e1 | ||
|
|
aa7c9a9a8d | ||
|
|
ca6ed0998a | ||
|
|
829d4d6d5b | ||
|
|
0be7975db5 | ||
|
|
40e4af828a | ||
|
|
0e57826ea5 | ||
|
|
eb2cd861b1 | ||
|
|
701cc647a7 | ||
|
|
be6ccaf044 | ||
|
|
1040625a88 | ||
|
|
6f4503b521 | ||
|
|
f5b6edc2e7 | ||
|
|
8197f0724c | ||
|
|
d40f7d2676 | ||
|
|
99fbcca7f4 | ||
|
|
04f9880c9a | ||
|
|
b9c858295d | ||
|
|
610f64c762 | ||
|
|
220333b3bb | ||
|
|
e194e95512 | ||
|
|
e796412c2c | ||
|
|
cb541521d2 | ||
|
|
2bc243f93e |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[bumpversion]
|
[bumpversion]
|
||||||
current_version = 0.1.12
|
current_version = 0.1.13
|
||||||
commit = True
|
commit = True
|
||||||
tag = True
|
tag = True
|
||||||
tag_name = v{new_version}
|
tag_name = v{new_version}
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
name: Build Windows
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build-windows:
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
runs-on: windows-latest
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Setup Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.12"
|
||||||
|
cache: "pip"
|
||||||
|
|
||||||
|
- name: Install Python dependencies
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
pip install pyinstaller
|
||||||
|
pip install -r backend/requirements.txt
|
||||||
|
|
||||||
|
- name: Build Python server
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
cd backend
|
||||||
|
python build_binary.py
|
||||||
|
|
||||||
|
PLATFORM=$(rustc --print host-tuple)
|
||||||
|
mkdir -p ../tauri/src-tauri/binaries
|
||||||
|
cp dist/voicebox-server.exe ../tauri/src-tauri/binaries/voicebox-server-${PLATFORM}.exe
|
||||||
|
echo "Built voicebox-server-${PLATFORM}.exe"
|
||||||
|
|
||||||
|
- name: Setup Bun
|
||||||
|
uses: oven-sh/setup-bun@v2
|
||||||
|
|
||||||
|
- name: Install Rust stable
|
||||||
|
uses: dtolnay/rust-toolchain@stable
|
||||||
|
|
||||||
|
- name: Rust cache
|
||||||
|
uses: swatinem/rust-cache@v2
|
||||||
|
with:
|
||||||
|
workspaces: "./tauri/src-tauri -> target"
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: bun install
|
||||||
|
|
||||||
|
- uses: tauri-apps/tauri-action@v0
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
with:
|
||||||
|
projectPath: tauri
|
||||||
|
tagName: v__VERSION__
|
||||||
|
releaseName: "voicebox v__VERSION__ (test build)"
|
||||||
|
releaseBody: "Test build for audio export fix"
|
||||||
|
releaseDraft: true
|
||||||
|
prerelease: true
|
||||||
|
args: ""
|
||||||
|
includeUpdaterJson: false
|
||||||
@@ -53,6 +53,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
- Audio export failing when Tauri save dialog returns object instead of string path
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
- **Makefile** - Comprehensive development workflow automation with commands for setup, development, building, testing, and code quality checks
|
- **Makefile** - Comprehensive development workflow automation with commands for setup, development, building, testing, and code quality checks
|
||||||
- Includes Python version detection and compatibility warnings
|
- Includes Python version detection and compatibility warnings
|
||||||
|
|||||||
@@ -59,7 +59,7 @@
|
|||||||
|
|
||||||
## What is Voicebox?
|
## What is Voicebox?
|
||||||
|
|
||||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as the **Ollama for voice** — download models, clone voices, and generate speech entirely on your machine.
|
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as a **local, free and open-source alternative to ElevenLabs** — download models, clone voices, and generate speech entirely on your machine.
|
||||||
|
|
||||||
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives you:
|
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives you:
|
||||||
|
|
||||||
@@ -80,10 +80,10 @@ Voicebox is available now for macOS and Windows.
|
|||||||
|
|
||||||
| Platform | Download |
|
| Platform | Download |
|
||||||
|----------|----------|
|
|----------|----------|
|
||||||
| macOS (Apple Silicon) | [voicebox_aarch64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_aarch64.app.tar.gz) |
|
| macOS (Apple Silicon) | [Voicebox_aarch64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/latest/download/Voicebox_aarch64.app.tar.gz) |
|
||||||
| macOS (Intel) | [voicebox_x64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_x64.app.tar.gz) |
|
| macOS (Intel) | [Voicebox_x64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/latest/download/Voicebox_x64.app.tar.gz) |
|
||||||
| Windows (MSI) | [voicebox_0.1.0_x64_en-US.msi](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_0.1.0_x64_en-US.msi) |
|
| Windows (MSI) | [Latest Windows MSI](https://github.com/jamiepine/voicebox/releases/latest) |
|
||||||
| Windows (Setup) | [voicebox_0.1.0_x64-setup.exe](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_0.1.0_x64-setup.exe) |
|
| Windows (Setup) | [Latest Windows Setup](https://github.com/jamiepine/voicebox/releases/latest) |
|
||||||
|
|
||||||
> **Linux builds coming soon** — Currently blocked by GitHub runner disk space limitations.
|
> **Linux builds coming soon** — Currently blocked by GitHub runner disk space limitations.
|
||||||
|
|
||||||
@@ -233,7 +233,7 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed setup and contribution guide
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Clone the repo
|
# Clone the repo
|
||||||
git clone https://github.com/voicebox-sh/voicebox.git
|
git clone https://github.com/jamiepine/voicebox.git
|
||||||
cd voicebox
|
cd voicebox
|
||||||
|
|
||||||
# Setup everything
|
# Setup everything
|
||||||
@@ -247,7 +247,7 @@ make dev
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Clone the repo
|
# Clone the repo
|
||||||
git clone https://github.com/voicebox-sh/voicebox.git
|
git clone https://github.com/jamiepine/voicebox.git
|
||||||
cd voicebox
|
cd voicebox
|
||||||
|
|
||||||
# Install dependencies
|
# Install dependencies
|
||||||
@@ -260,7 +260,7 @@ cd backend && pip install -r requirements.txt && cd ..
|
|||||||
bun run dev
|
bun run dev
|
||||||
```
|
```
|
||||||
|
|
||||||
**Prerequisites:** [Bun](https://bun.sh), [Rust](https://rustup.rs), [Python 3.11+](https://python.org).
|
**Prerequisites:** [Bun](https://bun.sh), [Rust](https://rustup.rs), [Python 3.11+](https://python.org). [XCode on macOS](https://developer.apple.com/xcode/).
|
||||||
|
|
||||||
**Performance:**
|
**Performance:**
|
||||||
- **Apple Silicon (M1/M2/M3)**: Uses MLX backend with native Metal acceleration for 4-5x faster inference
|
- **Apple Silicon (M1/M2/M3)**: Uses MLX backend with native Metal acceleration for 4-5x faster inference
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/app",
|
"name": "@voicebox/app",
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -124,6 +124,13 @@ export function AudioTab() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const handleChannelDelete = async (e, channelId) => {
|
||||||
|
e.stopPropagation();
|
||||||
|
if (await confirm('Delete this channel?')) {
|
||||||
|
deleteChannel.mutate(channelId);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
const allChannels = channels || [];
|
const allChannels = channels || [];
|
||||||
const allDevices = devices || [];
|
const allDevices = devices || [];
|
||||||
const selectedChannel = selectedChannelId
|
const selectedChannel = selectedChannelId
|
||||||
@@ -241,12 +248,7 @@ export function AudioTab() {
|
|||||||
variant="ghost"
|
variant="ghost"
|
||||||
size="sm"
|
size="sm"
|
||||||
className="h-8 w-8 p-0"
|
className="h-8 w-8 p-0"
|
||||||
onClick={(e) => {
|
onClick={(e) => handleChannelDelete(e, channel.id)}
|
||||||
e.stopPropagation();
|
|
||||||
if (confirm('Delete this channel?')) {
|
|
||||||
deleteChannel.mutate(channel.id);
|
|
||||||
}
|
|
||||||
}}
|
|
||||||
>
|
>
|
||||||
<Trash2 className="h-4 w-4" />
|
<Trash2 className="h-4 w-4" />
|
||||||
</Button>
|
</Button>
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
import { useMatchRoute } from '@tanstack/react-router';
|
import { useMatchRoute } from '@tanstack/react-router';
|
||||||
import { AnimatePresence, motion } from 'framer-motion';
|
import { AnimatePresence, motion } from 'framer-motion';
|
||||||
import { Loader2, MessageSquare, Sparkles } from 'lucide-react';
|
import { Loader2, SlidersHorizontal, Sparkles } from 'lucide-react';
|
||||||
import { useEffect, useRef, useState } from 'react';
|
import { useEffect, useRef, useState } from 'react';
|
||||||
import { Button } from '@/components/ui/button';
|
import { Button } from '@/components/ui/button';
|
||||||
import { Form, FormControl, FormField, FormItem, FormMessage } from '@/components/ui/form';
|
import { Form, FormControl, FormField, FormItem, FormMessage } from '@/components/ui/form';
|
||||||
@@ -187,7 +187,7 @@ export function FloatingGenerateBox({
|
|||||||
}}
|
}}
|
||||||
>
|
>
|
||||||
<motion.div
|
<motion.div
|
||||||
className="bg-background/30 backdrop-blur-2xl border border-accent/20 rounded-[2rem] shadow-2xl hover:bg-background/40 hover:border-accent/20 transition-all duration-300 overflow-hidden p-3"
|
className="bg-background/30 backdrop-blur-2xl border border-accent/20 rounded-[2rem] shadow-2xl hover:bg-background/40 hover:border-accent/20 transition-all duration-300 p-3"
|
||||||
transition={{ duration: 0.6, ease: 'easeInOut' }}
|
transition={{ duration: 0.6, ease: 'easeInOut' }}
|
||||||
>
|
>
|
||||||
<Form {...form}>
|
<Form {...form}>
|
||||||
@@ -274,7 +274,7 @@ export function FloatingGenerateBox({
|
|||||||
field.ref(node);
|
field.ref(node);
|
||||||
}
|
}
|
||||||
}}
|
}}
|
||||||
placeholder="Add delivery instructions..."
|
placeholder="e.g. very happy and excited"
|
||||||
className="resize-none bg-transparent border-none focus-visible:ring-0 focus-visible:ring-offset-0 focus:outline-none focus:ring-0 outline-none ring-0 rounded-2xl text-sm placeholder:text-muted-foreground/60 w-full"
|
className="resize-none bg-transparent border-none focus-visible:ring-0 focus-visible:ring-offset-0 focus:outline-none focus:ring-0 outline-none ring-0 rounded-2xl text-sm placeholder:text-muted-foreground/60 w-full"
|
||||||
style={{
|
style={{
|
||||||
minHeight: isExpanded ? '100px' : '32px',
|
minHeight: isExpanded ? '100px' : '32px',
|
||||||
@@ -294,18 +294,27 @@ export function FloatingGenerateBox({
|
|||||||
</motion.div>
|
</motion.div>
|
||||||
|
|
||||||
<div className="relative shrink-0">
|
<div className="relative shrink-0">
|
||||||
<Button
|
<div className="group relative">
|
||||||
type="submit"
|
<Button
|
||||||
disabled={isPending || !selectedProfileId}
|
type="submit"
|
||||||
className="h-10 w-10 rounded-full bg-accent hover:bg-accent/90 hover:scale-105 text-accent-foreground shadow-lg hover:shadow-accent/50 transition-all duration-200"
|
disabled={isPending || !selectedProfileId}
|
||||||
size="icon"
|
className="h-10 w-10 rounded-full bg-accent hover:bg-accent/90 hover:scale-105 text-accent-foreground shadow-lg hover:shadow-accent/50 transition-all duration-200"
|
||||||
>
|
size="icon"
|
||||||
{isPending ? (
|
>
|
||||||
<Loader2 className="h-4 w-4 animate-spin" />
|
{isPending ? (
|
||||||
) : (
|
<Loader2 className="h-4 w-4 animate-spin" />
|
||||||
<Sparkles className="h-4 w-4" />
|
) : (
|
||||||
)}
|
<Sparkles className="h-4 w-4" />
|
||||||
</Button>
|
)}
|
||||||
|
</Button>
|
||||||
|
<span className="pointer-events-none absolute bottom-full left-1/2 -translate-x-1/2 mb-2 whitespace-nowrap rounded-md bg-popover px-3 py-1.5 text-xs text-popover-foreground border border-border opacity-0 transition-opacity group-hover:opacity-100 z-[9999]">
|
||||||
|
{isPending
|
||||||
|
? 'Generating...'
|
||||||
|
: !selectedProfileId
|
||||||
|
? 'Select a voice profile first'
|
||||||
|
: 'Generate speech'}
|
||||||
|
</span>
|
||||||
|
</div>
|
||||||
<AnimatePresence>
|
<AnimatePresence>
|
||||||
{isExpanded && (
|
{isExpanded && (
|
||||||
<motion.div
|
<motion.div
|
||||||
@@ -315,20 +324,25 @@ export function FloatingGenerateBox({
|
|||||||
transition={{ duration: 0.2 }}
|
transition={{ duration: 0.2 }}
|
||||||
className="absolute top-0 right-[calc(100%+0.5rem)]"
|
className="absolute top-0 right-[calc(100%+0.5rem)]"
|
||||||
>
|
>
|
||||||
<Button
|
<div className="group relative">
|
||||||
type="button"
|
<Button
|
||||||
variant="ghost"
|
type="button"
|
||||||
size="icon"
|
variant="ghost"
|
||||||
onClick={() => setIsInstructMode(!isInstructMode)}
|
size="icon"
|
||||||
className={cn(
|
onClick={() => setIsInstructMode(!isInstructMode)}
|
||||||
'h-10 w-10 rounded-full transition-all duration-200',
|
className={cn(
|
||||||
isInstructMode
|
'h-10 w-10 rounded-full transition-all duration-200',
|
||||||
? 'bg-accent text-accent-foreground border border-accent hover:bg-accent/90'
|
isInstructMode
|
||||||
: 'bg-card border border-border hover:bg-background/50',
|
? 'bg-accent text-accent-foreground border border-accent hover:bg-accent/90'
|
||||||
)}
|
: 'bg-card border border-border hover:bg-background/50',
|
||||||
>
|
)}
|
||||||
<MessageSquare className="h-4 w-4" />
|
>
|
||||||
</Button>
|
<SlidersHorizontal className="h-4 w-4" />
|
||||||
|
</Button>
|
||||||
|
<span className="pointer-events-none absolute bottom-full left-1/2 -translate-x-1/2 mb-2 whitespace-nowrap rounded-md bg-popover px-3 py-1.5 text-xs text-popover-foreground border border-border opacity-0 transition-opacity group-hover:opacity-100 z-[9999]">
|
||||||
|
Fine tune instructions
|
||||||
|
</span>
|
||||||
|
</div>
|
||||||
</motion.div>
|
</motion.div>
|
||||||
)}
|
)}
|
||||||
</AnimatePresence>
|
</AnimatePresence>
|
||||||
|
|||||||
@@ -58,6 +58,7 @@ export function AudioSampleRecording({
|
|||||||
// Request microphone access when component mounts
|
// Request microphone access when component mounts
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
if (!showWaveform) return;
|
if (!showWaveform) return;
|
||||||
|
if (!navigator.mediaDevices || !navigator.mediaDevices.getUserMedia) return;
|
||||||
|
|
||||||
let stream: MediaStream | null = null;
|
let stream: MediaStream | null = null;
|
||||||
|
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ import {
|
|||||||
} from '@/lib/hooks/useProfiles';
|
} from '@/lib/hooks/useProfiles';
|
||||||
import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
||||||
import { useTranscription } from '@/lib/hooks/useTranscription';
|
import { useTranscription } from '@/lib/hooks/useTranscription';
|
||||||
import { formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
import { convertToWav, formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
||||||
import { usePlatform } from '@/platform/PlatformContext';
|
import { usePlatform } from '@/platform/PlatformContext';
|
||||||
import { useServerStore } from '@/stores/serverStore';
|
import { useServerStore } from '@/stores/serverStore';
|
||||||
import { type ProfileFormDraft, useUIStore } from '@/stores/uiStore';
|
import { type ProfileFormDraft, useUIStore } from '@/stores/uiStore';
|
||||||
@@ -505,10 +505,23 @@ export function ProfileForm() {
|
|||||||
language: data.language,
|
language: data.language,
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// Convert non-WAV uploads to WAV so the backend can always use soundfile.
|
||||||
|
// Recorded audio is already WAV (from useAudioRecording's convertToWav call).
|
||||||
|
let fileToUpload: File = sampleFile;
|
||||||
|
if (!sampleFile.type.includes('wav') && !sampleFile.name.toLowerCase().endsWith('.wav')) {
|
||||||
|
try {
|
||||||
|
const wavBlob = await convertToWav(sampleFile);
|
||||||
|
const wavName = sampleFile.name.replace(/\.[^.]+$/, '.wav');
|
||||||
|
fileToUpload = new File([wavBlob], wavName, { type: 'audio/wav' });
|
||||||
|
} catch {
|
||||||
|
// If browser can't decode the format, send the original and let the backend try.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
try {
|
try {
|
||||||
await addSample.mutateAsync({
|
await addSample.mutateAsync({
|
||||||
profileId: profile.id,
|
profileId: profile.id,
|
||||||
file: sampleFile,
|
file: fileToUpload,
|
||||||
referenceText: referenceText,
|
referenceText: referenceText,
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -79,8 +79,8 @@ export function VoicesTab() {
|
|||||||
setDialogOpen(true);
|
setDialogOpen(true);
|
||||||
};
|
};
|
||||||
|
|
||||||
const handleDelete = (profileId: string) => {
|
const handleProfileDelete = async (profileId: string) => {
|
||||||
if (confirm('Are you sure you want to delete this profile?')) {
|
if (await confirm('Are you sure you want to delete this profile?')) {
|
||||||
deleteProfile.mutate(profileId);
|
deleteProfile.mutate(profileId);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -147,7 +147,7 @@ export function VoicesTab() {
|
|||||||
channels={channels || []}
|
channels={channels || []}
|
||||||
onChannelChange={(channelIds) => handleChannelChange(profile.id, channelIds)}
|
onChannelChange={(channelIds) => handleChannelChange(profile.id, channelIds)}
|
||||||
onEdit={() => handleEdit(profile.id)}
|
onEdit={() => handleEdit(profile.id)}
|
||||||
onDelete={() => handleDelete(profile.id)}
|
onDelete={() => handleProfileDelete(profile.id)}
|
||||||
/>
|
/>
|
||||||
))}
|
))}
|
||||||
</TableBody>
|
</TableBody>
|
||||||
|
|||||||
@@ -20,11 +20,13 @@ export function useAudioRecording({
|
|||||||
const streamRef = useRef<MediaStream | null>(null);
|
const streamRef = useRef<MediaStream | null>(null);
|
||||||
const timerRef = useRef<number | null>(null);
|
const timerRef = useRef<number | null>(null);
|
||||||
const startTimeRef = useRef<number | null>(null);
|
const startTimeRef = useRef<number | null>(null);
|
||||||
|
const cancelledRef = useRef<boolean>(false);
|
||||||
|
|
||||||
const startRecording = useCallback(async () => {
|
const startRecording = useCallback(async () => {
|
||||||
try {
|
try {
|
||||||
setError(null);
|
setError(null);
|
||||||
chunksRef.current = [];
|
chunksRef.current = [];
|
||||||
|
cancelledRef.current = false;
|
||||||
setDuration(0);
|
setDuration(0);
|
||||||
|
|
||||||
// Check if getUserMedia is available
|
// Check if getUserMedia is available
|
||||||
@@ -87,31 +89,34 @@ export function useAudioRecording({
|
|||||||
};
|
};
|
||||||
|
|
||||||
mediaRecorder.onstop = async () => {
|
mediaRecorder.onstop = async () => {
|
||||||
|
// Snapshot the cancellation flag and recorded duration immediately —
|
||||||
|
// cancelRecording() clears chunks and sets cancelledRef synchronously
|
||||||
|
// before this async handler runs, so we must check it first.
|
||||||
|
const wasCancelled = cancelledRef.current;
|
||||||
|
const recordedDuration = startTimeRef.current
|
||||||
|
? (Date.now() - startTimeRef.current) / 1000
|
||||||
|
: undefined;
|
||||||
|
|
||||||
const webmBlob = new Blob(chunksRef.current, { type: 'audio/webm' });
|
const webmBlob = new Blob(chunksRef.current, { type: 'audio/webm' });
|
||||||
|
|
||||||
// Convert to WAV format to avoid needing ffmpeg on backend
|
// Stop all tracks now that we have the data
|
||||||
try {
|
|
||||||
const wavBlob = await convertToWav(webmBlob);
|
|
||||||
|
|
||||||
// Pass the actual recorded duration
|
|
||||||
const recordedDuration = startTimeRef.current
|
|
||||||
? (Date.now() - startTimeRef.current) / 1000
|
|
||||||
: undefined;
|
|
||||||
onRecordingComplete?.(wavBlob, recordedDuration);
|
|
||||||
} catch (err) {
|
|
||||||
console.error('Error converting audio to WAV:', err);
|
|
||||||
// Fallback to original blob if conversion fails
|
|
||||||
const recordedDuration = startTimeRef.current
|
|
||||||
? (Date.now() - startTimeRef.current) / 1000
|
|
||||||
: undefined;
|
|
||||||
onRecordingComplete?.(webmBlob, recordedDuration);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Stop all tracks
|
|
||||||
streamRef.current?.getTracks().forEach((track) => {
|
streamRef.current?.getTracks().forEach((track) => {
|
||||||
track.stop();
|
track.stop();
|
||||||
});
|
});
|
||||||
streamRef.current = null;
|
streamRef.current = null;
|
||||||
|
|
||||||
|
// Don't fire completion callback if the recording was cancelled
|
||||||
|
if (wasCancelled) return;
|
||||||
|
|
||||||
|
// Convert to WAV format to avoid needing ffmpeg on backend
|
||||||
|
try {
|
||||||
|
const wavBlob = await convertToWav(webmBlob);
|
||||||
|
onRecordingComplete?.(wavBlob, recordedDuration);
|
||||||
|
} catch (err) {
|
||||||
|
console.error('Error converting audio to WAV:', err);
|
||||||
|
// Fallback to original blob if conversion fails
|
||||||
|
onRecordingComplete?.(webmBlob, recordedDuration);
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
mediaRecorder.onerror = (event) => {
|
mediaRecorder.onerror = (event) => {
|
||||||
@@ -167,9 +172,10 @@ export function useAudioRecording({
|
|||||||
|
|
||||||
const cancelRecording = useCallback(() => {
|
const cancelRecording = useCallback(() => {
|
||||||
if (mediaRecorderRef.current) {
|
if (mediaRecorderRef.current) {
|
||||||
|
cancelledRef.current = true; // Must be set before stop() triggers onstop
|
||||||
|
chunksRef.current = [];
|
||||||
mediaRecorderRef.current.stop();
|
mediaRecorderRef.current.stop();
|
||||||
setIsRecording(false);
|
setIsRecording(false);
|
||||||
chunksRef.current = [];
|
|
||||||
setDuration(0);
|
setDuration(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+36
-18
@@ -22,6 +22,11 @@ export function formatAudioDuration(seconds: number): string {
|
|||||||
* If the file has a recordedDuration property (from recording hooks),
|
* If the file has a recordedDuration property (from recording hooks),
|
||||||
* use that instead of trying to read metadata. This fixes issues on Windows
|
* use that instead of trying to read metadata. This fixes issues on Windows
|
||||||
* where WebM files from MediaRecorder don't have proper duration metadata.
|
* where WebM files from MediaRecorder don't have proper duration metadata.
|
||||||
|
*
|
||||||
|
* For uploaded files we use AudioContext.decodeAudioData which fully decodes
|
||||||
|
* the audio and returns the exact duration. This is more reliable than
|
||||||
|
* HTMLMediaElement.duration which can return incorrect large values for VBR
|
||||||
|
* MP3 files that lack a proper XING/VBRI header.
|
||||||
*/
|
*/
|
||||||
export async function getAudioDuration(
|
export async function getAudioDuration(
|
||||||
file: File & { recordedDuration?: number },
|
file: File & { recordedDuration?: number },
|
||||||
@@ -30,26 +35,39 @@ export async function getAudioDuration(
|
|||||||
return file.recordedDuration;
|
return file.recordedDuration;
|
||||||
}
|
}
|
||||||
|
|
||||||
return new Promise((resolve, reject) => {
|
// Use Web Audio API for accurate duration — avoids VBR MP3 metadata issues.
|
||||||
const audio = new Audio();
|
try {
|
||||||
const url = URL.createObjectURL(file);
|
const audioContext = new AudioContext();
|
||||||
|
try {
|
||||||
|
const arrayBuffer = await file.arrayBuffer();
|
||||||
|
const audioBuffer = await audioContext.decodeAudioData(arrayBuffer);
|
||||||
|
return audioBuffer.duration;
|
||||||
|
} finally {
|
||||||
|
await audioContext.close();
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
// Fallback: read duration from the media element (less accurate but works for WAV).
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
const audio = new Audio();
|
||||||
|
const url = URL.createObjectURL(file);
|
||||||
|
|
||||||
audio.addEventListener('loadedmetadata', () => {
|
audio.addEventListener('loadedmetadata', () => {
|
||||||
URL.revokeObjectURL(url);
|
URL.revokeObjectURL(url);
|
||||||
if (Number.isFinite(audio.duration) && audio.duration > 0) {
|
if (Number.isFinite(audio.duration) && audio.duration > 0) {
|
||||||
resolve(audio.duration);
|
resolve(audio.duration);
|
||||||
} else {
|
} else {
|
||||||
reject(new Error('Audio file has invalid duration metadata'));
|
reject(new Error('Audio file has invalid duration metadata'));
|
||||||
}
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
audio.addEventListener('error', () => {
|
||||||
|
URL.revokeObjectURL(url);
|
||||||
|
reject(new Error('Failed to load audio file'));
|
||||||
|
});
|
||||||
|
|
||||||
|
audio.src = url;
|
||||||
});
|
});
|
||||||
|
}
|
||||||
audio.addEventListener('error', () => {
|
|
||||||
URL.revokeObjectURL(url);
|
|
||||||
reject(new Error('Failed to load audio file'));
|
|
||||||
});
|
|
||||||
|
|
||||||
audio.src = url;
|
|
||||||
});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
+1
-1
@@ -1,3 +1,3 @@
|
|||||||
# Backend package
|
# Backend package
|
||||||
|
|
||||||
__version__ = "0.1.12"
|
__version__ = "0.1.13"
|
||||||
|
|||||||
@@ -29,9 +29,23 @@ class PyTorchTTSBackend:
|
|||||||
"""Get the best available device."""
|
"""Get the best available device."""
|
||||||
if torch.cuda.is_available():
|
if torch.cuda.is_available():
|
||||||
return "cuda"
|
return "cuda"
|
||||||
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
# Intel Arc / Intel Xe GPU via intel-extension-for-pytorch (IPEX)
|
||||||
# MPS can have issues, use CPU for stability
|
try:
|
||||||
return "cpu"
|
import intel_extension_for_pytorch # noqa: F401
|
||||||
|
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||||
|
return "xpu"
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
# Any GPU on Windows via DirectML (torch-directml)
|
||||||
|
try:
|
||||||
|
import torch_directml
|
||||||
|
if torch_directml.device_count() > 0:
|
||||||
|
return torch_directml.device(0)
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
# MPS (Apple Silicon) — kept for completeness but MLX backend is preferred
|
||||||
|
if hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||||
|
return "cpu" # MPS disabled for stability; MLX backend handles Apple Silicon
|
||||||
return "cpu"
|
return "cpu"
|
||||||
|
|
||||||
def is_loaded(self) -> bool:
|
def is_loaded(self) -> bool:
|
||||||
@@ -166,11 +180,21 @@ class PyTorchTTSBackend:
|
|||||||
|
|
||||||
# Load the model (tqdm is patched, but filters out non-download progress)
|
# Load the model (tqdm is patched, but filters out non-download progress)
|
||||||
try:
|
try:
|
||||||
self.model = Qwen3TTSModel.from_pretrained(
|
# Don't pass device_map on CPU: accelerate's meta-tensor mechanism
|
||||||
model_path,
|
# causes "Cannot copy out of meta tensor" when moving to CPU.
|
||||||
device_map=self.device,
|
# Instead load directly then call .to(device) if needed.
|
||||||
torch_dtype=torch.float32 if self.device == "cpu" else torch.bfloat16,
|
if self.device == "cpu":
|
||||||
)
|
self.model = Qwen3TTSModel.from_pretrained(
|
||||||
|
model_path,
|
||||||
|
torch_dtype=torch.float32,
|
||||||
|
low_cpu_mem_usage=False,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
self.model = Qwen3TTSModel.from_pretrained(
|
||||||
|
model_path,
|
||||||
|
device_map=self.device,
|
||||||
|
torch_dtype=torch.bfloat16,
|
||||||
|
)
|
||||||
finally:
|
finally:
|
||||||
# Exit the patch context
|
# Exit the patch context
|
||||||
tracker_context.__exit__(None, None, None)
|
tracker_context.__exit__(None, None, None)
|
||||||
@@ -358,9 +382,22 @@ class PyTorchSTTBackend:
|
|||||||
"""Get the best available device."""
|
"""Get the best available device."""
|
||||||
if torch.cuda.is_available():
|
if torch.cuda.is_available():
|
||||||
return "cuda"
|
return "cuda"
|
||||||
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
# Intel Arc / Intel Xe GPU via intel-extension-for-pytorch (IPEX)
|
||||||
# MPS support for Whisper
|
try:
|
||||||
return "cpu" # Use CPU for stability
|
import intel_extension_for_pytorch # noqa: F401
|
||||||
|
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||||
|
return "xpu"
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
# Any GPU on Windows via DirectML (torch-directml)
|
||||||
|
try:
|
||||||
|
import torch_directml
|
||||||
|
if torch_directml.device_count() > 0:
|
||||||
|
return torch_directml.device(0)
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
if hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||||
|
return "cpu" # MPS disabled for stability
|
||||||
return "cpu"
|
return "cpu"
|
||||||
|
|
||||||
def is_loaded(self) -> bool:
|
def is_loaded(self) -> bool:
|
||||||
|
|||||||
@@ -83,9 +83,13 @@ def build_server():
|
|||||||
'--hidden-import', 'mlx_audio.stt',
|
'--hidden-import', 'mlx_audio.stt',
|
||||||
'--collect-submodules', 'mlx',
|
'--collect-submodules', 'mlx',
|
||||||
'--collect-submodules', 'mlx_audio',
|
'--collect-submodules', 'mlx_audio',
|
||||||
# Collect MLX data files including Metal shader libraries (.metallib)
|
# Use --collect-all so PyInstaller bundles both data files AND
|
||||||
'--collect-data', 'mlx',
|
# native shared libraries (.dylib, .metallib) for MLX.
|
||||||
'--collect-data', 'mlx_audio',
|
# Previously only --collect-data was used, which caused MLX to
|
||||||
|
# raise OSError at runtime inside the bundled binary because
|
||||||
|
# the Metal shader libraries were missing.
|
||||||
|
'--collect-all', 'mlx',
|
||||||
|
'--collect-all', 'mlx_audio',
|
||||||
])
|
])
|
||||||
else:
|
else:
|
||||||
print("Building for non-Apple Silicon platform - PyTorch only")
|
print("Building for non-Apple Silicon platform - PyTorch only")
|
||||||
|
|||||||
@@ -4,8 +4,17 @@ Configuration module for voicebox backend.
|
|||||||
Handles data directory configuration for production bundling.
|
Handles data directory configuration for production bundling.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
# Allow users to override the HuggingFace model download directory.
|
||||||
|
# Set VOICEBOX_MODELS_DIR to an absolute path before starting the server.
|
||||||
|
# This sets HF_HUB_CACHE so all huggingface_hub downloads go to that path.
|
||||||
|
_custom_models_dir = os.environ.get("VOICEBOX_MODELS_DIR")
|
||||||
|
if _custom_models_dir:
|
||||||
|
os.environ["HF_HUB_CACHE"] = _custom_models_dir
|
||||||
|
print(f"[config] Model download path set to: {_custom_models_dir}")
|
||||||
|
|
||||||
# Default data directory (used in development)
|
# Default data directory (used in development)
|
||||||
_data_dir = Path("data")
|
_data_dir = Path("data")
|
||||||
|
|
||||||
|
|||||||
+153
-39
@@ -22,6 +22,24 @@ import uuid
|
|||||||
import asyncio
|
import asyncio
|
||||||
import signal
|
import signal
|
||||||
import os
|
import os
|
||||||
|
from urllib.parse import quote
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_content_disposition(disposition_type: str, filename: str) -> str:
|
||||||
|
"""Build a Content-Disposition header that is safe for non-ASCII filenames.
|
||||||
|
|
||||||
|
Uses RFC 5987 ``filename*`` parameter so that browsers can decode
|
||||||
|
UTF-8 filenames while the ``filename`` fallback stays ASCII-only.
|
||||||
|
"""
|
||||||
|
ascii_name = "".join(
|
||||||
|
c for c in filename if c.isascii() and (c.isalnum() or c in " -_.")
|
||||||
|
).strip() or "download"
|
||||||
|
utf8_name = quote(filename, safe="")
|
||||||
|
return (
|
||||||
|
f'{disposition_type}; filename="{ascii_name}"; '
|
||||||
|
f"filename*=UTF-8''{utf8_name}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
from . import database, models, profiles, history, tts, transcribe, config, export_import, channels, stories, __version__
|
from . import database, models, profiles, history, tts, transcribe, config, export_import, channels, stories, __version__
|
||||||
from .database import get_db, Generation as DBGeneration, VoiceProfile as DBVoiceProfile
|
from .database import get_db, Generation as DBGeneration, VoiceProfile as DBVoiceProfile
|
||||||
@@ -77,10 +95,39 @@ async def health():
|
|||||||
tts_model = tts.get_tts_model()
|
tts_model = tts.get_tts_model()
|
||||||
backend_type = get_backend_type()
|
backend_type = get_backend_type()
|
||||||
|
|
||||||
# Check for GPU availability (CUDA or MPS)
|
# Check for GPU availability (CUDA, MPS, Intel Arc XPU, or DirectML)
|
||||||
has_cuda = torch.cuda.is_available()
|
has_cuda = torch.cuda.is_available()
|
||||||
has_mps = hasattr(torch.backends, 'mps') and torch.backends.mps.is_available()
|
has_mps = hasattr(torch.backends, 'mps') and torch.backends.mps.is_available()
|
||||||
gpu_available = has_cuda or has_mps
|
|
||||||
|
# Intel Arc / Intel Xe via intel-extension-for-pytorch (IPEX)
|
||||||
|
has_xpu = False
|
||||||
|
xpu_name = None
|
||||||
|
try:
|
||||||
|
import intel_extension_for_pytorch as ipex # noqa: F401
|
||||||
|
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||||
|
has_xpu = True
|
||||||
|
try:
|
||||||
|
xpu_name = torch.xpu.get_device_name(0)
|
||||||
|
except Exception:
|
||||||
|
xpu_name = "Intel GPU"
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# DirectML backend (torch-directml) for any Windows GPU
|
||||||
|
has_directml = False
|
||||||
|
directml_name = None
|
||||||
|
try:
|
||||||
|
import torch_directml
|
||||||
|
if torch_directml.device_count() > 0:
|
||||||
|
has_directml = True
|
||||||
|
try:
|
||||||
|
directml_name = torch_directml.device_name(0)
|
||||||
|
except Exception:
|
||||||
|
directml_name = "DirectML GPU"
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
gpu_available = has_cuda or has_mps or has_xpu or has_directml or backend_type == "mlx"
|
||||||
|
|
||||||
gpu_type = None
|
gpu_type = None
|
||||||
if has_cuda:
|
if has_cuda:
|
||||||
@@ -89,6 +136,10 @@ async def health():
|
|||||||
gpu_type = "MPS (Apple Silicon)"
|
gpu_type = "MPS (Apple Silicon)"
|
||||||
elif backend_type == "mlx":
|
elif backend_type == "mlx":
|
||||||
gpu_type = "Metal (Apple Silicon via MLX)"
|
gpu_type = "Metal (Apple Silicon via MLX)"
|
||||||
|
elif has_xpu:
|
||||||
|
gpu_type = f"XPU ({xpu_name})"
|
||||||
|
elif has_directml:
|
||||||
|
gpu_type = f"DirectML ({directml_name})"
|
||||||
|
|
||||||
vram_used = None
|
vram_used = None
|
||||||
if has_cuda:
|
if has_cuda:
|
||||||
@@ -252,12 +303,17 @@ async def add_profile_sample(
|
|||||||
db: Session = Depends(get_db),
|
db: Session = Depends(get_db),
|
||||||
):
|
):
|
||||||
"""Add a sample to a voice profile."""
|
"""Add a sample to a voice profile."""
|
||||||
# Save uploaded file to temporary location
|
# Preserve the uploaded file's extension so librosa can detect format correctly.
|
||||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
# Defaulting to .wav was causing soundfile to reject MP3/WebM content as invalid WAV.
|
||||||
|
_allowed_audio_exts = {'.wav', '.mp3', '.m4a', '.ogg', '.flac', '.aac', '.webm', '.opus'}
|
||||||
|
_uploaded_ext = Path(file.filename or '').suffix.lower()
|
||||||
|
file_suffix = _uploaded_ext if _uploaded_ext in _allowed_audio_exts else '.wav'
|
||||||
|
|
||||||
|
with tempfile.NamedTemporaryFile(suffix=file_suffix, delete=False) as tmp:
|
||||||
content = await file.read()
|
content = await file.read()
|
||||||
tmp.write(content)
|
tmp.write(content)
|
||||||
tmp_path = tmp.name
|
tmp_path = tmp.name
|
||||||
|
|
||||||
try:
|
try:
|
||||||
sample = await profiles.add_profile_sample(
|
sample = await profiles.add_profile_sample(
|
||||||
profile_id,
|
profile_id,
|
||||||
@@ -268,6 +324,8 @@ async def add_profile_sample(
|
|||||||
return sample
|
return sample
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
raise HTTPException(status_code=400, detail=str(e))
|
raise HTTPException(status_code=400, detail=str(e))
|
||||||
|
except Exception as e:
|
||||||
|
raise HTTPException(status_code=500, detail=f"Failed to process audio file: {str(e)}")
|
||||||
finally:
|
finally:
|
||||||
# Clean up temp file
|
# Clean up temp file
|
||||||
Path(tmp_path).unlink(missing_ok=True)
|
Path(tmp_path).unlink(missing_ok=True)
|
||||||
@@ -388,7 +446,7 @@ async def export_profile(
|
|||||||
io.BytesIO(zip_bytes),
|
io.BytesIO(zip_bytes),
|
||||||
media_type="application/zip",
|
media_type="application/zip",
|
||||||
headers={
|
headers={
|
||||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
@@ -542,47 +600,50 @@ async def generate_speech(
|
|||||||
if not profile:
|
if not profile:
|
||||||
raise HTTPException(status_code=404, detail="Profile not found")
|
raise HTTPException(status_code=404, detail="Profile not found")
|
||||||
|
|
||||||
# Create voice prompt from profile
|
|
||||||
voice_prompt = await profiles.create_voice_prompt_for_profile(
|
|
||||||
data.profile_id,
|
|
||||||
db,
|
|
||||||
)
|
|
||||||
|
|
||||||
# Generate audio
|
# Generate audio
|
||||||
|
|
||||||
|
# Resolve model size and load the correct model FIRST.
|
||||||
|
# This must happen before create_voice_prompt_for_profile because that
|
||||||
|
# function calls load_model_async(None), which falls back to self.model_size.
|
||||||
|
# If the model is already loaded with the right size at that point, it
|
||||||
|
# returns immediately and the voice prompt is created by the correct model.
|
||||||
tts_model = tts.get_tts_model()
|
tts_model = tts.get_tts_model()
|
||||||
# Load the requested model size if different from current (async to not block)
|
|
||||||
model_size = data.model_size or "1.7B"
|
model_size = data.model_size or "1.7B"
|
||||||
|
|
||||||
# Check if model needs to be downloaded first
|
# Check if model needs to be downloaded first
|
||||||
model_path = tts_model._get_model_path(model_size)
|
model_path = tts_model._get_model_path(model_size)
|
||||||
if model_path.startswith("Qwen/"):
|
if not tts_model._is_model_cached(model_size):
|
||||||
# Model not cached - check if it exists remotely or needs download
|
# Model is not fully cached — kick off a background download and tell
|
||||||
from huggingface_hub import constants as hf_constants
|
# the client to retry once it's ready.
|
||||||
repo_cache = Path(hf_constants.HF_HUB_CACHE) / ("models--" + model_path.replace("/", "--"))
|
model_name = f"qwen-tts-{model_size}"
|
||||||
if not repo_cache.exists():
|
|
||||||
# Start download in background
|
|
||||||
model_name = f"qwen-tts-{model_size}"
|
|
||||||
|
|
||||||
async def download_model_background():
|
async def download_model_background():
|
||||||
try:
|
try:
|
||||||
await tts_model.load_model_async(model_size)
|
await tts_model.load_model_async(model_size)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
task_manager.error_download(model_name, str(e))
|
task_manager.error_download(model_name, str(e))
|
||||||
|
|
||||||
task_manager.start_download(model_name)
|
task_manager.start_download(model_name)
|
||||||
asyncio.create_task(download_model_background())
|
asyncio.create_task(download_model_background())
|
||||||
|
|
||||||
# Return 202 Accepted with download info
|
raise HTTPException(
|
||||||
raise HTTPException(
|
status_code=202,
|
||||||
status_code=202,
|
detail={
|
||||||
detail={
|
"message": f"Model {model_size} is being downloaded. Please wait and try again.",
|
||||||
"message": f"Model {model_size} is being downloaded. Please wait and try again.",
|
"model_name": model_name,
|
||||||
"model_name": model_name,
|
"downloading": True,
|
||||||
"downloading": True
|
},
|
||||||
}
|
)
|
||||||
)
|
|
||||||
|
|
||||||
|
# Load (or switch to) the requested model before building the voice prompt
|
||||||
await tts_model.load_model_async(model_size)
|
await tts_model.load_model_async(model_size)
|
||||||
|
|
||||||
|
# Create voice prompt from profile (model is already loaded with correct size)
|
||||||
|
voice_prompt = await profiles.create_voice_prompt_for_profile(
|
||||||
|
data.profile_id,
|
||||||
|
db,
|
||||||
|
)
|
||||||
|
|
||||||
audio, sample_rate = await tts_model.generate(
|
audio, sample_rate = await tts_model.generate(
|
||||||
data.text,
|
data.text,
|
||||||
voice_prompt,
|
voice_prompt,
|
||||||
@@ -625,6 +686,59 @@ async def generate_speech(
|
|||||||
raise HTTPException(status_code=500, detail=str(e))
|
raise HTTPException(status_code=500, detail=str(e))
|
||||||
|
|
||||||
|
|
||||||
|
@app.post("/generate/stream")
|
||||||
|
async def stream_speech(
|
||||||
|
data: models.GenerationRequest,
|
||||||
|
db: Session = Depends(get_db),
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Generate speech and stream the WAV audio directly without saving to disk.
|
||||||
|
|
||||||
|
Returns raw WAV bytes via a StreamingResponse so the client can start
|
||||||
|
playing audio before the entire file has been received. This endpoint
|
||||||
|
does NOT create a history entry — use /generate for that.
|
||||||
|
"""
|
||||||
|
profile = await profiles.get_profile(data.profile_id, db)
|
||||||
|
if not profile:
|
||||||
|
raise HTTPException(status_code=404, detail="Profile not found")
|
||||||
|
|
||||||
|
tts_model = tts.get_tts_model()
|
||||||
|
model_size = data.model_size or "1.7B"
|
||||||
|
|
||||||
|
if not tts_model._is_model_cached(model_size):
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=400,
|
||||||
|
detail=f"Model {model_size} is not downloaded yet. Use /generate to trigger a download.",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Load the correct model before building the voice prompt (fixes issue #96)
|
||||||
|
await tts_model.load_model_async(model_size)
|
||||||
|
|
||||||
|
voice_prompt = await profiles.create_voice_prompt_for_profile(data.profile_id, db)
|
||||||
|
|
||||||
|
audio, sample_rate = await tts_model.generate(
|
||||||
|
data.text,
|
||||||
|
voice_prompt,
|
||||||
|
data.language,
|
||||||
|
data.seed,
|
||||||
|
data.instruct,
|
||||||
|
)
|
||||||
|
|
||||||
|
wav_bytes = tts.audio_to_wav_bytes(audio, sample_rate)
|
||||||
|
|
||||||
|
async def _wav_stream():
|
||||||
|
# Yield in chunks so large responses don't block the event loop
|
||||||
|
chunk_size = 64 * 1024 # 64 KB
|
||||||
|
for i in range(0, len(wav_bytes), chunk_size):
|
||||||
|
yield wav_bytes[i : i + chunk_size]
|
||||||
|
|
||||||
|
return StreamingResponse(
|
||||||
|
_wav_stream(),
|
||||||
|
media_type="audio/wav",
|
||||||
|
headers={"Content-Disposition": 'attachment; filename="speech.wav"'},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
# ============================================
|
# ============================================
|
||||||
# HISTORY ENDPOINTS
|
# HISTORY ENDPOINTS
|
||||||
# ============================================
|
# ============================================
|
||||||
@@ -753,7 +867,7 @@ async def export_generation(
|
|||||||
io.BytesIO(zip_bytes),
|
io.BytesIO(zip_bytes),
|
||||||
media_type="application/zip",
|
media_type="application/zip",
|
||||||
headers={
|
headers={
|
||||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
@@ -786,7 +900,7 @@ async def export_generation_audio(
|
|||||||
audio_path,
|
audio_path,
|
||||||
media_type="audio/wav",
|
media_type="audio/wav",
|
||||||
headers={
|
headers={
|
||||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -1054,7 +1168,7 @@ async def export_story_audio(
|
|||||||
io.BytesIO(audio_bytes),
|
io.BytesIO(audio_bytes),
|
||||||
media_type="audio/wav",
|
media_type="audio/wav",
|
||||||
headers={
|
headers={
|
||||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
except HTTPException:
|
except HTTPException:
|
||||||
|
|||||||
@@ -19,15 +19,17 @@ def is_apple_silicon() -> bool:
|
|||||||
def get_backend_type() -> Literal["mlx", "pytorch"]:
|
def get_backend_type() -> Literal["mlx", "pytorch"]:
|
||||||
"""
|
"""
|
||||||
Detect the best backend for the current platform.
|
Detect the best backend for the current platform.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
"mlx" on Apple Silicon (if MLX is available), "pytorch" otherwise
|
"mlx" on Apple Silicon (if MLX is available and functional), "pytorch" otherwise
|
||||||
"""
|
"""
|
||||||
if is_apple_silicon():
|
if is_apple_silicon():
|
||||||
try:
|
try:
|
||||||
import mlx
|
import mlx.core # noqa: F401 — triggers native lib loading
|
||||||
return "mlx"
|
return "mlx"
|
||||||
except ImportError:
|
except (ImportError, OSError, RuntimeError):
|
||||||
# MLX not installed, fallback to PyTorch
|
# MLX not installed, or native libraries failed to load inside a
|
||||||
|
# PyInstaller bundle (OSError on missing .dylib / .metallib).
|
||||||
|
# Fall through to PyTorch.
|
||||||
return "pytorch"
|
return "pytorch"
|
||||||
return "pytorch"
|
return "pytorch"
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ qwen-tts>=0.0.5
|
|||||||
librosa>=0.10.0
|
librosa>=0.10.0
|
||||||
soundfile>=0.12.0
|
soundfile>=0.12.0
|
||||||
numpy>=1.24.0
|
numpy>=1.24.0
|
||||||
|
numba>=0.60.0,<0.61.0
|
||||||
|
|
||||||
# Utilities
|
# Utilities
|
||||||
python-multipart>=0.0.6
|
python-multipart>=0.0.6
|
||||||
|
|||||||
@@ -32,11 +32,3 @@ def audio_to_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes:
|
|||||||
sf.write(buffer, audio, sample_rate, format="WAV")
|
sf.write(buffer, audio, sample_rate, format="WAV")
|
||||||
buffer.seek(0)
|
buffer.seek(0)
|
||||||
return buffer.read()
|
return buffer.read()
|
||||||
|
|
||||||
|
|
||||||
def audio_to_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes:
|
|
||||||
"""Convert audio array to WAV bytes."""
|
|
||||||
buffer = io.BytesIO()
|
|
||||||
sf.write(buffer, audio, sample_rate, format="WAV")
|
|
||||||
buffer.seek(0)
|
|
||||||
return buffer.read()
|
|
||||||
|
|||||||
@@ -6,8 +6,13 @@ from PyInstaller.utils.hooks import copy_metadata
|
|||||||
datas = []
|
datas = []
|
||||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
||||||
datas += collect_data_files('qwen_tts')
|
datas += collect_data_files('qwen_tts')
|
||||||
datas += collect_data_files('mlx')
|
# Use collect_all (not collect_data_files) so native .dylib and .metallib
|
||||||
datas += collect_data_files('mlx_audio')
|
# files are bundled as binaries, not data. Without this, MLX raises OSError
|
||||||
|
# when loading Metal shaders inside the PyInstaller bundle.
|
||||||
|
from PyInstaller.utils.hooks import collect_all as _collect_all
|
||||||
|
_mlx_datas, _mlx_bins, _mlx_hidden = _collect_all('mlx')
|
||||||
|
_mlxa_datas, _mlxa_bins, _mlxa_hidden = _collect_all('mlx_audio')
|
||||||
|
datas += _mlx_datas + _mlxa_datas
|
||||||
datas += copy_metadata('qwen-tts')
|
datas += copy_metadata('qwen-tts')
|
||||||
hiddenimports += collect_submodules('qwen_tts')
|
hiddenimports += collect_submodules('qwen_tts')
|
||||||
hiddenimports += collect_submodules('jaraco')
|
hiddenimports += collect_submodules('jaraco')
|
||||||
@@ -18,7 +23,7 @@ hiddenimports += collect_submodules('mlx_audio')
|
|||||||
a = Analysis(
|
a = Analysis(
|
||||||
['server.py'],
|
['server.py'],
|
||||||
pathex=[],
|
pathex=[],
|
||||||
binaries=[],
|
binaries=_mlx_bins + _mlxa_bins,
|
||||||
datas=datas,
|
datas=datas,
|
||||||
hiddenimports=hiddenimports,
|
hiddenimports=hiddenimports,
|
||||||
hookspath=[],
|
hookspath=[],
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ description: "Welcome to Voicebox - the open-source voice synthesis studio"
|
|||||||
|
|
||||||
## What is Voicebox?
|
## What is Voicebox?
|
||||||
|
|
||||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as the **Ollama for voice** — download models, clone voices, and generate speech entirely on your machine.
|
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as a **local, free and open-source alternative to ElevenLabs** — download models, clone voices, and generate speech entirely on your machine.
|
||||||
|
|
||||||
<Frame>
|
<Frame>
|
||||||
<img src="/images/app-screenshot-1.webp" alt="Voicebox App Screenshot" />
|
<img src="/images/app-screenshot-1.webp" alt="Voicebox App Screenshot" />
|
||||||
|
|||||||
@@ -0,0 +1,964 @@
|
|||||||
|
# TTS Provider Architecture
|
||||||
|
|
||||||
|
**Status:** Planned for v0.1.13
|
||||||
|
**Created:** 2025-01-31
|
||||||
|
**Problem:** GitHub 2GB release limit + poor UX for frequent updates requiring 2.4GB re-downloads
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Overview
|
||||||
|
|
||||||
|
Split the monolithic backend into modular components:
|
||||||
|
|
||||||
|
1. **Main App** (~150-200MB): Tauri + FastAPI backend + Whisper + UI/profiles/history
|
||||||
|
2. **TTS Providers** (downloadable plugins): Separate executables for model inference
|
||||||
|
|
||||||
|
This architecture solves:
|
||||||
|
|
||||||
|
- ✅ GitHub 2GB release artifact limit
|
||||||
|
- ✅ Frequent app updates without re-downloading large python binaries
|
||||||
|
- ✅ User choice of compute backend (CPU/GPU/Cloud)
|
||||||
|
- ✅ External provider support (OpenAI, custom servers)
|
||||||
|
- ✅ Future extensibility
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Architecture Diagram
|
||||||
|
|
||||||
|
```
|
||||||
|
┌─────────────────────────────────────────────────────────┐
|
||||||
|
│ Voicebox App (Tauri + Backend) ~150MB │
|
||||||
|
│ ├─ UI Layer (React) │
|
||||||
|
│ ├─ Backend (FastAPI) │
|
||||||
|
│ │ ├─ Voice Profiles │
|
||||||
|
│ │ ├─ Generation History │
|
||||||
|
│ │ ├─ Audio Editing / Stories │
|
||||||
|
│ │ └─ Provider Manager ◄──────────────┐ │
|
||||||
|
│ └─ Whisper (bundled, tiny ~50MB) │ │
|
||||||
|
└─────────────────────────────────────────┼────────────────┘
|
||||||
|
│
|
||||||
|
HTTP/IPC │
|
||||||
|
│
|
||||||
|
┌────────────────────────────────┼─────────────────┐
|
||||||
|
│ │ │
|
||||||
|
▼ ▼ ▼
|
||||||
|
┌─────────────────┐ ┌─────────────────┐ ┌──────────────────┐
|
||||||
|
│ TTS Provider: │ │ TTS Provider: │ │ TTS Provider: │
|
||||||
|
│ PyTorch CPU │ │ PyTorch CUDA │ │ MLX (Apple) │
|
||||||
|
│ │ │ │ │ │
|
||||||
|
│ ~300MB │ │ ~2.4GB │ │ ~800MB │
|
||||||
|
│ │ │ │ │ │
|
||||||
|
│ Local inference │ │ GPU inference │ │ Metal inference │
|
||||||
|
└─────────────────┘ └─────────────────┘ └──────────────────┘
|
||||||
|
│ │ │
|
||||||
|
└────────────────────────┴─────────────────────┘
|
||||||
|
│
|
||||||
|
┌─────────────▼──────────────┐
|
||||||
|
│ Future Providers: │
|
||||||
|
│ • Remote Server │
|
||||||
|
│ • OpenAI API │
|
||||||
|
│ • ElevenLabs │
|
||||||
|
│ • Custom Docker Container │
|
||||||
|
└────────────────────────────┘
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Problem Statement
|
||||||
|
|
||||||
|
### Current Architecture Issues
|
||||||
|
|
||||||
|
**Monolithic Binary:**
|
||||||
|
|
||||||
|
- CPU version: ~295MB
|
||||||
|
- CUDA version: ~2.37GB
|
||||||
|
- GitHub releases: 2GB file size limit (BLOCKED)
|
||||||
|
- Updates require re-downloading entire binary
|
||||||
|
- Poor UX: update app → restart → download CUDA update → restart again
|
||||||
|
|
||||||
|
**User Pain Points:**
|
||||||
|
|
||||||
|
1. Cannot release CUDA version on GitHub (over 2GB)
|
||||||
|
2. Every app update forces 2.4GB re-download for GPU users
|
||||||
|
3. No flexibility (can't use OpenAI, remote servers, etc.)
|
||||||
|
4. Wastes bandwidth for small bug fixes
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Solution: Pluggable TTS Providers
|
||||||
|
|
||||||
|
### Component Breakdown
|
||||||
|
|
||||||
|
#### 1. Main App (voicebox.exe / .app / .AppImage)
|
||||||
|
|
||||||
|
**Size:** ~100-150MB
|
||||||
|
|
||||||
|
**Includes:**
|
||||||
|
|
||||||
|
- Tauri runtime + React UI
|
||||||
|
- FastAPI backend (pure Python, no PyTorch)
|
||||||
|
- Whisper model (tiny, ~50MB)
|
||||||
|
- SQLite database
|
||||||
|
- Profile/history/audio editing logic
|
||||||
|
- Provider management system
|
||||||
|
|
||||||
|
**Does NOT include:**
|
||||||
|
|
||||||
|
- PyTorch (CPU or CUDA)
|
||||||
|
- TTS models (Qwen3-TTS)
|
||||||
|
- Heavy ML dependencies
|
||||||
|
|
||||||
|
**Updates frequently:** UI fixes, feature additions, non-ML changes
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
#### 2. TTS Provider: PyTorch CPU
|
||||||
|
|
||||||
|
**Binary:** `tts-provider-pytorch-cpu.exe`
|
||||||
|
**Size:** ~200MB
|
||||||
|
|
||||||
|
**Includes:**
|
||||||
|
|
||||||
|
- PyTorch CPU build
|
||||||
|
- Qwen3-TTS package
|
||||||
|
- Transformers
|
||||||
|
- No CUDA libraries
|
||||||
|
|
||||||
|
**Download source:** Cloudflare R2
|
||||||
|
**Updates rarely:** Only when model code changes
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
#### 3. TTS Provider: PyTorch CUDA
|
||||||
|
|
||||||
|
**Binary:** `tts-provider-pytorch-cuda.exe`
|
||||||
|
**Size:** ~2.4GB
|
||||||
|
|
||||||
|
**Includes:**
|
||||||
|
|
||||||
|
- PyTorch CUDA build (cu121)
|
||||||
|
- Qwen3-TTS package
|
||||||
|
- CUDA runtime, cuDNN, cuBLAS
|
||||||
|
- Transformers
|
||||||
|
|
||||||
|
**Download source:** Cloudflare R2
|
||||||
|
**Platform:** Windows + Linux (NVIDIA GPU)
|
||||||
|
**Updates rarely:** Only when model code or CUDA version changes
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
#### 4. TTS Provider: MLX
|
||||||
|
|
||||||
|
**Binary:** `tts-provider-mlx`
|
||||||
|
**Size:** ~150MB
|
||||||
|
|
||||||
|
**Includes:**
|
||||||
|
|
||||||
|
- MLX framework
|
||||||
|
- MLX-optimized Qwen3-TTS
|
||||||
|
- Metal acceleration
|
||||||
|
|
||||||
|
**Platform:** macOS only (Apple Silicon)
|
||||||
|
**Download source:** Cloudflare R2
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
#### 5. TTS Provider: Remote
|
||||||
|
|
||||||
|
**Binary:** None (built-in config)
|
||||||
|
**Size:** 0MB
|
||||||
|
|
||||||
|
**How it works:**
|
||||||
|
|
||||||
|
- User provides URL to their own TTS server
|
||||||
|
- Backend proxies requests to that server
|
||||||
|
- Implements API spec from `EXTERNAL_PROVIDERS.md`
|
||||||
|
|
||||||
|
**Use cases:**
|
||||||
|
|
||||||
|
- AMD GPU users running their own server
|
||||||
|
- Team deployments with shared GPU server
|
||||||
|
- Cloud hosting (Modal, RunPod, Replicate)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
#### 6. TTS Provider: OpenAI
|
||||||
|
|
||||||
|
**Binary:** None (API wrapper)
|
||||||
|
**Size:** 0MB
|
||||||
|
|
||||||
|
**How it works:**
|
||||||
|
|
||||||
|
- User provides OpenAI API key
|
||||||
|
- Backend wraps OpenAI Audio API
|
||||||
|
- Voice profiles map to OpenAI voices
|
||||||
|
|
||||||
|
**Benefits:**
|
||||||
|
|
||||||
|
- Zero local compute
|
||||||
|
- Pay-per-use
|
||||||
|
- Instant setup
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Communication Protocol
|
||||||
|
|
||||||
|
### Provider API Specification
|
||||||
|
|
||||||
|
All TTS providers must implement these endpoints:
|
||||||
|
|
||||||
|
#### POST /tts/generate
|
||||||
|
|
||||||
|
Generate speech from text.
|
||||||
|
|
||||||
|
**Request:**
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"text": "Hello world!",
|
||||||
|
"voice_prompt": {
|
||||||
|
/* voice prompt object */
|
||||||
|
},
|
||||||
|
"language": "en",
|
||||||
|
"seed": 12345,
|
||||||
|
"model_size": "1.7B"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Response:**
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"audio": "base64-encoded-audio",
|
||||||
|
"sample_rate": 24000,
|
||||||
|
"duration": 2.5
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### POST /tts/create_voice_prompt
|
||||||
|
|
||||||
|
Create voice prompt from reference audio.
|
||||||
|
|
||||||
|
**Request:** (multipart/form-data)
|
||||||
|
|
||||||
|
- `audio`: Audio file
|
||||||
|
- `reference_text`: Transcript
|
||||||
|
|
||||||
|
**Response:**
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"voice_prompt": {
|
||||||
|
/* serialized prompt */
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### GET /tts/health
|
||||||
|
|
||||||
|
Health check.
|
||||||
|
|
||||||
|
**Response:**
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"status": "healthy",
|
||||||
|
"provider": "pytorch-cuda",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||||
|
"device": "cuda:0"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### GET /tts/status
|
||||||
|
|
||||||
|
Model status.
|
||||||
|
|
||||||
|
**Response:**
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"model_loaded": true,
|
||||||
|
"model_size": "1.7B",
|
||||||
|
"available_sizes": ["0.6B", "1.7B"],
|
||||||
|
"gpu_available": true,
|
||||||
|
"vram_used_mb": 1234
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Backend Implementation
|
||||||
|
|
||||||
|
### Provider Manager
|
||||||
|
|
||||||
|
**File:** `backend/providers/__init__.py`
|
||||||
|
|
||||||
|
```python
|
||||||
|
class ProviderManager:
|
||||||
|
"""Manages TTS provider lifecycle."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.active_provider: Optional[Provider] = None
|
||||||
|
self.config = load_provider_config()
|
||||||
|
|
||||||
|
async def start_provider(self, provider_type: str) -> str:
|
||||||
|
"""Start a TTS provider process."""
|
||||||
|
if provider_type == "pytorch-cpu":
|
||||||
|
return await self._start_local_provider("tts-provider-pytorch-cpu.exe")
|
||||||
|
elif provider_type == "pytorch-cuda":
|
||||||
|
return await self._start_local_provider("tts-provider-pytorch-cuda.exe")
|
||||||
|
elif provider_type == "mlx":
|
||||||
|
return await self._start_local_provider("tts-provider-mlx")
|
||||||
|
elif provider_type == "remote":
|
||||||
|
return self.config["remote_url"]
|
||||||
|
elif provider_type == "openai":
|
||||||
|
return None # No subprocess, API wrapper
|
||||||
|
|
||||||
|
async def _start_local_provider(self, binary_name: str) -> str:
|
||||||
|
"""Start local provider subprocess."""
|
||||||
|
provider_path = get_provider_binary_path(binary_name)
|
||||||
|
|
||||||
|
if not provider_path.exists():
|
||||||
|
raise ProviderNotInstalledException(binary_name)
|
||||||
|
|
||||||
|
# Start subprocess on random port
|
||||||
|
port = get_free_port()
|
||||||
|
process = subprocess.Popen([
|
||||||
|
str(provider_path),
|
||||||
|
"--port", str(port),
|
||||||
|
"--data-dir", str(config.get_data_dir())
|
||||||
|
])
|
||||||
|
|
||||||
|
# Wait for provider to be ready
|
||||||
|
await wait_for_provider_health(f"http://localhost:{port}")
|
||||||
|
|
||||||
|
self.active_provider = Provider(process, port)
|
||||||
|
return f"http://localhost:{port}"
|
||||||
|
|
||||||
|
async def stop_provider(self):
|
||||||
|
"""Stop active provider."""
|
||||||
|
if self.active_provider:
|
||||||
|
self.active_provider.process.terminate()
|
||||||
|
self.active_provider = None
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Provider Abstraction
|
||||||
|
|
||||||
|
**File:** `backend/providers/base.py`
|
||||||
|
|
||||||
|
```python
|
||||||
|
class TTSProvider(ABC):
|
||||||
|
"""Abstract base for TTS providers."""
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
async def generate(
|
||||||
|
self,
|
||||||
|
text: str,
|
||||||
|
voice_prompt: dict,
|
||||||
|
language: str,
|
||||||
|
seed: Optional[int]
|
||||||
|
) -> tuple[np.ndarray, int]:
|
||||||
|
"""Generate speech audio."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
async def create_voice_prompt(
|
||||||
|
self,
|
||||||
|
audio_path: str,
|
||||||
|
reference_text: str
|
||||||
|
) -> dict:
|
||||||
|
"""Create voice prompt from reference audio."""
|
||||||
|
pass
|
||||||
|
```
|
||||||
|
|
||||||
|
**File:** `backend/providers/local.py`
|
||||||
|
|
||||||
|
```python
|
||||||
|
class LocalProvider(TTSProvider):
|
||||||
|
"""Provider that communicates with local subprocess via HTTP."""
|
||||||
|
|
||||||
|
def __init__(self, base_url: str):
|
||||||
|
self.base_url = base_url
|
||||||
|
self.client = httpx.AsyncClient()
|
||||||
|
|
||||||
|
async def generate(self, text, voice_prompt, language, seed):
|
||||||
|
response = await self.client.post(
|
||||||
|
f"{self.base_url}/tts/generate",
|
||||||
|
json={
|
||||||
|
"text": text,
|
||||||
|
"voice_prompt": voice_prompt,
|
||||||
|
"language": language,
|
||||||
|
"seed": seed
|
||||||
|
}
|
||||||
|
)
|
||||||
|
data = response.json()
|
||||||
|
audio = np.frombuffer(base64.b64decode(data["audio"]), dtype=np.float32)
|
||||||
|
return audio, data["sample_rate"]
|
||||||
|
```
|
||||||
|
|
||||||
|
**File:** `backend/providers/openai.py`
|
||||||
|
|
||||||
|
```python
|
||||||
|
class OpenAIProvider(TTSProvider):
|
||||||
|
"""Provider that wraps OpenAI Audio API."""
|
||||||
|
|
||||||
|
def __init__(self, api_key: str):
|
||||||
|
self.client = OpenAI(api_key=api_key)
|
||||||
|
|
||||||
|
async def generate(self, text, voice_prompt, language, seed):
|
||||||
|
# Map voice_prompt to OpenAI voice name
|
||||||
|
voice = map_profile_to_openai_voice(voice_prompt)
|
||||||
|
|
||||||
|
response = await self.client.audio.speech.create(
|
||||||
|
model="tts-1",
|
||||||
|
voice=voice,
|
||||||
|
input=text
|
||||||
|
)
|
||||||
|
|
||||||
|
# Convert to numpy array
|
||||||
|
audio_data = response.content
|
||||||
|
audio, sr = load_audio_from_bytes(audio_data)
|
||||||
|
return audio, sr
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Provider Installation
|
||||||
|
|
||||||
|
### Download Manager
|
||||||
|
|
||||||
|
**File:** `backend/providers/installer.py`
|
||||||
|
|
||||||
|
```python
|
||||||
|
class ProviderInstaller:
|
||||||
|
"""Handles provider download and installation."""
|
||||||
|
|
||||||
|
async def download_provider(self, provider_type: str):
|
||||||
|
"""Download provider binary from R2."""
|
||||||
|
|
||||||
|
binary_name = {
|
||||||
|
"pytorch-cpu": "tts-provider-pytorch-cpu.exe",
|
||||||
|
"pytorch-cuda": "tts-provider-pytorch-cuda.exe",
|
||||||
|
"mlx": "tts-provider-mlx"
|
||||||
|
}[provider_type]
|
||||||
|
|
||||||
|
download_url = f"https://downloads.voicebox.sh/providers/v{PROVIDER_VERSION}/{binary_name}"
|
||||||
|
|
||||||
|
# Download with progress tracking (reuse existing SSE system)
|
||||||
|
await download_with_progress(
|
||||||
|
url=download_url,
|
||||||
|
destination=get_provider_install_path(binary_name),
|
||||||
|
progress_key=f"provider-{provider_type}"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Provider Storage Location:**
|
||||||
|
|
||||||
|
- Windows: `%APPDATA%/voicebox/providers/`
|
||||||
|
- macOS: `~/Library/Application Support/voicebox/providers/`
|
||||||
|
- Linux: `~/.local/share/voicebox/providers/`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Frontend Implementation
|
||||||
|
|
||||||
|
### Provider Settings UI
|
||||||
|
|
||||||
|
**Component:** `app/src/components/ServerSettings/ProviderSettings.tsx`
|
||||||
|
|
||||||
|
```tsx
|
||||||
|
export function ProviderSettings() {
|
||||||
|
const [selectedProvider, setSelectedProvider] =
|
||||||
|
useState<ProviderType>("auto");
|
||||||
|
const {data: installedProviders} = useQuery({
|
||||||
|
queryKey: ["providers", "installed"],
|
||||||
|
queryFn: () => apiClient.getInstalledProviders(),
|
||||||
|
});
|
||||||
|
|
||||||
|
return (
|
||||||
|
<Card>
|
||||||
|
<CardHeader>
|
||||||
|
<CardTitle>TTS Provider</CardTitle>
|
||||||
|
<CardDescription>Choose how Voicebox generates speech</CardDescription>
|
||||||
|
</CardHeader>
|
||||||
|
<CardContent>
|
||||||
|
<RadioGroup
|
||||||
|
value={selectedProvider}
|
||||||
|
onValueChange={setSelectedProvider}
|
||||||
|
>
|
||||||
|
{/* Auto-detect */}
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem value="auto" id="auto" />
|
||||||
|
<Label htmlFor="auto">
|
||||||
|
<div className="font-medium">Auto-detect (Recommended)</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
Automatically choose the best available provider
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
{/* PyTorch CUDA */}
|
||||||
|
<div className="flex items-center justify-between">
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem
|
||||||
|
value="pytorch-cuda"
|
||||||
|
id="cuda"
|
||||||
|
disabled={!gpuAvailable}
|
||||||
|
/>
|
||||||
|
<Label htmlFor="cuda">
|
||||||
|
<div className="font-medium">PyTorch CUDA (NVIDIA GPU)</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
4-5x faster inference on NVIDIA GPUs
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
{!installedProviders?.includes("pytorch-cuda") && gpuAvailable && (
|
||||||
|
<Button
|
||||||
|
onClick={() => downloadProvider("pytorch-cuda")}
|
||||||
|
size="sm"
|
||||||
|
>
|
||||||
|
Download (2.4GB)
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
|
||||||
|
{/* PyTorch CPU */}
|
||||||
|
<div className="flex items-center justify-between">
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem value="pytorch-cpu" id="cpu" />
|
||||||
|
<Label htmlFor="cpu">
|
||||||
|
<div className="font-medium">PyTorch CPU</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
Works on any system, slower inference
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
{!installedProviders?.includes("pytorch-cpu") && (
|
||||||
|
<Button onClick={() => downloadProvider("pytorch-cpu")} size="sm">
|
||||||
|
Download (300MB)
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
|
||||||
|
{/* MLX (macOS only) */}
|
||||||
|
{isMacOS && (
|
||||||
|
<div className="flex items-center justify-between">
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem value="mlx" id="mlx" />
|
||||||
|
<Label htmlFor="mlx">
|
||||||
|
<div className="font-medium">MLX (Apple Silicon)</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
Optimized for M1/M2/M3 chips
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
{!installedProviders?.includes("mlx") && (
|
||||||
|
<Button onClick={() => downloadProvider("mlx")} size="sm">
|
||||||
|
Download (800MB)
|
||||||
|
</Button>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
|
{/* Remote */}
|
||||||
|
<div className="space-y-2">
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem value="remote" id="remote" />
|
||||||
|
<Label htmlFor="remote">
|
||||||
|
<div className="font-medium">Remote Server</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
Connect to your own TTS server
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
{selectedProvider === "remote" && (
|
||||||
|
<Input placeholder="http://your-server:8000" className="ml-6" />
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
|
||||||
|
{/* OpenAI */}
|
||||||
|
<div className="space-y-2">
|
||||||
|
<div className="flex items-center space-x-2">
|
||||||
|
<RadioGroupItem value="openai" id="openai" />
|
||||||
|
<Label htmlFor="openai">
|
||||||
|
<div className="font-medium">OpenAI API</div>
|
||||||
|
<div className="text-sm text-muted-foreground">
|
||||||
|
Use OpenAI's TTS API (requires API key)
|
||||||
|
</div>
|
||||||
|
</Label>
|
||||||
|
</div>
|
||||||
|
{selectedProvider === "openai" && (
|
||||||
|
<Input type="password" placeholder="sk-..." className="ml-6" />
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
</RadioGroup>
|
||||||
|
</CardContent>
|
||||||
|
</Card>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## File Structure
|
||||||
|
|
||||||
|
```
|
||||||
|
voicebox/
|
||||||
|
├── backend/
|
||||||
|
│ ├── main.py # Main FastAPI app (no TTS code)
|
||||||
|
│ ├── providers/
|
||||||
|
│ │ ├── __init__.py # ProviderManager
|
||||||
|
│ │ ├── base.py # TTSProvider ABC
|
||||||
|
│ │ ├── local.py # LocalProvider (subprocess)
|
||||||
|
│ │ ├── remote.py # RemoteProvider (HTTP)
|
||||||
|
│ │ ├── openai.py # OpenAIProvider (API wrapper)
|
||||||
|
│ │ └── installer.py # Provider download logic
|
||||||
|
│ ├── profiles.py # Voice profile management
|
||||||
|
│ ├── history.py # Generation history
|
||||||
|
│ ├── transcribe.py # Whisper (still bundled)
|
||||||
|
│ └── ... (other backend modules)
|
||||||
|
│
|
||||||
|
├── providers/
|
||||||
|
│ ├── pytorch-cpu/
|
||||||
|
│ │ ├── main.py # FastAPI server for TTS
|
||||||
|
│ │ ├── tts_backend.py # PyTorch TTS logic
|
||||||
|
│ │ ├── requirements.txt # torch (CPU), qwen-tts, transformers
|
||||||
|
│ │ └── build.spec # PyInstaller spec
|
||||||
|
│ │
|
||||||
|
│ ├── pytorch-cuda/
|
||||||
|
│ │ ├── main.py # FastAPI server for TTS
|
||||||
|
│ │ ├── tts_backend.py # PyTorch TTS logic
|
||||||
|
│ │ ├── requirements.txt # torch+cu121, qwen-tts, transformers
|
||||||
|
│ │ └── build.spec # PyInstaller spec
|
||||||
|
│ │
|
||||||
|
│ └── mlx/
|
||||||
|
│ ├── main.py # FastAPI server for TTS
|
||||||
|
│ ├── mlx_backend.py # MLX TTS logic
|
||||||
|
│ ├── requirements.txt # mlx, qwen-tts-mlx
|
||||||
|
│ └── build.spec # PyInstaller spec
|
||||||
|
│
|
||||||
|
├── app/ # Frontend (Tauri + React)
|
||||||
|
│ └── src/
|
||||||
|
│ └── components/
|
||||||
|
│ └── ServerSettings/
|
||||||
|
│ └── ProviderSettings.tsx
|
||||||
|
│
|
||||||
|
└── tauri/
|
||||||
|
└── src-tauri/
|
||||||
|
└── tauri.conf.json # No externalBin for providers
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Migration Path
|
||||||
|
|
||||||
|
### Phase 1: Refactor Backend (No User Changes)
|
||||||
|
|
||||||
|
**Goal:** Abstract TTS behind provider interface
|
||||||
|
|
||||||
|
1. Create `backend/providers/` module structure
|
||||||
|
2. Implement `TTSProvider` abstract base class
|
||||||
|
3. Create `LocalProvider` wrapper for current PyTorch code
|
||||||
|
4. Modify `backend/tts.py` to use provider abstraction
|
||||||
|
5. Keep PyTorch bundled in main app
|
||||||
|
|
||||||
|
**Result:** Code is prepared, but user experience unchanged
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Phase 2: Build Provider Binaries
|
||||||
|
|
||||||
|
**Goal:** Create standalone TTS provider executables
|
||||||
|
|
||||||
|
1. Create separate PyInstaller specs for each provider
|
||||||
|
2. Build provider executables:
|
||||||
|
- `tts-provider-pytorch-cpu.exe` (~300MB)
|
||||||
|
- `tts-provider-pytorch-cuda.exe` (~2.4GB)
|
||||||
|
- `tts-provider-mlx` (~800MB, macOS)
|
||||||
|
3. Test subprocess communication
|
||||||
|
4. Upload providers to Cloudflare R2
|
||||||
|
|
||||||
|
**Result:** Provider binaries exist but aren't used yet
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Phase 3: Remove PyTorch from Main App
|
||||||
|
|
||||||
|
**Goal:** Split main app from providers
|
||||||
|
|
||||||
|
1. Exclude PyTorch/Qwen3-TTS from main app PyInstaller spec
|
||||||
|
2. Main app now requires provider download
|
||||||
|
3. Update GitHub CI to build multiple artifacts:
|
||||||
|
- `voicebox-{version}-{platform}.exe` (~150MB)
|
||||||
|
- `tts-provider-pytorch-cpu-{version}.exe`
|
||||||
|
- `tts-provider-pytorch-cuda-{version}.exe`
|
||||||
|
- `tts-provider-mlx-{version}` (macOS)
|
||||||
|
|
||||||
|
**Result:** Main app is small, providers downloaded separately
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Phase 4: Add Provider UI
|
||||||
|
|
||||||
|
**Goal:** User-facing provider management
|
||||||
|
|
||||||
|
1. Create Provider Settings page
|
||||||
|
2. Implement provider download UI
|
||||||
|
3. Add provider status indicators
|
||||||
|
4. Show active provider in UI
|
||||||
|
|
||||||
|
**Result:** Users can choose and download providers
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Phase 5: External Providers
|
||||||
|
|
||||||
|
**Goal:** Enable remote and cloud providers
|
||||||
|
|
||||||
|
1. Implement `RemoteProvider` (HTTP client)
|
||||||
|
2. Implement `OpenAIProvider` (API wrapper)
|
||||||
|
3. Add provider configuration UI (URLs, API keys)
|
||||||
|
4. Document external provider API spec
|
||||||
|
|
||||||
|
**Result:** Full provider ecosystem
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Provider Versioning
|
||||||
|
|
||||||
|
### Independent Versioning
|
||||||
|
|
||||||
|
Providers have their own version numbers, independent of the main app:
|
||||||
|
|
||||||
|
- **App version:** `v0.2.0` (frequent updates)
|
||||||
|
- **Provider version:** `v1.0.0` (rare updates)
|
||||||
|
|
||||||
|
### Compatibility Matrix
|
||||||
|
|
||||||
|
**Example:**
|
||||||
|
|
||||||
|
| App Version | Min Provider Version | Max Provider Version |
|
||||||
|
| ----------- | -------------------- | -------------------- |
|
||||||
|
| v0.2.0 | v1.0.0 | v1.x.x |
|
||||||
|
| v0.3.0 | v1.0.0 | v1.x.x |
|
||||||
|
| v0.4.0 | v1.2.0 | v1.x.x |
|
||||||
|
| v1.0.0 | v2.0.0 | v2.x.x |
|
||||||
|
|
||||||
|
**Backend checks compatibility:**
|
||||||
|
|
||||||
|
```python
|
||||||
|
async def check_provider_compatibility(provider_version: str) -> bool:
|
||||||
|
"""Check if provider version is compatible with current app."""
|
||||||
|
min_version = "1.0.0"
|
||||||
|
max_version = "1.999.999"
|
||||||
|
return min_version <= provider_version < max_version
|
||||||
|
```
|
||||||
|
|
||||||
|
**UI shows warning if incompatible:**
|
||||||
|
|
||||||
|
```
|
||||||
|
⚠️ Provider version 0.9.0 is outdated. Update to v1.0.0+
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## User Flows
|
||||||
|
|
||||||
|
### First-Time Setup
|
||||||
|
|
||||||
|
1. User downloads and installs Voicebox (~150MB)
|
||||||
|
2. App launches → detects no TTS provider installed
|
||||||
|
3. Shows setup wizard:
|
||||||
|
|
||||||
|
```
|
||||||
|
Choose your TTS provider:
|
||||||
|
|
||||||
|
[ ] PyTorch CUDA (2.4GB) [Download]
|
||||||
|
✓ Fastest on NVIDIA GPUs
|
||||||
|
✗ Requires NVIDIA GPU
|
||||||
|
|
||||||
|
[●] PyTorch CPU (300MB) [Download]
|
||||||
|
✓ Works on any system
|
||||||
|
✗ Slower inference
|
||||||
|
|
||||||
|
[ ] MLX (800MB) [Download]
|
||||||
|
✓ Fast on Apple Silicon
|
||||||
|
✗ macOS only (M1/M2/M3)
|
||||||
|
|
||||||
|
[ ] Remote Server
|
||||||
|
URL: ___________________
|
||||||
|
|
||||||
|
[ ] OpenAI API
|
||||||
|
API Key: ________________
|
||||||
|
```
|
||||||
|
|
||||||
|
4. User selects provider → downloads with progress bar
|
||||||
|
5. Provider installs to AppData/Application Support
|
||||||
|
6. App starts provider → ready to use
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### App Update Flow (No Provider Change)
|
||||||
|
|
||||||
|
**Scenario:** Bug fix in UI, no backend changes
|
||||||
|
|
||||||
|
1. User gets update notification: "Voicebox v0.2.1 available"
|
||||||
|
2. Downloads update (~150MB, not 2.4GB!)
|
||||||
|
3. Installs and restarts
|
||||||
|
4. **Provider stays the same** (no re-download needed)
|
||||||
|
5. App starts using existing provider
|
||||||
|
|
||||||
|
**User experience:** Fast updates, no multi-GB downloads
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Provider Update Flow
|
||||||
|
|
||||||
|
**Scenario:** New Qwen3-TTS model version released
|
||||||
|
|
||||||
|
1. User opens Settings → Provider tab
|
||||||
|
2. Sees notification: "Provider update available (v1.1.0)"
|
||||||
|
3. Clicks "Update Provider"
|
||||||
|
4. Downloads new provider binary
|
||||||
|
5. Old provider binary is replaced
|
||||||
|
6. Restart app to use new provider
|
||||||
|
|
||||||
|
**Frequency:** Rare (only when TTS model/backend changes)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Switching Providers
|
||||||
|
|
||||||
|
**Scenario:** User upgrades to NVIDIA GPU
|
||||||
|
|
||||||
|
1. User goes to Settings → Provider
|
||||||
|
2. Selects "PyTorch CUDA"
|
||||||
|
3. Clicks "Download" → downloads 2.4GB
|
||||||
|
4. Download completes → restarts app
|
||||||
|
5. App now uses CUDA provider
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Benefits
|
||||||
|
|
||||||
|
| Benefit | Details |
|
||||||
|
| ----------------------------- | --------------------------------------------------------- |
|
||||||
|
| **GitHub Releases Work** | Main app ~150MB << 2GB limit |
|
||||||
|
| **Fast Updates** | UI/feature updates don't require re-downloading providers |
|
||||||
|
| **User Choice** | CPU, CUDA, MLX, OpenAI, remote server |
|
||||||
|
| **External Provider Support** | Users can run their own TTS servers |
|
||||||
|
| **Bandwidth Savings** | Only download provider once, app updates are small |
|
||||||
|
| **Future-Proof** | Easy to add new providers (ElevenLabs, custom models) |
|
||||||
|
| **Team Deployments** | Multiple users share one remote provider |
|
||||||
|
| **Cloud-Ready** | Works with Modal, Replicate, RunPod, etc. |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
### 1. Provider Versioning
|
||||||
|
|
||||||
|
**Question:** Should providers have independent versions or match app version?
|
||||||
|
|
||||||
|
**Options:**
|
||||||
|
|
||||||
|
- A. Independent (providers: v1.x, app: v0.2.x)
|
||||||
|
- B. Matched (both use v0.2.x)
|
||||||
|
|
||||||
|
**Recommendation:** Independent versioning with compatibility matrix
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 2. Auto-Update Providers
|
||||||
|
|
||||||
|
**Question:** Should providers auto-update separately from app?
|
||||||
|
|
||||||
|
**Options:**
|
||||||
|
|
||||||
|
- A. Manual updates only (user clicks "Update Provider")
|
||||||
|
- B. Optional auto-update (user can enable)
|
||||||
|
- C. Always auto-update
|
||||||
|
|
||||||
|
**Recommendation:** Optional auto-update (default off)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 3. Provider Discovery
|
||||||
|
|
||||||
|
**Question:** How does app find installed providers?
|
||||||
|
|
||||||
|
**Options:**
|
||||||
|
|
||||||
|
- A. Check standard paths in AppData/Application Support
|
||||||
|
- B. Registry (Windows) / plist (macOS)
|
||||||
|
- C. Config file with provider locations
|
||||||
|
|
||||||
|
**Recommendation:** Standard paths + config fallback
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 4. Fallback Behavior
|
||||||
|
|
||||||
|
**Question:** What if no provider is installed?
|
||||||
|
|
||||||
|
**Options:**
|
||||||
|
|
||||||
|
- A. Show setup wizard on first launch
|
||||||
|
- B. Block app until provider installed
|
||||||
|
- C. Allow app to run in "demo mode" (transcription only)
|
||||||
|
|
||||||
|
**Recommendation:** Setup wizard on first launch
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 5. Provider Auto-Start
|
||||||
|
|
||||||
|
**Question:** Should provider start automatically with app?
|
||||||
|
|
||||||
|
**Options:**
|
||||||
|
|
||||||
|
- A. Always start selected provider on app launch
|
||||||
|
- B. Start on-demand (when user generates speech)
|
||||||
|
- C. User preference
|
||||||
|
|
||||||
|
**Recommendation:** Auto-start (configurable in settings)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Future Enhancements
|
||||||
|
|
||||||
|
- [ ] **Provider Marketplace:** Built-in directory of community providers
|
||||||
|
- [ ] **Multi-Provider Support:** Use different providers per voice/language
|
||||||
|
- [ ] **Provider Health Monitoring:** Automatic failover if provider crashes
|
||||||
|
- [ ] **Cost Tracking:** Monitor API usage for OpenAI/cloud providers
|
||||||
|
- [ ] **Performance Metrics:** Latency, throughput, VRAM usage dashboards
|
||||||
|
- [ ] **Docker Providers:** Run providers in Docker containers
|
||||||
|
- [ ] **Provider Plugins:** Load custom providers from user scripts
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Related Documents
|
||||||
|
|
||||||
|
- [EXTERNAL_PROVIDERS.md](./EXTERNAL_PROVIDERS.md) - External provider support plan
|
||||||
|
- [OPENAI_SUPPORT.md](./OPENAI_SUPPORT.md) - OpenAI API compatibility
|
||||||
|
- [github-2gb-limit-issue.md](../github-2gb-limit-issue.md) - Original problem
|
||||||
|
- [r2-setup.md](../r2-setup.md) - Cloudflare R2 configuration
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Contributing
|
||||||
|
|
||||||
|
If you want to build a custom TTS provider:
|
||||||
|
|
||||||
|
1. Implement the provider API spec (see above)
|
||||||
|
2. Test with Voicebox locally
|
||||||
|
3. Package as executable (PyInstaller, Docker, etc.)
|
||||||
|
4. Share in GitHub Discussions
|
||||||
|
|
||||||
|
**Questions?**
|
||||||
|
|
||||||
|
- GitHub Issues: [voicebox/issues](https://github.com/jamiepine/voicebox/issues)
|
||||||
|
- Discord: Coming soon
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/landing",
|
"name": "@voicebox/landing",
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"description": "Landing page for voicebox.sh",
|
"description": "Landing page for voicebox.sh",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "bun --bun next dev --turbo",
|
"dev": "bun --bun next dev --turbo",
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
import type { Metadata } from 'next';
|
import type { Metadata } from 'next';
|
||||||
import { Inter } from 'next/font/google';
|
import { Inter } from 'next/font/google';
|
||||||
import './globals.css';
|
import './globals.css';
|
||||||
|
import { Banner } from '@/components/Banner';
|
||||||
import { Footer } from '@/components/Footer';
|
import { Footer } from '@/components/Footer';
|
||||||
import { Header } from '@/components/Header';
|
import { Header } from '@/components/Header';
|
||||||
|
|
||||||
@@ -31,6 +32,7 @@ export default function RootLayout({ children }: { children: React.ReactNode })
|
|||||||
<html lang="en" suppressHydrationWarning className="dark">
|
<html lang="en" suppressHydrationWarning className="dark">
|
||||||
<body className={inter.variable}>
|
<body className={inter.variable}>
|
||||||
<div className="relative min-h-screen bg-background font-sans flex flex-col">
|
<div className="relative min-h-screen bg-background font-sans flex flex-col">
|
||||||
|
<Banner />
|
||||||
<Header />
|
<Header />
|
||||||
<main className="container mx-auto px-4 sm:px-6 md:px-4 flex-1 py-4 sm:py-6 md:py-0">
|
<main className="container mx-auto px-4 sm:px-6 md:px-4 flex-1 py-4 sm:py-6 md:py-0">
|
||||||
{children}
|
{children}
|
||||||
|
|||||||
@@ -239,8 +239,9 @@ export default function Home() {
|
|||||||
<div className="space-y-6 text-lg text-foreground/80 text-center">
|
<div className="space-y-6 text-lg text-foreground/80 text-center">
|
||||||
<p>
|
<p>
|
||||||
Voicebox is a <strong>local-first voice cloning studio</strong> with DAW-like features
|
Voicebox is a <strong>local-first voice cloning studio</strong> with DAW-like features
|
||||||
for professional voice synthesis. Think of it as the <strong>Ollama for voice</strong>{' '}
|
for professional voice synthesis. Think of it as a{' '}
|
||||||
— download models, clone voices, and generate speech entirely on your machine.
|
<strong>local, free and open-source alternative to ElevenLabs</strong> — download
|
||||||
|
models, clone voices, and generate speech entirely on your machine.
|
||||||
</p>
|
</p>
|
||||||
<p>
|
<p>
|
||||||
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives
|
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
import { ArrowRight } from 'lucide-react';
|
||||||
|
|
||||||
|
export function Banner() {
|
||||||
|
return (
|
||||||
|
<div className="bg-primary/[0.06] border-b border-border backdrop-blur-sm">
|
||||||
|
<div className="container mx-auto px-4">
|
||||||
|
<div className="flex items-center justify-center h-10 text-sm">
|
||||||
|
<a
|
||||||
|
href="https://spacebot.sh"
|
||||||
|
target="_blank"
|
||||||
|
rel="noopener noreferrer"
|
||||||
|
className="flex items-center gap-2 text-muted-foreground hover:text-foreground transition-colors group"
|
||||||
|
>
|
||||||
|
<span>
|
||||||
|
Also by the creator of Voicebox:{' '}
|
||||||
|
<strong className="text-foreground/90">Spacebot</strong>, an AI agent OS for teams.
|
||||||
|
Connect Discord, Slack, or Telegram in one click.
|
||||||
|
</span>
|
||||||
|
<ArrowRight className="h-3.5 w-3.5 transition-transform group-hover:translate-x-0.5" />
|
||||||
|
</a>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
);
|
||||||
|
}
|
||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "voicebox",
|
"name": "voicebox",
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"private": true,
|
"private": true,
|
||||||
"workspaces": [
|
"workspaces": [
|
||||||
"app",
|
"app",
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
uvicorn
|
||||||
|
fastapi
|
||||||
|
sqlalchemy
|
||||||
|
torch
|
||||||
|
torchvision
|
||||||
|
soundfile
|
||||||
|
librosa
|
||||||
|
python-multipart
|
||||||
|
huggingface_hub
|
||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/tauri",
|
"name": "@voicebox/tauri",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Generated
+1
-1
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.1.11"
|
version = "0.1.12"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"core-foundation-sys",
|
"core-foundation-sys",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.1.12"
|
version = "0.1.13"
|
||||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||||
authors = ["you"]
|
authors = ["you"]
|
||||||
license = ""
|
license = ""
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
use crate::audio_capture::AudioCaptureState;
|
||||||
|
|
||||||
|
pub async fn start_capture(
|
||||||
|
state: &AudioCaptureState,
|
||||||
|
max_duration_secs: u32,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
todo!("implement Linux audio capture")
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn stop_capture(state: &AudioCaptureState) -> Result<String, String> {
|
||||||
|
todo!("implement Linux audio capture stop")
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn is_supported() -> bool {
|
||||||
|
false
|
||||||
|
}
|
||||||
@@ -2,11 +2,15 @@
|
|||||||
mod macos;
|
mod macos;
|
||||||
#[cfg(target_os = "windows")]
|
#[cfg(target_os = "windows")]
|
||||||
mod windows;
|
mod windows;
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
mod linux;
|
||||||
|
|
||||||
#[cfg(target_os = "macos")]
|
#[cfg(target_os = "macos")]
|
||||||
pub use macos::*;
|
pub use macos::*;
|
||||||
#[cfg(target_os = "windows")]
|
#[cfg(target_os = "windows")]
|
||||||
pub use windows::*;
|
pub use windows::*;
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
pub use linux::*;
|
||||||
|
|
||||||
use std::sync::{Arc, Mutex};
|
use std::sync::{Arc, Mutex};
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://schema.tauri.app/config/2",
|
"$schema": "https://schema.tauri.app/config/2",
|
||||||
"productName": "Voicebox",
|
"productName": "Voicebox",
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"identifier": "sh.voicebox.app",
|
"identifier": "sh.voicebox.app",
|
||||||
"build": {
|
"build": {
|
||||||
"beforeDevCommand": "bun run dev",
|
"beforeDevCommand": "bun run dev",
|
||||||
@@ -12,7 +12,7 @@
|
|||||||
"bundle": {
|
"bundle": {
|
||||||
"active": true,
|
"active": true,
|
||||||
"targets": "all",
|
"targets": "all",
|
||||||
"createUpdaterArtifacts": true,
|
"createUpdaterArtifacts": false,
|
||||||
"externalBin": ["binaries/voicebox-server"],
|
"externalBin": ["binaries/voicebox-server"],
|
||||||
"icon": [
|
"icon": [
|
||||||
"icons/32x32.png",
|
"icons/32x32.png",
|
||||||
|
|||||||
@@ -2,29 +2,25 @@ import type { PlatformFilesystem, FileFilter } from '@/platform/types';
|
|||||||
|
|
||||||
export const tauriFilesystem: PlatformFilesystem = {
|
export const tauriFilesystem: PlatformFilesystem = {
|
||||||
async saveFile(filename: string, blob: Blob, filters?: FileFilter[]) {
|
async saveFile(filename: string, blob: Blob, filters?: FileFilter[]) {
|
||||||
try {
|
const { save } = await import('@tauri-apps/plugin-dialog');
|
||||||
const { save } = await import('@tauri-apps/plugin-dialog');
|
const { writeFile } = await import('@tauri-apps/plugin-fs');
|
||||||
const filePath = await save({
|
|
||||||
defaultPath: filename,
|
|
||||||
filters: filters || [],
|
|
||||||
});
|
|
||||||
|
|
||||||
if (filePath) {
|
const filePath = await save({
|
||||||
const { writeBinaryFile } = await import('@tauri-apps/plugin-fs');
|
defaultPath: filename,
|
||||||
const arrayBuffer = await blob.arrayBuffer();
|
filters: filters || [],
|
||||||
await writeBinaryFile(filePath, new Uint8Array(arrayBuffer));
|
});
|
||||||
}
|
|
||||||
} catch (error) {
|
if (!filePath) return; // User cancelled the dialog
|
||||||
console.error('Failed to use Tauri dialog, falling back to browser download:', error);
|
|
||||||
// Fall back to browser download if Tauri dialog fails
|
const resolvedPath = typeof filePath === 'string'
|
||||||
const url = window.URL.createObjectURL(blob);
|
? filePath
|
||||||
const a = document.createElement('a');
|
: (filePath as { path: string }).path;
|
||||||
a.href = url;
|
|
||||||
a.download = filename;
|
if (!resolvedPath) {
|
||||||
document.body.appendChild(a);
|
throw new Error('Failed to resolve save path from dialog');
|
||||||
a.click();
|
|
||||||
window.URL.revokeObjectURL(url);
|
|
||||||
document.body.removeChild(a);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const arrayBuffer = await blob.arrayBuffer();
|
||||||
|
await writeFile(resolvedPath, new Uint8Array(arrayBuffer));
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
|
|||||||
+2
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/web",
|
"name": "@voicebox/web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.12",
|
"version": "0.1.13",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
@@ -21,6 +21,7 @@
|
|||||||
"@types/react-dom": "^18.3.0",
|
"@types/react-dom": "^18.3.0",
|
||||||
"@typescript-eslint/eslint-plugin": "^7.0.0",
|
"@typescript-eslint/eslint-plugin": "^7.0.0",
|
||||||
"@typescript-eslint/parser": "^7.0.0",
|
"@typescript-eslint/parser": "^7.0.0",
|
||||||
|
"@tailwindcss/vite": "^4.0.0",
|
||||||
"@vitejs/plugin-react": "^4.3.0",
|
"@vitejs/plugin-react": "^4.3.0",
|
||||||
"eslint": "^8.57.0",
|
"eslint": "^8.57.0",
|
||||||
"eslint-plugin-react-hooks": "^4.6.0",
|
"eslint-plugin-react-hooks": "^4.6.0",
|
||||||
|
|||||||
+2
-1
@@ -1,9 +1,10 @@
|
|||||||
import path from 'node:path';
|
import path from 'node:path';
|
||||||
import react from '@vitejs/plugin-react';
|
import react from '@vitejs/plugin-react';
|
||||||
|
import tailwindcss from '@tailwindcss/vite';
|
||||||
import { defineConfig } from 'vite';
|
import { defineConfig } from 'vite';
|
||||||
|
|
||||||
export default defineConfig({
|
export default defineConfig({
|
||||||
plugins: [react()],
|
plugins: [react(), tailwindcss()],
|
||||||
resolve: {
|
resolve: {
|
||||||
alias: {
|
alias: {
|
||||||
'@': path.resolve(__dirname, '../app/src'),
|
'@': path.resolve(__dirname, '../app/src'),
|
||||||
|
|||||||
Reference in New Issue
Block a user