mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-28 14:45:16 -07:00
Compare commits
42
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
38bf96ff20 | ||
|
|
0b14cb1b2c | ||
|
|
4d24e69012 | ||
|
|
90436e428d | ||
|
|
e4bb288904 | ||
|
|
6f8bc7f23b | ||
|
|
baca111d50 | ||
|
|
cc298fe6d8 | ||
|
|
46b8f6b882 | ||
|
|
162cf4fb84 | ||
|
|
68558243d9 | ||
|
|
8d5ad926f9 | ||
|
|
334f037dce | ||
|
|
f6522eea80 | ||
|
|
7615a08f81 | ||
|
|
31ea3c68a5 | ||
|
|
54d72ddfd0 | ||
|
|
d4794f78e1 | ||
|
|
aa7c9a9a8d | ||
|
|
ca6ed0998a | ||
|
|
829d4d6d5b | ||
|
|
0be7975db5 | ||
|
|
40e4af828a | ||
|
|
0e57826ea5 | ||
|
|
eb2cd861b1 | ||
|
|
701cc647a7 | ||
|
|
be6ccaf044 | ||
|
|
1040625a88 | ||
|
|
6f4503b521 | ||
|
|
f5b6edc2e7 | ||
|
|
8197f0724c | ||
|
|
d40f7d2676 | ||
|
|
99fbcca7f4 | ||
|
|
04f9880c9a | ||
|
|
b9c858295d | ||
|
|
610f64c762 | ||
|
|
220333b3bb | ||
|
|
e194e95512 | ||
|
|
e796412c2c | ||
|
|
cb541521d2 | ||
|
|
2bc243f93e | ||
|
|
0209008d73 |
+1
-1
@@ -1,5 +1,5 @@
|
||||
[bumpversion]
|
||||
current_version = 0.1.12
|
||||
current_version = 0.1.13
|
||||
commit = True
|
||||
tag = True
|
||||
tag_name = v{new_version}
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
name: Build Windows
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
build-windows:
|
||||
permissions:
|
||||
contents: write
|
||||
runs-on: windows-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: "pip"
|
||||
|
||||
- name: Install Python dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install pyinstaller
|
||||
pip install -r backend/requirements.txt
|
||||
|
||||
- name: Build Python server
|
||||
shell: bash
|
||||
run: |
|
||||
cd backend
|
||||
python build_binary.py
|
||||
|
||||
PLATFORM=$(rustc --print host-tuple)
|
||||
mkdir -p ../tauri/src-tauri/binaries
|
||||
cp dist/voicebox-server.exe ../tauri/src-tauri/binaries/voicebox-server-${PLATFORM}.exe
|
||||
echo "Built voicebox-server-${PLATFORM}.exe"
|
||||
|
||||
- name: Setup Bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
|
||||
- name: Install Rust stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Rust cache
|
||||
uses: swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: "./tauri/src-tauri -> target"
|
||||
|
||||
- name: Install dependencies
|
||||
run: bun install
|
||||
|
||||
- uses: tauri-apps/tauri-action@v0
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
projectPath: tauri
|
||||
tagName: v__VERSION__
|
||||
releaseName: "voicebox v__VERSION__ (test build)"
|
||||
releaseBody: "Test build for audio export fix"
|
||||
releaseDraft: true
|
||||
prerelease: true
|
||||
args: ""
|
||||
includeUpdaterJson: false
|
||||
@@ -4,7 +4,7 @@ on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
- "v*"
|
||||
|
||||
jobs:
|
||||
release:
|
||||
@@ -14,22 +14,22 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- platform: 'macos-latest'
|
||||
args: '--target aarch64-apple-darwin'
|
||||
python-version: '3.12'
|
||||
backend: 'mlx'
|
||||
- platform: 'macos-15-intel'
|
||||
args: '--target x86_64-apple-darwin'
|
||||
python-version: '3.12'
|
||||
backend: 'pytorch'
|
||||
- platform: "macos-latest"
|
||||
args: "--target aarch64-apple-darwin"
|
||||
python-version: "3.12"
|
||||
backend: "mlx"
|
||||
- platform: "macos-15-intel"
|
||||
args: "--target x86_64-apple-darwin"
|
||||
python-version: "3.12"
|
||||
backend: "pytorch"
|
||||
# - platform: 'ubuntu-22.04'
|
||||
# args: ''
|
||||
# python-version: '3.12'
|
||||
# backend: 'pytorch'
|
||||
- platform: 'windows-latest'
|
||||
args: ''
|
||||
python-version: '3.12'
|
||||
backend: 'pytorch'
|
||||
- platform: "windows-latest"
|
||||
args: ""
|
||||
python-version: "3.12"
|
||||
backend: "pytorch"
|
||||
|
||||
runs-on: ${{ matrix.platform }}
|
||||
|
||||
@@ -53,7 +53,7 @@ jobs:
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
cache: 'pip'
|
||||
cache: "pip"
|
||||
|
||||
- name: Install Python dependencies
|
||||
run: |
|
||||
@@ -66,24 +66,24 @@ jobs:
|
||||
run: |
|
||||
pip install -r backend/requirements-mlx.txt
|
||||
|
||||
# - name: Install PyTorch with CUDA (Windows only)
|
||||
# if: matrix.platform == 'windows-latest'
|
||||
# run: |
|
||||
# pip install torch --index-url https://download.pytorch.org/whl/cu121 --force-reinstall --no-deps
|
||||
# pip install torchvision torchaudio --index-url https://download.pytorch.org/whl/cu121
|
||||
|
||||
- name: Build Python server (Linux/macOS)
|
||||
if: matrix.platform != 'windows-latest'
|
||||
run: |
|
||||
chmod +x scripts/build-server.sh
|
||||
./scripts/build-server.sh
|
||||
|
||||
- name: Build CPU Python server (Windows)
|
||||
- name: Build Python server (Windows)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
shell: bash
|
||||
run: |
|
||||
cd backend
|
||||
|
||||
echo "Installing CPU-only PyTorch..."
|
||||
pip uninstall -y torch torchvision torchaudio
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
|
||||
|
||||
echo "Building CPU server binary..."
|
||||
python build_binary.py cpu
|
||||
python build_binary.py
|
||||
|
||||
# Get platform tuple
|
||||
PLATFORM=$(rustc --print host-tuple)
|
||||
@@ -91,31 +91,9 @@ jobs:
|
||||
# Create binaries directory
|
||||
mkdir -p ../tauri/src-tauri/binaries
|
||||
|
||||
# Copy CPU version (default for installer)
|
||||
# Copy with platform suffix
|
||||
cp dist/voicebox-server.exe ../tauri/src-tauri/binaries/voicebox-server-${PLATFORM}.exe
|
||||
echo "Built CPU server: voicebox-server-${PLATFORM}.exe (~500MB)"
|
||||
|
||||
- name: Build CUDA Python server (Windows)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
shell: bash
|
||||
run: |
|
||||
cd backend
|
||||
|
||||
echo "Installing CUDA PyTorch..."
|
||||
pip uninstall -y torch torchvision torchaudio
|
||||
pip install torch --index-url https://download.pytorch.org/whl/cu121 --force-reinstall --no-deps
|
||||
pip install torchvision torchaudio --index-url https://download.pytorch.org/whl/cu121
|
||||
|
||||
echo "Building CUDA server binary..."
|
||||
python build_binary.py cuda
|
||||
|
||||
# Get platform tuple
|
||||
PLATFORM=$(rustc --print host-tuple)
|
||||
|
||||
# Copy CUDA version for separate upload
|
||||
mkdir -p cuda-release
|
||||
cp dist/voicebox-server-cuda.exe cuda-release/voicebox-server-cuda-${PLATFORM}.exe
|
||||
echo "Built CUDA server: voicebox-server-cuda-${PLATFORM}.exe (~3GB)"
|
||||
echo "Built voicebox-server-${PLATFORM}.exe"
|
||||
|
||||
- name: Setup Bun
|
||||
uses: oven-sh/setup-bun@v2
|
||||
@@ -128,7 +106,7 @@ jobs:
|
||||
- name: Rust cache
|
||||
uses: swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: './tauri/src-tauri -> target'
|
||||
workspaces: "./tauri/src-tauri -> target"
|
||||
|
||||
- name: Install dependencies
|
||||
run: bun install
|
||||
@@ -164,7 +142,7 @@ jobs:
|
||||
with:
|
||||
projectPath: tauri
|
||||
tagName: v__VERSION__
|
||||
releaseName: 'voicebox v__VERSION__'
|
||||
releaseName: "voicebox v__VERSION__"
|
||||
releaseBody: |
|
||||
## What's Changed
|
||||
See the assets below to download and install this version.
|
||||
@@ -172,41 +150,11 @@ jobs:
|
||||
### Installation
|
||||
- **macOS (Apple Silicon)**: Download the `aarch64.dmg` file - uses MLX for fast native inference
|
||||
- **macOS (Intel)**: Download the `x64.dmg` file - uses PyTorch
|
||||
- **Windows**: Download the `.msi` installer - includes CPU-only inference (~500MB)
|
||||
- **Windows**: Download the `.msi` installer
|
||||
- **Linux**: Download the `.AppImage` or `.deb` package
|
||||
|
||||
### NVIDIA GPU Acceleration (Windows)
|
||||
Windows users with NVIDIA GPUs can enable CUDA for 4-5x faster inference:
|
||||
1. Install the app normally (CPU version included in installer)
|
||||
2. The app will detect your GPU and offer to download CUDA support automatically
|
||||
3. Or manually download: [voicebox-server-cuda-x86_64-pc-windows-msvc.exe](https://downloads.voicebox.sh/cuda/__VERSION__/voicebox-server-cuda-x86_64-pc-windows-msvc.exe) (~2.4GB)
|
||||
|
||||
The app includes automatic updates - future updates will be installed automatically.
|
||||
releaseDraft: true
|
||||
prerelease: false
|
||||
args: ${{ matrix.args }}
|
||||
includeUpdaterJson: true
|
||||
|
||||
- name: Upload CUDA server to Cloudflare R2 (Windows only)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
|
||||
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
|
||||
run: |
|
||||
# Install AWS CLI if not available
|
||||
pip install awscli
|
||||
|
||||
# Get version from tag
|
||||
VERSION=${GITHUB_REF#refs/tags/}
|
||||
|
||||
# Get platform tuple
|
||||
PLATFORM=$(rustc --print host-tuple)
|
||||
|
||||
# Upload to R2
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-${PLATFORM}.exe \
|
||||
s3://voicebox/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe \
|
||||
--endpoint-url $R2_ENDPOINT \
|
||||
--acl public-read
|
||||
|
||||
echo "CUDA binary uploaded to: https://downloads.voicebox.sh/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe"
|
||||
|
||||
@@ -53,6 +53,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
- Audio export failing when Tauri save dialog returns object instead of string path
|
||||
|
||||
### Added
|
||||
- **Makefile** - Comprehensive development workflow automation with commands for setup, development, building, testing, and code quality checks
|
||||
- Includes Python version detection and compatibility warnings
|
||||
|
||||
@@ -59,7 +59,7 @@
|
||||
|
||||
## What is Voicebox?
|
||||
|
||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as the **Ollama for voice** — download models, clone voices, and generate speech entirely on your machine.
|
||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as a **local, free and open-source alternative to ElevenLabs** — download models, clone voices, and generate speech entirely on your machine.
|
||||
|
||||
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives you:
|
||||
|
||||
@@ -80,10 +80,10 @@ Voicebox is available now for macOS and Windows.
|
||||
|
||||
| Platform | Download |
|
||||
|----------|----------|
|
||||
| macOS (Apple Silicon) | [voicebox_aarch64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_aarch64.app.tar.gz) |
|
||||
| macOS (Intel) | [voicebox_x64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_x64.app.tar.gz) |
|
||||
| Windows (MSI) | [voicebox_0.1.0_x64_en-US.msi](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_0.1.0_x64_en-US.msi) |
|
||||
| Windows (Setup) | [voicebox_0.1.0_x64-setup.exe](https://github.com/jamiepine/voicebox/releases/download/v0.1.0/voicebox_0.1.0_x64-setup.exe) |
|
||||
| macOS (Apple Silicon) | [Voicebox_aarch64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/latest/download/Voicebox_aarch64.app.tar.gz) |
|
||||
| macOS (Intel) | [Voicebox_x64.app.tar.gz](https://github.com/jamiepine/voicebox/releases/latest/download/Voicebox_x64.app.tar.gz) |
|
||||
| Windows (MSI) | [Latest Windows MSI](https://github.com/jamiepine/voicebox/releases/latest) |
|
||||
| Windows (Setup) | [Latest Windows Setup](https://github.com/jamiepine/voicebox/releases/latest) |
|
||||
|
||||
> **Linux builds coming soon** — Currently blocked by GitHub runner disk space limitations.
|
||||
|
||||
@@ -233,7 +233,7 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed setup and contribution guide
|
||||
|
||||
```bash
|
||||
# Clone the repo
|
||||
git clone https://github.com/voicebox-sh/voicebox.git
|
||||
git clone https://github.com/jamiepine/voicebox.git
|
||||
cd voicebox
|
||||
|
||||
# Setup everything
|
||||
@@ -247,7 +247,7 @@ make dev
|
||||
|
||||
```bash
|
||||
# Clone the repo
|
||||
git clone https://github.com/voicebox-sh/voicebox.git
|
||||
git clone https://github.com/jamiepine/voicebox.git
|
||||
cd voicebox
|
||||
|
||||
# Install dependencies
|
||||
@@ -260,7 +260,7 @@ cd backend && pip install -r requirements.txt && cd ..
|
||||
bun run dev
|
||||
```
|
||||
|
||||
**Prerequisites:** [Bun](https://bun.sh), [Rust](https://rustup.rs), [Python 3.11+](https://python.org).
|
||||
**Prerequisites:** [Bun](https://bun.sh), [Rust](https://rustup.rs), [Python 3.11+](https://python.org). [XCode on macOS](https://developer.apple.com/xcode/).
|
||||
|
||||
**Performance:**
|
||||
- **Apple Silicon (M1/M2/M3)**: Uses MLX backend with native Metal acceleration for 4-5x faster inference
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@voicebox/app",
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
|
||||
@@ -124,6 +124,13 @@ export function AudioTab() {
|
||||
);
|
||||
}
|
||||
|
||||
const handleChannelDelete = async (e, channelId) => {
|
||||
e.stopPropagation();
|
||||
if (await confirm('Delete this channel?')) {
|
||||
deleteChannel.mutate(channelId);
|
||||
}
|
||||
}
|
||||
|
||||
const allChannels = channels || [];
|
||||
const allDevices = devices || [];
|
||||
const selectedChannel = selectedChannelId
|
||||
@@ -241,12 +248,7 @@ export function AudioTab() {
|
||||
variant="ghost"
|
||||
size="sm"
|
||||
className="h-8 w-8 p-0"
|
||||
onClick={(e) => {
|
||||
e.stopPropagation();
|
||||
if (confirm('Delete this channel?')) {
|
||||
deleteChannel.mutate(channel.id);
|
||||
}
|
||||
}}
|
||||
onClick={(e) => handleChannelDelete(e, channel.id)}
|
||||
>
|
||||
<Trash2 className="h-4 w-4" />
|
||||
</Button>
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { useMatchRoute } from '@tanstack/react-router';
|
||||
import { AnimatePresence, motion } from 'framer-motion';
|
||||
import { Loader2, MessageSquare, Sparkles } from 'lucide-react';
|
||||
import { Loader2, SlidersHorizontal, Sparkles } from 'lucide-react';
|
||||
import { useEffect, useRef, useState } from 'react';
|
||||
import { Button } from '@/components/ui/button';
|
||||
import { Form, FormControl, FormField, FormItem, FormMessage } from '@/components/ui/form';
|
||||
@@ -187,7 +187,7 @@ export function FloatingGenerateBox({
|
||||
}}
|
||||
>
|
||||
<motion.div
|
||||
className="bg-background/30 backdrop-blur-2xl border border-accent/20 rounded-[2rem] shadow-2xl hover:bg-background/40 hover:border-accent/20 transition-all duration-300 overflow-hidden p-3"
|
||||
className="bg-background/30 backdrop-blur-2xl border border-accent/20 rounded-[2rem] shadow-2xl hover:bg-background/40 hover:border-accent/20 transition-all duration-300 p-3"
|
||||
transition={{ duration: 0.6, ease: 'easeInOut' }}
|
||||
>
|
||||
<Form {...form}>
|
||||
@@ -274,7 +274,7 @@ export function FloatingGenerateBox({
|
||||
field.ref(node);
|
||||
}
|
||||
}}
|
||||
placeholder="Add delivery instructions..."
|
||||
placeholder="e.g. very happy and excited"
|
||||
className="resize-none bg-transparent border-none focus-visible:ring-0 focus-visible:ring-offset-0 focus:outline-none focus:ring-0 outline-none ring-0 rounded-2xl text-sm placeholder:text-muted-foreground/60 w-full"
|
||||
style={{
|
||||
minHeight: isExpanded ? '100px' : '32px',
|
||||
@@ -294,18 +294,27 @@ export function FloatingGenerateBox({
|
||||
</motion.div>
|
||||
|
||||
<div className="relative shrink-0">
|
||||
<Button
|
||||
type="submit"
|
||||
disabled={isPending || !selectedProfileId}
|
||||
className="h-10 w-10 rounded-full bg-accent hover:bg-accent/90 hover:scale-105 text-accent-foreground shadow-lg hover:shadow-accent/50 transition-all duration-200"
|
||||
size="icon"
|
||||
>
|
||||
{isPending ? (
|
||||
<Loader2 className="h-4 w-4 animate-spin" />
|
||||
) : (
|
||||
<Sparkles className="h-4 w-4" />
|
||||
)}
|
||||
</Button>
|
||||
<div className="group relative">
|
||||
<Button
|
||||
type="submit"
|
||||
disabled={isPending || !selectedProfileId}
|
||||
className="h-10 w-10 rounded-full bg-accent hover:bg-accent/90 hover:scale-105 text-accent-foreground shadow-lg hover:shadow-accent/50 transition-all duration-200"
|
||||
size="icon"
|
||||
>
|
||||
{isPending ? (
|
||||
<Loader2 className="h-4 w-4 animate-spin" />
|
||||
) : (
|
||||
<Sparkles className="h-4 w-4" />
|
||||
)}
|
||||
</Button>
|
||||
<span className="pointer-events-none absolute bottom-full left-1/2 -translate-x-1/2 mb-2 whitespace-nowrap rounded-md bg-popover px-3 py-1.5 text-xs text-popover-foreground border border-border opacity-0 transition-opacity group-hover:opacity-100 z-[9999]">
|
||||
{isPending
|
||||
? 'Generating...'
|
||||
: !selectedProfileId
|
||||
? 'Select a voice profile first'
|
||||
: 'Generate speech'}
|
||||
</span>
|
||||
</div>
|
||||
<AnimatePresence>
|
||||
{isExpanded && (
|
||||
<motion.div
|
||||
@@ -315,20 +324,25 @@ export function FloatingGenerateBox({
|
||||
transition={{ duration: 0.2 }}
|
||||
className="absolute top-0 right-[calc(100%+0.5rem)]"
|
||||
>
|
||||
<Button
|
||||
type="button"
|
||||
variant="ghost"
|
||||
size="icon"
|
||||
onClick={() => setIsInstructMode(!isInstructMode)}
|
||||
className={cn(
|
||||
'h-10 w-10 rounded-full transition-all duration-200',
|
||||
isInstructMode
|
||||
? 'bg-accent text-accent-foreground border border-accent hover:bg-accent/90'
|
||||
: 'bg-card border border-border hover:bg-background/50',
|
||||
)}
|
||||
>
|
||||
<MessageSquare className="h-4 w-4" />
|
||||
</Button>
|
||||
<div className="group relative">
|
||||
<Button
|
||||
type="button"
|
||||
variant="ghost"
|
||||
size="icon"
|
||||
onClick={() => setIsInstructMode(!isInstructMode)}
|
||||
className={cn(
|
||||
'h-10 w-10 rounded-full transition-all duration-200',
|
||||
isInstructMode
|
||||
? 'bg-accent text-accent-foreground border border-accent hover:bg-accent/90'
|
||||
: 'bg-card border border-border hover:bg-background/50',
|
||||
)}
|
||||
>
|
||||
<SlidersHorizontal className="h-4 w-4" />
|
||||
</Button>
|
||||
<span className="pointer-events-none absolute bottom-full left-1/2 -translate-x-1/2 mb-2 whitespace-nowrap rounded-md bg-popover px-3 py-1.5 text-xs text-popover-foreground border border-border opacity-0 transition-opacity group-hover:opacity-100 z-[9999]">
|
||||
Fine tune instructions
|
||||
</span>
|
||||
</div>
|
||||
</motion.div>
|
||||
)}
|
||||
</AnimatePresence>
|
||||
|
||||
@@ -58,6 +58,7 @@ export function AudioSampleRecording({
|
||||
// Request microphone access when component mounts
|
||||
useEffect(() => {
|
||||
if (!showWaveform) return;
|
||||
if (!navigator.mediaDevices || !navigator.mediaDevices.getUserMedia) return;
|
||||
|
||||
let stream: MediaStream | null = null;
|
||||
|
||||
|
||||
@@ -43,7 +43,7 @@ import {
|
||||
} from '@/lib/hooks/useProfiles';
|
||||
import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
||||
import { useTranscription } from '@/lib/hooks/useTranscription';
|
||||
import { formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
||||
import { convertToWav, formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
||||
import { usePlatform } from '@/platform/PlatformContext';
|
||||
import { useServerStore } from '@/stores/serverStore';
|
||||
import { type ProfileFormDraft, useUIStore } from '@/stores/uiStore';
|
||||
@@ -505,10 +505,23 @@ export function ProfileForm() {
|
||||
language: data.language,
|
||||
});
|
||||
|
||||
// Convert non-WAV uploads to WAV so the backend can always use soundfile.
|
||||
// Recorded audio is already WAV (from useAudioRecording's convertToWav call).
|
||||
let fileToUpload: File = sampleFile;
|
||||
if (!sampleFile.type.includes('wav') && !sampleFile.name.toLowerCase().endsWith('.wav')) {
|
||||
try {
|
||||
const wavBlob = await convertToWav(sampleFile);
|
||||
const wavName = sampleFile.name.replace(/\.[^.]+$/, '.wav');
|
||||
fileToUpload = new File([wavBlob], wavName, { type: 'audio/wav' });
|
||||
} catch {
|
||||
// If browser can't decode the format, send the original and let the backend try.
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
await addSample.mutateAsync({
|
||||
profileId: profile.id,
|
||||
file: sampleFile,
|
||||
file: fileToUpload,
|
||||
referenceText: referenceText,
|
||||
});
|
||||
|
||||
|
||||
@@ -79,8 +79,8 @@ export function VoicesTab() {
|
||||
setDialogOpen(true);
|
||||
};
|
||||
|
||||
const handleDelete = (profileId: string) => {
|
||||
if (confirm('Are you sure you want to delete this profile?')) {
|
||||
const handleProfileDelete = async (profileId: string) => {
|
||||
if (await confirm('Are you sure you want to delete this profile?')) {
|
||||
deleteProfile.mutate(profileId);
|
||||
}
|
||||
};
|
||||
@@ -147,7 +147,7 @@ export function VoicesTab() {
|
||||
channels={channels || []}
|
||||
onChannelChange={(channelIds) => handleChannelChange(profile.id, channelIds)}
|
||||
onEdit={() => handleEdit(profile.id)}
|
||||
onDelete={() => handleDelete(profile.id)}
|
||||
onDelete={() => handleProfileDelete(profile.id)}
|
||||
/>
|
||||
))}
|
||||
</TableBody>
|
||||
|
||||
@@ -20,11 +20,13 @@ export function useAudioRecording({
|
||||
const streamRef = useRef<MediaStream | null>(null);
|
||||
const timerRef = useRef<number | null>(null);
|
||||
const startTimeRef = useRef<number | null>(null);
|
||||
const cancelledRef = useRef<boolean>(false);
|
||||
|
||||
const startRecording = useCallback(async () => {
|
||||
try {
|
||||
setError(null);
|
||||
chunksRef.current = [];
|
||||
cancelledRef.current = false;
|
||||
setDuration(0);
|
||||
|
||||
// Check if getUserMedia is available
|
||||
@@ -87,31 +89,34 @@ export function useAudioRecording({
|
||||
};
|
||||
|
||||
mediaRecorder.onstop = async () => {
|
||||
// Snapshot the cancellation flag and recorded duration immediately —
|
||||
// cancelRecording() clears chunks and sets cancelledRef synchronously
|
||||
// before this async handler runs, so we must check it first.
|
||||
const wasCancelled = cancelledRef.current;
|
||||
const recordedDuration = startTimeRef.current
|
||||
? (Date.now() - startTimeRef.current) / 1000
|
||||
: undefined;
|
||||
|
||||
const webmBlob = new Blob(chunksRef.current, { type: 'audio/webm' });
|
||||
|
||||
// Convert to WAV format to avoid needing ffmpeg on backend
|
||||
try {
|
||||
const wavBlob = await convertToWav(webmBlob);
|
||||
|
||||
// Pass the actual recorded duration
|
||||
const recordedDuration = startTimeRef.current
|
||||
? (Date.now() - startTimeRef.current) / 1000
|
||||
: undefined;
|
||||
onRecordingComplete?.(wavBlob, recordedDuration);
|
||||
} catch (err) {
|
||||
console.error('Error converting audio to WAV:', err);
|
||||
// Fallback to original blob if conversion fails
|
||||
const recordedDuration = startTimeRef.current
|
||||
? (Date.now() - startTimeRef.current) / 1000
|
||||
: undefined;
|
||||
onRecordingComplete?.(webmBlob, recordedDuration);
|
||||
}
|
||||
|
||||
// Stop all tracks
|
||||
// Stop all tracks now that we have the data
|
||||
streamRef.current?.getTracks().forEach((track) => {
|
||||
track.stop();
|
||||
});
|
||||
streamRef.current = null;
|
||||
|
||||
// Don't fire completion callback if the recording was cancelled
|
||||
if (wasCancelled) return;
|
||||
|
||||
// Convert to WAV format to avoid needing ffmpeg on backend
|
||||
try {
|
||||
const wavBlob = await convertToWav(webmBlob);
|
||||
onRecordingComplete?.(wavBlob, recordedDuration);
|
||||
} catch (err) {
|
||||
console.error('Error converting audio to WAV:', err);
|
||||
// Fallback to original blob if conversion fails
|
||||
onRecordingComplete?.(webmBlob, recordedDuration);
|
||||
}
|
||||
};
|
||||
|
||||
mediaRecorder.onerror = (event) => {
|
||||
@@ -167,9 +172,10 @@ export function useAudioRecording({
|
||||
|
||||
const cancelRecording = useCallback(() => {
|
||||
if (mediaRecorderRef.current) {
|
||||
cancelledRef.current = true; // Must be set before stop() triggers onstop
|
||||
chunksRef.current = [];
|
||||
mediaRecorderRef.current.stop();
|
||||
setIsRecording(false);
|
||||
chunksRef.current = [];
|
||||
setDuration(0);
|
||||
}
|
||||
|
||||
|
||||
+36
-18
@@ -22,6 +22,11 @@ export function formatAudioDuration(seconds: number): string {
|
||||
* If the file has a recordedDuration property (from recording hooks),
|
||||
* use that instead of trying to read metadata. This fixes issues on Windows
|
||||
* where WebM files from MediaRecorder don't have proper duration metadata.
|
||||
*
|
||||
* For uploaded files we use AudioContext.decodeAudioData which fully decodes
|
||||
* the audio and returns the exact duration. This is more reliable than
|
||||
* HTMLMediaElement.duration which can return incorrect large values for VBR
|
||||
* MP3 files that lack a proper XING/VBRI header.
|
||||
*/
|
||||
export async function getAudioDuration(
|
||||
file: File & { recordedDuration?: number },
|
||||
@@ -30,26 +35,39 @@ export async function getAudioDuration(
|
||||
return file.recordedDuration;
|
||||
}
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
const audio = new Audio();
|
||||
const url = URL.createObjectURL(file);
|
||||
// Use Web Audio API for accurate duration — avoids VBR MP3 metadata issues.
|
||||
try {
|
||||
const audioContext = new AudioContext();
|
||||
try {
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const audioBuffer = await audioContext.decodeAudioData(arrayBuffer);
|
||||
return audioBuffer.duration;
|
||||
} finally {
|
||||
await audioContext.close();
|
||||
}
|
||||
} catch {
|
||||
// Fallback: read duration from the media element (less accurate but works for WAV).
|
||||
return new Promise((resolve, reject) => {
|
||||
const audio = new Audio();
|
||||
const url = URL.createObjectURL(file);
|
||||
|
||||
audio.addEventListener('loadedmetadata', () => {
|
||||
URL.revokeObjectURL(url);
|
||||
if (Number.isFinite(audio.duration) && audio.duration > 0) {
|
||||
resolve(audio.duration);
|
||||
} else {
|
||||
reject(new Error('Audio file has invalid duration metadata'));
|
||||
}
|
||||
audio.addEventListener('loadedmetadata', () => {
|
||||
URL.revokeObjectURL(url);
|
||||
if (Number.isFinite(audio.duration) && audio.duration > 0) {
|
||||
resolve(audio.duration);
|
||||
} else {
|
||||
reject(new Error('Audio file has invalid duration metadata'));
|
||||
}
|
||||
});
|
||||
|
||||
audio.addEventListener('error', () => {
|
||||
URL.revokeObjectURL(url);
|
||||
reject(new Error('Failed to load audio file'));
|
||||
});
|
||||
|
||||
audio.src = url;
|
||||
});
|
||||
|
||||
audio.addEventListener('error', () => {
|
||||
URL.revokeObjectURL(url);
|
||||
reject(new Error('Failed to load audio file'));
|
||||
});
|
||||
|
||||
audio.src = url;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
# Backend package
|
||||
|
||||
__version__ = "0.1.12"
|
||||
__version__ = "0.1.13"
|
||||
|
||||
@@ -29,9 +29,23 @@ class PyTorchTTSBackend:
|
||||
"""Get the best available device."""
|
||||
if torch.cuda.is_available():
|
||||
return "cuda"
|
||||
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||
# MPS can have issues, use CPU for stability
|
||||
return "cpu"
|
||||
# Intel Arc / Intel Xe GPU via intel-extension-for-pytorch (IPEX)
|
||||
try:
|
||||
import intel_extension_for_pytorch # noqa: F401
|
||||
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||
return "xpu"
|
||||
except ImportError:
|
||||
pass
|
||||
# Any GPU on Windows via DirectML (torch-directml)
|
||||
try:
|
||||
import torch_directml
|
||||
if torch_directml.device_count() > 0:
|
||||
return torch_directml.device(0)
|
||||
except ImportError:
|
||||
pass
|
||||
# MPS (Apple Silicon) — kept for completeness but MLX backend is preferred
|
||||
if hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||
return "cpu" # MPS disabled for stability; MLX backend handles Apple Silicon
|
||||
return "cpu"
|
||||
|
||||
def is_loaded(self) -> bool:
|
||||
@@ -166,11 +180,21 @@ class PyTorchTTSBackend:
|
||||
|
||||
# Load the model (tqdm is patched, but filters out non-download progress)
|
||||
try:
|
||||
self.model = Qwen3TTSModel.from_pretrained(
|
||||
model_path,
|
||||
device_map=self.device,
|
||||
torch_dtype=torch.float32 if self.device == "cpu" else torch.bfloat16,
|
||||
)
|
||||
# Don't pass device_map on CPU: accelerate's meta-tensor mechanism
|
||||
# causes "Cannot copy out of meta tensor" when moving to CPU.
|
||||
# Instead load directly then call .to(device) if needed.
|
||||
if self.device == "cpu":
|
||||
self.model = Qwen3TTSModel.from_pretrained(
|
||||
model_path,
|
||||
torch_dtype=torch.float32,
|
||||
low_cpu_mem_usage=False,
|
||||
)
|
||||
else:
|
||||
self.model = Qwen3TTSModel.from_pretrained(
|
||||
model_path,
|
||||
device_map=self.device,
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
finally:
|
||||
# Exit the patch context
|
||||
tracker_context.__exit__(None, None, None)
|
||||
@@ -358,9 +382,22 @@ class PyTorchSTTBackend:
|
||||
"""Get the best available device."""
|
||||
if torch.cuda.is_available():
|
||||
return "cuda"
|
||||
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||
# MPS support for Whisper
|
||||
return "cpu" # Use CPU for stability
|
||||
# Intel Arc / Intel Xe GPU via intel-extension-for-pytorch (IPEX)
|
||||
try:
|
||||
import intel_extension_for_pytorch # noqa: F401
|
||||
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||
return "xpu"
|
||||
except ImportError:
|
||||
pass
|
||||
# Any GPU on Windows via DirectML (torch-directml)
|
||||
try:
|
||||
import torch_directml
|
||||
if torch_directml.device_count() > 0:
|
||||
return torch_directml.device(0)
|
||||
except ImportError:
|
||||
pass
|
||||
if hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||
return "cpu" # MPS disabled for stability
|
||||
return "cpu"
|
||||
|
||||
def is_loaded(self) -> bool:
|
||||
|
||||
+13
-27
@@ -5,7 +5,6 @@ PyInstaller build script for creating standalone Python server binary.
|
||||
import PyInstaller.__main__
|
||||
import os
|
||||
import platform
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@@ -14,27 +13,15 @@ def is_apple_silicon():
|
||||
return platform.system() == "Darwin" and platform.machine() == "arm64"
|
||||
|
||||
|
||||
def build_server(variant="cpu"):
|
||||
"""Build Python server as standalone binary.
|
||||
|
||||
Args:
|
||||
variant: 'cpu' for CPU-only build (~500MB) or 'cuda' for CUDA build (~3GB)
|
||||
"""
|
||||
def build_server():
|
||||
"""Build Python server as standalone binary."""
|
||||
backend_dir = Path(__file__).parent
|
||||
|
||||
if variant not in ['cpu', 'cuda']:
|
||||
raise ValueError(f"Invalid variant: {variant}. Must be 'cpu' or 'cuda'")
|
||||
|
||||
# Set binary name based on variant
|
||||
binary_name = f'voicebox-server-{variant}' if variant == 'cuda' else 'voicebox-server'
|
||||
|
||||
print(f"Building {variant.upper()} variant: {binary_name}")
|
||||
|
||||
# PyInstaller arguments
|
||||
args = [
|
||||
'server.py', # Use server.py as entry point instead of main.py
|
||||
'--onefile',
|
||||
'--name', binary_name,
|
||||
'--name', 'voicebox-server',
|
||||
]
|
||||
|
||||
# Add local qwen_tts path if specified (for editable installs)
|
||||
@@ -96,9 +83,13 @@ def build_server(variant="cpu"):
|
||||
'--hidden-import', 'mlx_audio.stt',
|
||||
'--collect-submodules', 'mlx',
|
||||
'--collect-submodules', 'mlx_audio',
|
||||
# Collect MLX data files including Metal shader libraries (.metallib)
|
||||
'--collect-data', 'mlx',
|
||||
'--collect-data', 'mlx_audio',
|
||||
# Use --collect-all so PyInstaller bundles both data files AND
|
||||
# native shared libraries (.dylib, .metallib) for MLX.
|
||||
# Previously only --collect-data was used, which caused MLX to
|
||||
# raise OSError at runtime inside the bundled binary because
|
||||
# the Metal shader libraries were missing.
|
||||
'--collect-all', 'mlx',
|
||||
'--collect-all', 'mlx_audio',
|
||||
])
|
||||
else:
|
||||
print("Building for non-Apple Silicon platform - PyTorch only")
|
||||
@@ -113,14 +104,9 @@ def build_server(variant="cpu"):
|
||||
|
||||
# Run PyInstaller
|
||||
PyInstaller.__main__.run(args)
|
||||
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Build complete: {variant.upper()} variant")
|
||||
print(f"Binary: {backend_dir / 'dist' / binary_name}")
|
||||
print(f"{'='*60}\n")
|
||||
|
||||
print(f"Binary built in {backend_dir / 'dist' / 'voicebox-server'}")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# Accept variant as command line argument
|
||||
variant = sys.argv[1] if len(sys.argv) > 1 else 'cpu'
|
||||
build_server(variant)
|
||||
build_server()
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
@echo off
|
||||
REM Build both CPU and CUDA server binaries for Windows
|
||||
|
||||
echo ============================================================
|
||||
echo Building BOTH server binaries (CPU + CUDA)
|
||||
echo This will take a while...
|
||||
echo ============================================================
|
||||
|
||||
call build_cpu.bat
|
||||
if errorlevel 1 (
|
||||
echo CPU build failed!
|
||||
exit /b 1
|
||||
)
|
||||
|
||||
echo.
|
||||
echo.
|
||||
|
||||
call build_cuda.bat
|
||||
if errorlevel 1 (
|
||||
echo CUDA build failed!
|
||||
exit /b 1
|
||||
)
|
||||
|
||||
echo.
|
||||
echo ============================================================
|
||||
echo Both binaries built successfully!
|
||||
echo ============================================================
|
||||
echo CPU binary: dist\voicebox-server.exe (~500MB)
|
||||
echo CUDA binary: dist\voicebox-server-cuda.exe (~3GB)
|
||||
echo ============================================================
|
||||
@@ -1,28 +0,0 @@
|
||||
@echo off
|
||||
REM Build CPU-only server binary for Windows
|
||||
REM This creates a ~500MB binary without CUDA support
|
||||
|
||||
echo ============================================================
|
||||
echo Building CPU-only server binary
|
||||
echo ============================================================
|
||||
|
||||
echo.
|
||||
echo Step 1: Installing CPU-only PyTorch...
|
||||
pip uninstall -y torch torchvision torchaudio
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
|
||||
|
||||
echo.
|
||||
echo Step 2: Building binary with PyInstaller...
|
||||
python build_binary.py cpu
|
||||
|
||||
echo.
|
||||
echo Step 3: Restoring CUDA PyTorch for development...
|
||||
pip uninstall -y torch torchvision torchaudio
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu121
|
||||
|
||||
echo.
|
||||
echo ============================================================
|
||||
echo CPU binary built successfully!
|
||||
echo Location: dist\voicebox-server.exe
|
||||
echo Size: ~500MB
|
||||
echo ============================================================
|
||||
@@ -1,30 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Build CPU-only server binary
|
||||
# This creates a ~500MB binary without CUDA support
|
||||
|
||||
set -e
|
||||
|
||||
echo "============================================================"
|
||||
echo "Building CPU-only server binary"
|
||||
echo "============================================================"
|
||||
|
||||
echo ""
|
||||
echo "Step 1: Installing CPU-only PyTorch..."
|
||||
pip uninstall -y torch torchvision torchaudio || true
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
|
||||
|
||||
echo ""
|
||||
echo "Step 2: Building binary with PyInstaller..."
|
||||
python build_binary.py cpu
|
||||
|
||||
echo ""
|
||||
echo "Step 3: Restoring CUDA PyTorch for development..."
|
||||
pip uninstall -y torch torchvision torchaudio || true
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu121
|
||||
|
||||
echo ""
|
||||
echo "============================================================"
|
||||
echo "CPU binary built successfully!"
|
||||
echo "Location: dist/voicebox-server"
|
||||
echo "Size: ~500MB"
|
||||
echo "============================================================"
|
||||
@@ -1,22 +0,0 @@
|
||||
@echo off
|
||||
REM Build CUDA server binary for Windows
|
||||
REM This creates a ~3GB binary with CUDA support
|
||||
|
||||
echo ============================================================
|
||||
echo Building CUDA server binary
|
||||
echo ============================================================
|
||||
|
||||
echo.
|
||||
echo Step 1: Ensuring CUDA PyTorch is installed...
|
||||
pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu121 --upgrade
|
||||
|
||||
echo.
|
||||
echo Step 2: Building binary with PyInstaller...
|
||||
python build_binary.py cuda
|
||||
|
||||
echo.
|
||||
echo ============================================================
|
||||
echo CUDA binary built successfully!
|
||||
echo Location: dist\voicebox-server-cuda.exe
|
||||
echo Size: ~3GB
|
||||
echo ============================================================
|
||||
@@ -4,8 +4,17 @@ Configuration module for voicebox backend.
|
||||
Handles data directory configuration for production bundling.
|
||||
"""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Allow users to override the HuggingFace model download directory.
|
||||
# Set VOICEBOX_MODELS_DIR to an absolute path before starting the server.
|
||||
# This sets HF_HUB_CACHE so all huggingface_hub downloads go to that path.
|
||||
_custom_models_dir = os.environ.get("VOICEBOX_MODELS_DIR")
|
||||
if _custom_models_dir:
|
||||
os.environ["HF_HUB_CACHE"] = _custom_models_dir
|
||||
print(f"[config] Model download path set to: {_custom_models_dir}")
|
||||
|
||||
# Default data directory (used in development)
|
||||
_data_dir = Path("data")
|
||||
|
||||
|
||||
+153
-39
@@ -22,6 +22,24 @@ import uuid
|
||||
import asyncio
|
||||
import signal
|
||||
import os
|
||||
from urllib.parse import quote
|
||||
|
||||
|
||||
def _safe_content_disposition(disposition_type: str, filename: str) -> str:
|
||||
"""Build a Content-Disposition header that is safe for non-ASCII filenames.
|
||||
|
||||
Uses RFC 5987 ``filename*`` parameter so that browsers can decode
|
||||
UTF-8 filenames while the ``filename`` fallback stays ASCII-only.
|
||||
"""
|
||||
ascii_name = "".join(
|
||||
c for c in filename if c.isascii() and (c.isalnum() or c in " -_.")
|
||||
).strip() or "download"
|
||||
utf8_name = quote(filename, safe="")
|
||||
return (
|
||||
f'{disposition_type}; filename="{ascii_name}"; '
|
||||
f"filename*=UTF-8''{utf8_name}"
|
||||
)
|
||||
|
||||
|
||||
from . import database, models, profiles, history, tts, transcribe, config, export_import, channels, stories, __version__
|
||||
from .database import get_db, Generation as DBGeneration, VoiceProfile as DBVoiceProfile
|
||||
@@ -77,10 +95,39 @@ async def health():
|
||||
tts_model = tts.get_tts_model()
|
||||
backend_type = get_backend_type()
|
||||
|
||||
# Check for GPU availability (CUDA or MPS)
|
||||
# Check for GPU availability (CUDA, MPS, Intel Arc XPU, or DirectML)
|
||||
has_cuda = torch.cuda.is_available()
|
||||
has_mps = hasattr(torch.backends, 'mps') and torch.backends.mps.is_available()
|
||||
gpu_available = has_cuda or has_mps
|
||||
|
||||
# Intel Arc / Intel Xe via intel-extension-for-pytorch (IPEX)
|
||||
has_xpu = False
|
||||
xpu_name = None
|
||||
try:
|
||||
import intel_extension_for_pytorch as ipex # noqa: F401
|
||||
if hasattr(torch, 'xpu') and torch.xpu.is_available():
|
||||
has_xpu = True
|
||||
try:
|
||||
xpu_name = torch.xpu.get_device_name(0)
|
||||
except Exception:
|
||||
xpu_name = "Intel GPU"
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# DirectML backend (torch-directml) for any Windows GPU
|
||||
has_directml = False
|
||||
directml_name = None
|
||||
try:
|
||||
import torch_directml
|
||||
if torch_directml.device_count() > 0:
|
||||
has_directml = True
|
||||
try:
|
||||
directml_name = torch_directml.device_name(0)
|
||||
except Exception:
|
||||
directml_name = "DirectML GPU"
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
gpu_available = has_cuda or has_mps or has_xpu or has_directml or backend_type == "mlx"
|
||||
|
||||
gpu_type = None
|
||||
if has_cuda:
|
||||
@@ -89,6 +136,10 @@ async def health():
|
||||
gpu_type = "MPS (Apple Silicon)"
|
||||
elif backend_type == "mlx":
|
||||
gpu_type = "Metal (Apple Silicon via MLX)"
|
||||
elif has_xpu:
|
||||
gpu_type = f"XPU ({xpu_name})"
|
||||
elif has_directml:
|
||||
gpu_type = f"DirectML ({directml_name})"
|
||||
|
||||
vram_used = None
|
||||
if has_cuda:
|
||||
@@ -252,12 +303,17 @@ async def add_profile_sample(
|
||||
db: Session = Depends(get_db),
|
||||
):
|
||||
"""Add a sample to a voice profile."""
|
||||
# Save uploaded file to temporary location
|
||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp:
|
||||
# Preserve the uploaded file's extension so librosa can detect format correctly.
|
||||
# Defaulting to .wav was causing soundfile to reject MP3/WebM content as invalid WAV.
|
||||
_allowed_audio_exts = {'.wav', '.mp3', '.m4a', '.ogg', '.flac', '.aac', '.webm', '.opus'}
|
||||
_uploaded_ext = Path(file.filename or '').suffix.lower()
|
||||
file_suffix = _uploaded_ext if _uploaded_ext in _allowed_audio_exts else '.wav'
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=file_suffix, delete=False) as tmp:
|
||||
content = await file.read()
|
||||
tmp.write(content)
|
||||
tmp_path = tmp.name
|
||||
|
||||
|
||||
try:
|
||||
sample = await profiles.add_profile_sample(
|
||||
profile_id,
|
||||
@@ -268,6 +324,8 @@ async def add_profile_sample(
|
||||
return sample
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=f"Failed to process audio file: {str(e)}")
|
||||
finally:
|
||||
# Clean up temp file
|
||||
Path(tmp_path).unlink(missing_ok=True)
|
||||
@@ -388,7 +446,7 @@ async def export_profile(
|
||||
io.BytesIO(zip_bytes),
|
||||
media_type="application/zip",
|
||||
headers={
|
||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
||||
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||
}
|
||||
)
|
||||
except ValueError as e:
|
||||
@@ -542,47 +600,50 @@ async def generate_speech(
|
||||
if not profile:
|
||||
raise HTTPException(status_code=404, detail="Profile not found")
|
||||
|
||||
# Create voice prompt from profile
|
||||
voice_prompt = await profiles.create_voice_prompt_for_profile(
|
||||
data.profile_id,
|
||||
db,
|
||||
)
|
||||
|
||||
# Generate audio
|
||||
|
||||
# Resolve model size and load the correct model FIRST.
|
||||
# This must happen before create_voice_prompt_for_profile because that
|
||||
# function calls load_model_async(None), which falls back to self.model_size.
|
||||
# If the model is already loaded with the right size at that point, it
|
||||
# returns immediately and the voice prompt is created by the correct model.
|
||||
tts_model = tts.get_tts_model()
|
||||
# Load the requested model size if different from current (async to not block)
|
||||
model_size = data.model_size or "1.7B"
|
||||
|
||||
# Check if model needs to be downloaded first
|
||||
model_path = tts_model._get_model_path(model_size)
|
||||
if model_path.startswith("Qwen/"):
|
||||
# Model not cached - check if it exists remotely or needs download
|
||||
from huggingface_hub import constants as hf_constants
|
||||
repo_cache = Path(hf_constants.HF_HUB_CACHE) / ("models--" + model_path.replace("/", "--"))
|
||||
if not repo_cache.exists():
|
||||
# Start download in background
|
||||
model_name = f"qwen-tts-{model_size}"
|
||||
if not tts_model._is_model_cached(model_size):
|
||||
# Model is not fully cached — kick off a background download and tell
|
||||
# the client to retry once it's ready.
|
||||
model_name = f"qwen-tts-{model_size}"
|
||||
|
||||
async def download_model_background():
|
||||
try:
|
||||
await tts_model.load_model_async(model_size)
|
||||
except Exception as e:
|
||||
task_manager.error_download(model_name, str(e))
|
||||
async def download_model_background():
|
||||
try:
|
||||
await tts_model.load_model_async(model_size)
|
||||
except Exception as e:
|
||||
task_manager.error_download(model_name, str(e))
|
||||
|
||||
task_manager.start_download(model_name)
|
||||
asyncio.create_task(download_model_background())
|
||||
task_manager.start_download(model_name)
|
||||
asyncio.create_task(download_model_background())
|
||||
|
||||
# Return 202 Accepted with download info
|
||||
raise HTTPException(
|
||||
status_code=202,
|
||||
detail={
|
||||
"message": f"Model {model_size} is being downloaded. Please wait and try again.",
|
||||
"model_name": model_name,
|
||||
"downloading": True
|
||||
}
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=202,
|
||||
detail={
|
||||
"message": f"Model {model_size} is being downloaded. Please wait and try again.",
|
||||
"model_name": model_name,
|
||||
"downloading": True,
|
||||
},
|
||||
)
|
||||
|
||||
# Load (or switch to) the requested model before building the voice prompt
|
||||
await tts_model.load_model_async(model_size)
|
||||
|
||||
# Create voice prompt from profile (model is already loaded with correct size)
|
||||
voice_prompt = await profiles.create_voice_prompt_for_profile(
|
||||
data.profile_id,
|
||||
db,
|
||||
)
|
||||
|
||||
audio, sample_rate = await tts_model.generate(
|
||||
data.text,
|
||||
voice_prompt,
|
||||
@@ -625,6 +686,59 @@ async def generate_speech(
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
@app.post("/generate/stream")
|
||||
async def stream_speech(
|
||||
data: models.GenerationRequest,
|
||||
db: Session = Depends(get_db),
|
||||
):
|
||||
"""
|
||||
Generate speech and stream the WAV audio directly without saving to disk.
|
||||
|
||||
Returns raw WAV bytes via a StreamingResponse so the client can start
|
||||
playing audio before the entire file has been received. This endpoint
|
||||
does NOT create a history entry — use /generate for that.
|
||||
"""
|
||||
profile = await profiles.get_profile(data.profile_id, db)
|
||||
if not profile:
|
||||
raise HTTPException(status_code=404, detail="Profile not found")
|
||||
|
||||
tts_model = tts.get_tts_model()
|
||||
model_size = data.model_size or "1.7B"
|
||||
|
||||
if not tts_model._is_model_cached(model_size):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Model {model_size} is not downloaded yet. Use /generate to trigger a download.",
|
||||
)
|
||||
|
||||
# Load the correct model before building the voice prompt (fixes issue #96)
|
||||
await tts_model.load_model_async(model_size)
|
||||
|
||||
voice_prompt = await profiles.create_voice_prompt_for_profile(data.profile_id, db)
|
||||
|
||||
audio, sample_rate = await tts_model.generate(
|
||||
data.text,
|
||||
voice_prompt,
|
||||
data.language,
|
||||
data.seed,
|
||||
data.instruct,
|
||||
)
|
||||
|
||||
wav_bytes = tts.audio_to_wav_bytes(audio, sample_rate)
|
||||
|
||||
async def _wav_stream():
|
||||
# Yield in chunks so large responses don't block the event loop
|
||||
chunk_size = 64 * 1024 # 64 KB
|
||||
for i in range(0, len(wav_bytes), chunk_size):
|
||||
yield wav_bytes[i : i + chunk_size]
|
||||
|
||||
return StreamingResponse(
|
||||
_wav_stream(),
|
||||
media_type="audio/wav",
|
||||
headers={"Content-Disposition": 'attachment; filename="speech.wav"'},
|
||||
)
|
||||
|
||||
|
||||
# ============================================
|
||||
# HISTORY ENDPOINTS
|
||||
# ============================================
|
||||
@@ -753,7 +867,7 @@ async def export_generation(
|
||||
io.BytesIO(zip_bytes),
|
||||
media_type="application/zip",
|
||||
headers={
|
||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
||||
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||
}
|
||||
)
|
||||
except ValueError as e:
|
||||
@@ -786,7 +900,7 @@ async def export_generation_audio(
|
||||
audio_path,
|
||||
media_type="audio/wav",
|
||||
headers={
|
||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
||||
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||
}
|
||||
)
|
||||
|
||||
@@ -1054,7 +1168,7 @@ async def export_story_audio(
|
||||
io.BytesIO(audio_bytes),
|
||||
media_type="audio/wav",
|
||||
headers={
|
||||
"Content-Disposition": f'attachment; filename="{filename}"'
|
||||
"Content-Disposition": _safe_content_disposition("attachment", filename)
|
||||
}
|
||||
)
|
||||
except HTTPException:
|
||||
|
||||
@@ -19,15 +19,17 @@ def is_apple_silicon() -> bool:
|
||||
def get_backend_type() -> Literal["mlx", "pytorch"]:
|
||||
"""
|
||||
Detect the best backend for the current platform.
|
||||
|
||||
|
||||
Returns:
|
||||
"mlx" on Apple Silicon (if MLX is available), "pytorch" otherwise
|
||||
"mlx" on Apple Silicon (if MLX is available and functional), "pytorch" otherwise
|
||||
"""
|
||||
if is_apple_silicon():
|
||||
try:
|
||||
import mlx
|
||||
import mlx.core # noqa: F401 — triggers native lib loading
|
||||
return "mlx"
|
||||
except ImportError:
|
||||
# MLX not installed, fallback to PyTorch
|
||||
except (ImportError, OSError, RuntimeError):
|
||||
# MLX not installed, or native libraries failed to load inside a
|
||||
# PyInstaller bundle (OSError on missing .dylib / .metallib).
|
||||
# Fall through to PyTorch.
|
||||
return "pytorch"
|
||||
return "pytorch"
|
||||
|
||||
@@ -18,6 +18,7 @@ qwen-tts>=0.0.5
|
||||
librosa>=0.10.0
|
||||
soundfile>=0.12.0
|
||||
numpy>=1.24.0
|
||||
numba>=0.60.0,<0.61.0
|
||||
|
||||
# Utilities
|
||||
python-multipart>=0.0.6
|
||||
|
||||
@@ -1,137 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test CUDA binary compression to verify it fits under GitHub's 2GB release asset limit.
|
||||
|
||||
Usage:
|
||||
python test_cuda_compression.py [path/to/voicebox-server-cuda.exe]
|
||||
|
||||
If no path provided, looks for the binary in ./dist/
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def format_size(bytes_size):
|
||||
"""Format bytes into human-readable size."""
|
||||
for unit in ['B', 'KB', 'MB', 'GB']:
|
||||
if bytes_size < 1024.0:
|
||||
return f"{bytes_size:.2f} {unit}"
|
||||
bytes_size /= 1024.0
|
||||
return f"{bytes_size:.2f} TB"
|
||||
|
||||
|
||||
def get_file_size(filepath):
|
||||
"""Get file size in bytes."""
|
||||
return os.path.getsize(filepath)
|
||||
|
||||
|
||||
def compress_with_7z(input_file, output_file):
|
||||
"""Compress file using 7z with maximum compression."""
|
||||
print(f"\nCompressing with 7z (maximum compression)...")
|
||||
print(f"This may take several minutes for a ~2.5GB file...\n")
|
||||
|
||||
cmd = [
|
||||
'7z', 'a',
|
||||
'-t7z', # 7z format
|
||||
'-m0=lzma2', # LZMA2 compression
|
||||
'-mx=9', # Maximum compression
|
||||
'-mfb=64', # Fast bytes
|
||||
'-md=32m', # Dictionary size
|
||||
'-ms=on', # Solid archive
|
||||
output_file,
|
||||
input_file
|
||||
]
|
||||
|
||||
try:
|
||||
subprocess.run(cmd, check=True, capture_output=True, text=True)
|
||||
return True
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error during compression: {e}")
|
||||
print(f"stderr: {e.stderr}")
|
||||
return False
|
||||
except FileNotFoundError:
|
||||
print("ERROR: 7z not found. Please install 7-Zip:")
|
||||
print(" Windows: https://www.7-zip.org/download.html")
|
||||
print(" macOS: brew install p7zip")
|
||||
print(" Linux: apt-get install p7zip-full")
|
||||
return False
|
||||
|
||||
|
||||
def main():
|
||||
# Find CUDA binary
|
||||
if len(sys.argv) > 1:
|
||||
cuda_binary = Path(sys.argv[1])
|
||||
else:
|
||||
# Look in dist directory
|
||||
dist_dir = Path(__file__).parent / 'dist'
|
||||
candidates = list(dist_dir.glob('voicebox-server-cuda*.exe'))
|
||||
|
||||
if not candidates:
|
||||
print("ERROR: CUDA binary not found in ./dist/")
|
||||
print("Please provide the path as an argument:")
|
||||
print(" python test_cuda_compression.py path/to/voicebox-server-cuda.exe")
|
||||
sys.exit(1)
|
||||
|
||||
cuda_binary = candidates[0]
|
||||
|
||||
if not cuda_binary.exists():
|
||||
print(f"ERROR: File not found: {cuda_binary}")
|
||||
sys.exit(1)
|
||||
|
||||
print("=" * 70)
|
||||
print("CUDA Binary Compression Test")
|
||||
print("=" * 70)
|
||||
|
||||
# Get original size
|
||||
original_size = get_file_size(cuda_binary)
|
||||
print(f"\nOriginal file: {cuda_binary.name}")
|
||||
print(f"Original size: {format_size(original_size)} ({original_size:,} bytes)")
|
||||
|
||||
# Check if already over 2GB
|
||||
github_limit = 2 * 1024 * 1024 * 1024 # 2GB in bytes
|
||||
print(f"GitHub limit: {format_size(github_limit)} ({github_limit:,} bytes)")
|
||||
|
||||
if original_size > github_limit:
|
||||
print(f"\n[WARNING] Original file exceeds GitHub limit by {format_size(original_size - github_limit)}")
|
||||
else:
|
||||
print(f"\n[OK] Original file is under GitHub limit")
|
||||
|
||||
# Compress
|
||||
output_file = cuda_binary.parent / f"{cuda_binary.stem}.7z"
|
||||
if output_file.exists():
|
||||
print(f"\nRemoving existing compressed file: {output_file.name}")
|
||||
output_file.unlink()
|
||||
|
||||
success = compress_with_7z(cuda_binary, output_file)
|
||||
|
||||
if not success:
|
||||
sys.exit(1)
|
||||
|
||||
# Check compressed size
|
||||
compressed_size = get_file_size(output_file)
|
||||
compression_ratio = (1 - compressed_size / original_size) * 100
|
||||
|
||||
print("\n" + "=" * 70)
|
||||
print("Compression Results")
|
||||
print("=" * 70)
|
||||
print(f"\nCompressed file: {output_file.name}")
|
||||
print(f"Compressed size: {format_size(compressed_size)} ({compressed_size:,} bytes)")
|
||||
print(f"Compression ratio: {compression_ratio:.1f}%")
|
||||
print(f"Space saved: {format_size(original_size - compressed_size)}")
|
||||
|
||||
if compressed_size <= github_limit:
|
||||
print(f"\n[SUCCESS] Compressed file fits under GitHub's 2GB limit!")
|
||||
print(f" Margin: {format_size(github_limit - compressed_size)} remaining")
|
||||
else:
|
||||
print(f"\n[FAILED] Compressed file still exceeds GitHub limit")
|
||||
print(f" Over by: {format_size(compressed_size - github_limit)}")
|
||||
print(f"\n Alternative: Host on external storage (S3, Azure Blob, etc.)")
|
||||
|
||||
print("\n" + "=" * 70)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -1,86 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Test R2 upload locally before running in CI
|
||||
|
||||
set -e
|
||||
|
||||
echo "============================================================"
|
||||
echo "Cloudflare R2 Upload Test"
|
||||
echo "============================================================"
|
||||
|
||||
# Check for required environment variables
|
||||
if [ -z "$AWS_ACCESS_KEY_ID" ] || [ -z "$AWS_SECRET_ACCESS_KEY" ] || [ -z "$R2_ENDPOINT" ]; then
|
||||
echo "ERROR: Missing required environment variables"
|
||||
echo ""
|
||||
echo "Please set:"
|
||||
echo " export AWS_ACCESS_KEY_ID='your-r2-access-key-id'"
|
||||
echo " export AWS_SECRET_ACCESS_KEY='your-r2-secret-access-key'"
|
||||
echo " export R2_ENDPOINT='https://your-account-id.r2.cloudflarestorage.com'"
|
||||
echo ""
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check for AWS CLI
|
||||
if ! command -v aws &> /dev/null; then
|
||||
echo "Installing AWS CLI..."
|
||||
pip install awscli
|
||||
fi
|
||||
|
||||
# Find CUDA binary
|
||||
CUDA_BINARY=$(ls dist/voicebox-server-cuda*.exe 2>/dev/null | head -1)
|
||||
|
||||
if [ -z "$CUDA_BINARY" ]; then
|
||||
echo "ERROR: CUDA binary not found in dist/"
|
||||
echo "Run: bash build_cuda.bat"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Found CUDA binary: $CUDA_BINARY"
|
||||
echo "Size: $(du -h "$CUDA_BINARY" | cut -f1)"
|
||||
echo ""
|
||||
|
||||
# Test version
|
||||
VERSION="v0.1.12-test"
|
||||
PLATFORM="x86_64-pc-windows-msvc"
|
||||
FILENAME="voicebox-server-cuda-${PLATFORM}.exe"
|
||||
|
||||
echo "Test upload configuration:"
|
||||
echo " Version: $VERSION"
|
||||
echo " Platform: $PLATFORM"
|
||||
echo " Endpoint: $R2_ENDPOINT"
|
||||
echo " Bucket: voicebox"
|
||||
echo " Path: cuda/$VERSION/$FILENAME"
|
||||
echo ""
|
||||
|
||||
read -p "Proceed with upload? (y/n) " -n 1 -r
|
||||
echo
|
||||
if [[ ! $REPLY =~ ^[Yy]$ ]]; then
|
||||
echo "Aborted."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Uploading to R2..."
|
||||
|
||||
aws s3 cp "$CUDA_BINARY" \
|
||||
"s3://voicebox/cuda/${VERSION}/${FILENAME}" \
|
||||
--endpoint-url "$R2_ENDPOINT" \
|
||||
--acl public-read
|
||||
|
||||
if [ $? -eq 0 ]; then
|
||||
echo ""
|
||||
echo "============================================================"
|
||||
echo "Upload successful!"
|
||||
echo "============================================================"
|
||||
echo ""
|
||||
echo "Download URL:"
|
||||
echo "https://downloads.voicebox.sh/cuda/${VERSION}/${FILENAME}"
|
||||
echo ""
|
||||
echo "Test with:"
|
||||
echo "curl -I https://downloads.voicebox.sh/cuda/${VERSION}/${FILENAME}"
|
||||
echo ""
|
||||
else
|
||||
echo ""
|
||||
echo "Upload failed!"
|
||||
exit 1
|
||||
fi
|
||||
@@ -32,11 +32,3 @@ def audio_to_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes:
|
||||
sf.write(buffer, audio, sample_rate, format="WAV")
|
||||
buffer.seek(0)
|
||||
return buffer.read()
|
||||
|
||||
|
||||
def audio_to_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes:
|
||||
"""Convert audio array to WAV bytes."""
|
||||
buffer = io.BytesIO()
|
||||
sf.write(buffer, audio, sample_rate, format="WAV")
|
||||
buffer.seek(0)
|
||||
return buffer.read()
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
# -*- mode: python ; coding: utf-8 -*-
|
||||
from PyInstaller.utils.hooks import collect_data_files
|
||||
from PyInstaller.utils.hooks import collect_submodules
|
||||
from PyInstaller.utils.hooks import copy_metadata
|
||||
|
||||
datas = []
|
||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern']
|
||||
datas += collect_data_files('qwen_tts')
|
||||
datas += copy_metadata('qwen-tts')
|
||||
hiddenimports += collect_submodules('qwen_tts')
|
||||
hiddenimports += collect_submodules('jaraco')
|
||||
|
||||
|
||||
a = Analysis(
|
||||
['server.py'],
|
||||
pathex=[],
|
||||
binaries=[],
|
||||
datas=datas,
|
||||
hiddenimports=hiddenimports,
|
||||
hookspath=[],
|
||||
hooksconfig={},
|
||||
runtime_hooks=[],
|
||||
excludes=[],
|
||||
noarchive=False,
|
||||
optimize=0,
|
||||
)
|
||||
pyz = PYZ(a.pure)
|
||||
|
||||
exe = EXE(
|
||||
pyz,
|
||||
a.scripts,
|
||||
a.binaries,
|
||||
a.datas,
|
||||
[],
|
||||
name='voicebox-server-cuda',
|
||||
debug=False,
|
||||
bootloader_ignore_signals=False,
|
||||
strip=False,
|
||||
upx=True,
|
||||
upx_exclude=[],
|
||||
runtime_tmpdir=None,
|
||||
console=True,
|
||||
disable_windowed_traceback=False,
|
||||
argv_emulation=False,
|
||||
target_arch=None,
|
||||
codesign_identity=None,
|
||||
entitlements_file=None,
|
||||
)
|
||||
@@ -4,17 +4,26 @@ from PyInstaller.utils.hooks import collect_submodules
|
||||
from PyInstaller.utils.hooks import copy_metadata
|
||||
|
||||
datas = []
|
||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern']
|
||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
||||
datas += collect_data_files('qwen_tts')
|
||||
# Use collect_all (not collect_data_files) so native .dylib and .metallib
|
||||
# files are bundled as binaries, not data. Without this, MLX raises OSError
|
||||
# when loading Metal shaders inside the PyInstaller bundle.
|
||||
from PyInstaller.utils.hooks import collect_all as _collect_all
|
||||
_mlx_datas, _mlx_bins, _mlx_hidden = _collect_all('mlx')
|
||||
_mlxa_datas, _mlxa_bins, _mlxa_hidden = _collect_all('mlx_audio')
|
||||
datas += _mlx_datas + _mlxa_datas
|
||||
datas += copy_metadata('qwen-tts')
|
||||
hiddenimports += collect_submodules('qwen_tts')
|
||||
hiddenimports += collect_submodules('jaraco')
|
||||
hiddenimports += collect_submodules('mlx')
|
||||
hiddenimports += collect_submodules('mlx_audio')
|
||||
|
||||
|
||||
a = Analysis(
|
||||
['server.py'],
|
||||
pathex=[],
|
||||
binaries=[],
|
||||
binaries=_mlx_bins + _mlxa_bins,
|
||||
datas=datas,
|
||||
hiddenimports=hiddenimports,
|
||||
hookspath=[],
|
||||
|
||||
@@ -1,620 +0,0 @@
|
||||
# CUDA Distribution Problem - Complete Analysis
|
||||
|
||||
## Table of Contents
|
||||
1. [Problem Overview](#problem-overview)
|
||||
2. [Root Cause](#root-cause)
|
||||
3. [Attempted Solutions](#attempted-solutions)
|
||||
4. [Current Status](#current-status)
|
||||
5. [Available Options](#available-options)
|
||||
6. [Technical Details](#technical-details)
|
||||
7. [Cost Analysis](#cost-analysis)
|
||||
8. [Recommendations](#recommendations)
|
||||
|
||||
---
|
||||
|
||||
## Problem Overview
|
||||
|
||||
### Timeline of Issues
|
||||
|
||||
**Original Problem (v0.1.0 - v0.1.11)**
|
||||
- Single server binary with CUDA support
|
||||
- Size: ~2.9GB
|
||||
- Issue: MSI installer build fails in GitHub Actions CI
|
||||
- Error: WiX Toolset cannot handle 3GB files efficiently
|
||||
|
||||
**First Solution: Dual Binary System (v0.1.12)**
|
||||
- Split into CPU (295MB) and CUDA (2.37GB) binaries
|
||||
- CPU ships with installer
|
||||
- CUDA as optional download
|
||||
- Issue: GitHub Release assets have 2GB limit
|
||||
|
||||
**Current Problem (Discovered during implementation)**
|
||||
- GitHub Release Asset Limit: **2GB hard maximum**
|
||||
- CUDA binary: **2.37GB** (370MB over limit)
|
||||
- Cannot upload to GitHub Releases
|
||||
|
||||
---
|
||||
|
||||
## Root Cause
|
||||
|
||||
### Why Is The CUDA Binary So Large?
|
||||
|
||||
The size difference between CPU and CUDA builds:
|
||||
|
||||
| Component | CPU Build | CUDA Build | Difference |
|
||||
|-----------|-----------|------------|------------|
|
||||
| PyTorch Core | ~150MB | ~150MB | - |
|
||||
| CPU Libraries (MKL/OpenBLAS) | ~100MB | - | -100MB |
|
||||
| CUDA Runtime | - | ~500MB | +500MB |
|
||||
| cuBLAS | - | ~350MB | +350MB |
|
||||
| cuDNN | - | ~1.2GB | +1.2GB |
|
||||
| NVRTC (CUDA Compiler) | - | ~90MB | +90MB |
|
||||
| Other CUDA libs | - | ~100MB | +100MB |
|
||||
| **Total** | **~295MB** | **~2.37GB** | **+2.07GB** |
|
||||
|
||||
### CUDA Dependencies Breakdown
|
||||
|
||||
```
|
||||
torch/lib/ (CUDA build):
|
||||
├── cudart64_12.dll (~0.5 MB) - CUDA Runtime
|
||||
├── cublas64_12.dll (~100 MB) - Basic Linear Algebra
|
||||
├── cublasLt64_12.dll (~200 MB) - Linear Algebra (optimized)
|
||||
├── cudnn64_9.dll (~800 MB) - Deep Neural Networks
|
||||
├── cudnn_*_infer64_9.dll (~400 MB) - DNN Inference ops
|
||||
├── nvrtc64_*.dll (~50 MB) - Runtime Compiler
|
||||
├── nvrtc-builtins64_*.dll (~40 MB) - Compiler builtins
|
||||
├── torch_cuda.dll (~200 MB) - PyTorch CUDA bridge
|
||||
└── c10_cuda.dll (~20 MB) - Core CUDA utilities
|
||||
```
|
||||
|
||||
**Why These Are Required:**
|
||||
- cuDNN is essential for neural network operations
|
||||
- cuBLAS handles all matrix operations (core of ML)
|
||||
- Cannot split or remove without breaking functionality
|
||||
|
||||
---
|
||||
|
||||
## Attempted Solutions
|
||||
|
||||
### Solution 1: Dual Binary System ✅ (Partially Successful)
|
||||
|
||||
**Goal**: Split CPU and CUDA into separate downloads
|
||||
|
||||
**Implementation**:
|
||||
```bash
|
||||
# Build CPU-only (295MB)
|
||||
pip install torch --index-url https://download.pytorch.org/whl/cpu
|
||||
python build_binary.py cpu
|
||||
|
||||
# Build CUDA (2.37GB)
|
||||
pip install torch --index-url https://download.pytorch.org/whl/cu121
|
||||
python build_binary.py cuda
|
||||
```
|
||||
|
||||
**Results**:
|
||||
- ✅ CPU binary: 295MB (fits in installer)
|
||||
- ✅ CI builds successfully
|
||||
- ✅ Installer size reduced from 3GB to ~500MB
|
||||
- ❌ CUDA binary still too large for GitHub
|
||||
|
||||
**See**: `docs/dual-server-binaries.md`
|
||||
|
||||
### Solution 2: Compression Testing ❌ (Failed)
|
||||
|
||||
**Goal**: Compress CUDA binary to fit under 2GB
|
||||
|
||||
**Method**: 7z with maximum compression settings
|
||||
```bash
|
||||
7z a -t7z -m0=lzma2 -mx=9 -mfb=64 -md=32m -ms=on \
|
||||
voicebox-server-cuda.7z voicebox-server-cuda.exe
|
||||
```
|
||||
|
||||
**Results**:
|
||||
```
|
||||
Original: 2.37 GB (2,545,086,396 bytes)
|
||||
Compressed: 2.35 GB (2,519,381,264 bytes)
|
||||
Compression: 1.0% (only 24.5MB saved)
|
||||
GitHub Limit: 2.00 GB (2,147,483,648 bytes)
|
||||
Over by: 354.67 MB
|
||||
|
||||
Status: FAILED - Still exceeds limit by 354MB
|
||||
```
|
||||
|
||||
**Why Compression Failed**:
|
||||
- CUDA binaries are already optimized machine code
|
||||
- No redundant data to compress
|
||||
- Neural network kernels are highly compact
|
||||
- Libraries are already stripped of debug symbols
|
||||
|
||||
**Conclusion**: Compression is not viable
|
||||
|
||||
---
|
||||
|
||||
## Current Status
|
||||
|
||||
### What Works
|
||||
- ✅ CPU binary builds successfully (295MB)
|
||||
- ✅ CUDA binary builds successfully (2.37GB)
|
||||
- ✅ Build scripts for both variants
|
||||
- ✅ CI workflow updated for dual binaries
|
||||
- ✅ Installer can be created with CPU binary
|
||||
|
||||
### What Doesn't Work
|
||||
- ❌ Cannot upload CUDA binary to GitHub Releases (exceeds 2GB limit)
|
||||
- ❌ Compression doesn't reduce size enough
|
||||
- ❌ No automated distribution path for CUDA binary
|
||||
|
||||
### Branch Status
|
||||
- Branch: `feat/dual-server-binaries`
|
||||
- Commits: Implementation complete
|
||||
- Testing: Local builds successful
|
||||
- Blocker: CUDA distribution path
|
||||
|
||||
---
|
||||
|
||||
## Available Options
|
||||
|
||||
### Option 1: AWS S3 Hosting (Recommended)
|
||||
|
||||
**Description**: Host CUDA binary in Amazon S3 bucket
|
||||
|
||||
**Pros**:
|
||||
- ✅ No file size limits (can handle multi-GB files)
|
||||
- ✅ Fast global CDN (CloudFront)
|
||||
- ✅ Reliable (99.99% uptime)
|
||||
- ✅ Pay only for usage
|
||||
- ✅ Easy CI integration
|
||||
- ✅ Version control (keep multiple releases)
|
||||
|
||||
**Cons**:
|
||||
- ❌ Requires AWS account
|
||||
- ❌ Monthly costs (~$1-5/month)
|
||||
- ❌ Additional infrastructure to manage
|
||||
|
||||
**Cost Estimate**:
|
||||
```
|
||||
Storage: 2.37 GB × $0.023/GB = $0.05/month
|
||||
Transfer: 100 downloads × 2.37GB × $0.09/GB = $21.33/month
|
||||
Total: ~$21-25/month for 100 downloads
|
||||
~$2-5/month for 10-20 downloads
|
||||
```
|
||||
|
||||
**Implementation**:
|
||||
```yaml
|
||||
# .github/workflows/release.yml
|
||||
- name: Upload CUDA to S3
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
|
||||
run: |
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
s3://voicebox-releases/cuda/${{ github.ref_name }}/ \
|
||||
--acl public-read
|
||||
|
||||
# Generate download URL
|
||||
echo "CUDA_URL=https://voicebox-releases.s3.amazonaws.com/cuda/${{ github.ref_name }}/voicebox-server-cuda-x86_64-pc-windows-msvc.exe" >> release_notes.txt
|
||||
```
|
||||
|
||||
**User Experience**:
|
||||
1. Install app normally (500MB installer)
|
||||
2. App detects NVIDIA GPU
|
||||
3. Shows: "Download CUDA support? (2.4GB)"
|
||||
4. Downloads from S3: `https://voicebox-releases.s3.amazonaws.com/cuda/v0.1.12/voicebox-server-cuda.exe`
|
||||
5. Saves to `%APPDATA%/voicebox/binaries/`
|
||||
6. App restarts with CUDA server
|
||||
|
||||
---
|
||||
|
||||
### Option 2: Azure Blob Storage
|
||||
|
||||
**Description**: Microsoft Azure alternative to S3
|
||||
|
||||
**Pros**:
|
||||
- ✅ Similar to S3 (no size limits, CDN, reliable)
|
||||
- ✅ Good if already using Azure
|
||||
- ✅ Competitive pricing
|
||||
- ✅ Global CDN with Azure CDN
|
||||
|
||||
**Cons**:
|
||||
- ❌ Requires Azure account
|
||||
- ❌ Similar monthly costs
|
||||
- ❌ Less common in open source projects
|
||||
|
||||
**Cost Estimate**:
|
||||
```
|
||||
Storage: $0.018/GB = $0.04/month
|
||||
Transfer: ~$20-25/month for 100 downloads
|
||||
```
|
||||
|
||||
**Implementation**:
|
||||
```yaml
|
||||
- name: Upload to Azure Blob
|
||||
env:
|
||||
AZURE_STORAGE_CONNECTION_STRING: ${{ secrets.AZURE_STORAGE }}
|
||||
run: |
|
||||
az storage blob upload \
|
||||
--account-name voiceboxreleases \
|
||||
--container-name cuda-binaries \
|
||||
--name v${{ github.ref_name }}/voicebox-server-cuda.exe \
|
||||
--file backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
--tier Hot
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Option 3: Cloudflare R2
|
||||
|
||||
**Description**: Cloudflare's S3-compatible object storage
|
||||
|
||||
**Pros**:
|
||||
- ✅ S3-compatible API
|
||||
- ✅ **FREE egress (no bandwidth charges!)**
|
||||
- ✅ Cheaper than S3/Azure
|
||||
- ✅ Cloudflare CDN included
|
||||
- ✅ Good for open source projects
|
||||
|
||||
**Cons**:
|
||||
- ❌ Requires Cloudflare account
|
||||
- ❌ Newer service (less mature than S3)
|
||||
|
||||
**Cost Estimate**:
|
||||
```
|
||||
Storage: $0.015/GB = $0.04/month
|
||||
Egress: $0.00 (FREE!)
|
||||
Class A ops: Negligible
|
||||
Total: ~$0.04/month (essentially free!)
|
||||
```
|
||||
|
||||
**Why This Is Attractive**:
|
||||
- Zero bandwidth costs (huge savings)
|
||||
- Perfect for open source distribution
|
||||
- S3-compatible (easy migration if needed)
|
||||
|
||||
**Implementation**:
|
||||
Same as S3 (R2 is S3-compatible):
|
||||
```yaml
|
||||
- name: Upload to R2
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
|
||||
AWS_ENDPOINT_URL: https://<account-id>.r2.cloudflarestorage.com
|
||||
run: |
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
s3://voicebox-releases/cuda/${{ github.ref_name }}/ \
|
||||
--endpoint-url=$AWS_ENDPOINT_URL
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Option 4: GitHub Packages (Container Registry)
|
||||
|
||||
**Description**: Package CUDA binary as OCI/Docker artifact
|
||||
|
||||
**Pros**:
|
||||
- ✅ Stays in GitHub ecosystem
|
||||
- ✅ No additional accounts needed
|
||||
- ✅ Free for public repos
|
||||
|
||||
**Cons**:
|
||||
- ❌ Complex for desktop app distribution
|
||||
- ❌ Users need to extract from container
|
||||
- ❌ Awkward UX (not designed for binary distribution)
|
||||
- ❌ Requires Docker understanding
|
||||
|
||||
**Not Recommended**: Containers aren't designed for desktop app binaries
|
||||
|
||||
---
|
||||
|
||||
### Option 5: Self-Hosted Server
|
||||
|
||||
**Description**: Host on your own VPS/server
|
||||
|
||||
**Pros**:
|
||||
- ✅ Full control
|
||||
- ✅ No cloud provider dependency
|
||||
- ✅ Predictable costs
|
||||
|
||||
**Cons**:
|
||||
- ❌ Requires server maintenance
|
||||
- ❌ Bandwidth costs can be high
|
||||
- ❌ Uptime responsibility
|
||||
- ❌ Scaling challenges
|
||||
|
||||
**Cost Estimate**:
|
||||
```
|
||||
VPS: $5-20/month (DigitalOcean, Linode)
|
||||
Bandwidth: $0.01-0.02/GB
|
||||
Total: $10-50/month depending on traffic
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Option 6: Manual Distribution
|
||||
|
||||
**Description**: Don't automate - provide manual download instructions
|
||||
|
||||
**Pros**:
|
||||
- ✅ Zero cost
|
||||
- ✅ Zero infrastructure
|
||||
- ✅ Simple
|
||||
|
||||
**Cons**:
|
||||
- ❌ Poor user experience
|
||||
- ❌ Manual upload to file host each release
|
||||
- ❌ Users must manually download and install
|
||||
- ❌ No automatic updates for CUDA binary
|
||||
- ❌ Increases support burden
|
||||
|
||||
**Implementation**:
|
||||
```
|
||||
Release notes:
|
||||
"Windows users with NVIDIA GPUs can download CUDA support:
|
||||
1. Download voicebox-server-cuda.exe from [Google Drive/Mega/etc]
|
||||
2. Place in C:\Users\<YourName>\AppData\Roaming\voicebox\binaries\
|
||||
3. Restart the app"
|
||||
```
|
||||
|
||||
**Not Recommended**: Creates friction, support issues
|
||||
|
||||
---
|
||||
|
||||
### Option 7: Split CUDA Binary
|
||||
|
||||
**Description**: Break CUDA binary into multiple <2GB chunks
|
||||
|
||||
**Technical Approach**:
|
||||
```python
|
||||
# Split binary
|
||||
split -b 2000M voicebox-server-cuda.exe cuda_part_
|
||||
|
||||
# Upload parts to GitHub (each <2GB)
|
||||
cuda_part_aa (2.0 GB)
|
||||
cuda_part_ab (0.37 GB)
|
||||
|
||||
# App downloads and reassembles
|
||||
cat cuda_part_* > voicebox-server-cuda.exe
|
||||
```
|
||||
|
||||
**Pros**:
|
||||
- ✅ Stays on GitHub
|
||||
- ✅ No external hosting
|
||||
|
||||
**Cons**:
|
||||
- ❌ Complex download logic (multiple files)
|
||||
- ❌ Integrity checking required
|
||||
- ❌ More points of failure
|
||||
- ❌ Users must wait for multiple downloads
|
||||
- ❌ Still hacky solution
|
||||
|
||||
**Complexity**: Medium-High
|
||||
|
||||
---
|
||||
|
||||
## Technical Details
|
||||
|
||||
### Current Build Output
|
||||
|
||||
```
|
||||
backend/dist/
|
||||
├── voicebox-server.exe 295 MB (CPU-only)
|
||||
└── voicebox-server-cuda.exe 2.37 GB (CUDA)
|
||||
|
||||
# After compression test:
|
||||
backend/dist/
|
||||
└── voicebox-server-cuda.7z 2.35 GB (not viable)
|
||||
```
|
||||
|
||||
### CI Workflow Changes Required
|
||||
|
||||
For external hosting (S3/R2/Azure):
|
||||
|
||||
```yaml
|
||||
# Current workflow (fails)
|
||||
- name: Upload CUDA server binary (Windows only)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
uses: softprops/action-gh-release@v1
|
||||
with:
|
||||
files: backend/cuda-release/voicebox-server-cuda-*.exe # ❌ Fails: >2GB
|
||||
draft: true
|
||||
|
||||
# New workflow (S3 example)
|
||||
- name: Upload CUDA to S3 (Windows only)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
|
||||
run: |
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
s3://voicebox-releases/cuda/${{ github.ref_name }}/ \
|
||||
--acl public-read
|
||||
|
||||
# Generate release notes with download URL
|
||||
cat >> release_notes.md <<EOF
|
||||
|
||||
### GPU Acceleration (Windows)
|
||||
Download CUDA support for NVIDIA GPUs:
|
||||
[voicebox-server-cuda.exe](https://voicebox-releases.s3.amazonaws.com/cuda/${{ github.ref_name }}/voicebox-server-cuda-x86_64-pc-windows-msvc.exe)
|
||||
Size: 2.37 GB
|
||||
EOF
|
||||
```
|
||||
|
||||
### App Changes Required
|
||||
|
||||
**Frontend (Tauri)**: Download manager
|
||||
```typescript
|
||||
// src/lib/cuda-downloader.ts
|
||||
const CUDA_DOWNLOAD_URL =
|
||||
"https://voicebox-releases.s3.amazonaws.com/cuda/v{VERSION}/voicebox-server-cuda.exe";
|
||||
|
||||
async function downloadCudaBinary(version: string) {
|
||||
const url = CUDA_DOWNLOAD_URL.replace("{VERSION}", version);
|
||||
const savePath = path.join(app.getPath("userData"), "binaries", "voicebox-server-cuda.exe");
|
||||
|
||||
// Download with progress
|
||||
await downloadFile(url, savePath, (progress) => {
|
||||
// Update UI: "Downloading CUDA support: 45% (1.2GB / 2.4GB)"
|
||||
});
|
||||
|
||||
// Verify checksum
|
||||
const checksum = await calculateChecksum(savePath);
|
||||
if (checksum !== EXPECTED_CHECKSUM) {
|
||||
throw new Error("Download corrupted");
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**Backend**: Already supports both binaries (no changes needed)
|
||||
|
||||
---
|
||||
|
||||
## Cost Analysis
|
||||
|
||||
### Monthly Cost Comparison (100 downloads/month)
|
||||
|
||||
| Option | Storage | Bandwidth | Total/Month | Notes |
|
||||
|--------|---------|-----------|-------------|-------|
|
||||
| **Cloudflare R2** | $0.04 | $0.00 | **$0.04** | Best for open source |
|
||||
| AWS S3 | $0.05 | $21.33 | $21.38 | Good reliability |
|
||||
| Azure Blob | $0.04 | $20.00 | $20.04 | Azure ecosystem |
|
||||
| Self-hosted VPS | $10.00 | $2.37 | $12.37 | Maintenance overhead |
|
||||
| Manual | $0.00 | $0.00 | $0.00 | Poor UX |
|
||||
|
||||
### Annual Cost Comparison
|
||||
|
||||
| Option | Year 1 | Year 2+ | Notes |
|
||||
|--------|--------|---------|-------|
|
||||
| **Cloudflare R2** | **$0.50** | **$0.50** | Essentially free |
|
||||
| AWS S3 | $256 | $256 | Predictable |
|
||||
| Self-hosted | $144 | $144 | Time cost |
|
||||
|
||||
**Recommendation**: Cloudflare R2 (free egress = huge savings)
|
||||
|
||||
---
|
||||
|
||||
## Recommendations
|
||||
|
||||
### Recommended Solution: Cloudflare R2
|
||||
|
||||
**Why**:
|
||||
1. **Cost**: Essentially free (~$0.04/month)
|
||||
2. **Bandwidth**: Zero egress charges (unlimited downloads)
|
||||
3. **CDN**: Cloudflare's global network included
|
||||
4. **Compatibility**: S3-compatible API (easy to use)
|
||||
5. **Perfect for open source**: No surprise bandwidth bills
|
||||
|
||||
### Implementation Priority
|
||||
|
||||
**Phase 1: Setup (1-2 hours)**
|
||||
1. Create Cloudflare R2 account
|
||||
2. Create bucket: `voicebox-releases`
|
||||
3. Generate API credentials
|
||||
4. Add to GitHub Secrets
|
||||
|
||||
**Phase 2: CI Integration (1-2 hours)**
|
||||
1. Update `.github/workflows/release.yml`
|
||||
2. Add R2 upload step
|
||||
3. Generate release notes with download URL
|
||||
4. Test with draft release
|
||||
|
||||
**Phase 3: App Integration (4-6 hours)**
|
||||
1. Add GPU detection on startup
|
||||
2. Implement download manager UI
|
||||
3. Add progress indicators
|
||||
4. Implement checksum verification
|
||||
5. Server restart logic
|
||||
|
||||
**Phase 4: Documentation (1 hour)**
|
||||
1. Update README with GPU instructions
|
||||
2. Add troubleshooting guide
|
||||
3. Document manual download process
|
||||
|
||||
**Total Time**: ~8-12 hours of development
|
||||
|
||||
### Alternative: AWS S3 (If Already Using AWS)
|
||||
|
||||
If you're already using AWS for other infrastructure, S3 is also a solid choice:
|
||||
- More mature than R2
|
||||
- Extensive documentation
|
||||
- Familiar tooling
|
||||
- ~$20/month for moderate usage
|
||||
|
||||
---
|
||||
|
||||
## Open Questions
|
||||
|
||||
1. **Expected Download Volume**: How many CUDA downloads per month?
|
||||
- Affects cost calculations
|
||||
- Determines if R2's free egress is significant
|
||||
|
||||
2. **Update Strategy**: How to handle CUDA updates?
|
||||
- Option A: Version in URL path (keep all versions)
|
||||
- Option B: Overwrite latest (save space)
|
||||
|
||||
3. **Fallback Strategy**: What if cloud provider is down?
|
||||
- Mirror on multiple providers?
|
||||
- Graceful degradation to CPU?
|
||||
|
||||
4. **Telemetry**: Track CUDA download stats?
|
||||
- Helps with cost forecasting
|
||||
- User behavior insights
|
||||
|
||||
---
|
||||
|
||||
## Next Steps
|
||||
|
||||
1. **Research Phase** (You are here)
|
||||
- Evaluate cloud providers
|
||||
- Check terms of service
|
||||
- Test account creation
|
||||
|
||||
2. **Decision Phase**
|
||||
- Choose provider (Cloudflare R2 recommended)
|
||||
- Set up account
|
||||
- Configure billing alerts
|
||||
|
||||
3. **Implementation Phase**
|
||||
- Update CI workflow
|
||||
- Implement download manager
|
||||
- Test end-to-end flow
|
||||
|
||||
4. **Launch Phase**
|
||||
- Deploy to production
|
||||
- Monitor downloads
|
||||
- Gather user feedback
|
||||
|
||||
---
|
||||
|
||||
## References
|
||||
|
||||
- **GitHub Release Limits**: https://docs.github.com/en/repositories/releasing-projects-on-github/about-releases
|
||||
- **Cloudflare R2 Pricing**: https://developers.cloudflare.com/r2/pricing/
|
||||
- **AWS S3 Pricing**: https://aws.amazon.com/s3/pricing/
|
||||
- **Compression Test Results**: `backend/test_cuda_compression.py`
|
||||
- **Dual Binary Implementation**: `docs/dual-server-binaries.md`
|
||||
|
||||
---
|
||||
|
||||
## Appendix: Alternative Approaches Considered
|
||||
|
||||
### A. Dynamic CUDA Loading
|
||||
**Idea**: Load CUDA DLLs dynamically at runtime
|
||||
**Why Not**: PyTorch requires CUDA DLLs at import time, can't lazy-load
|
||||
|
||||
### B. CUDA as Separate Package
|
||||
**Idea**: Python package with just CUDA libs
|
||||
**Why Not**: Still 2GB+, same problem
|
||||
|
||||
### C. Model Quantization
|
||||
**Idea**: Use smaller quantized models
|
||||
**Why Not**: Doesn't reduce CUDA runtime size
|
||||
|
||||
### D. Docker Distribution
|
||||
**Idea**: Distribute as Docker container
|
||||
**Why Not**: Poor fit for desktop app, requires Docker installed
|
||||
|
||||
---
|
||||
|
||||
**Document Version**: 1.0
|
||||
**Last Updated**: 2026-01-31
|
||||
**Status**: Research Phase
|
||||
**Next Review**: After cloud provider decision
|
||||
@@ -1,177 +0,0 @@
|
||||
# Dual Server Binary System
|
||||
|
||||
## Overview
|
||||
|
||||
Voicebox now uses a dual-binary approach to manage the size difference between CPU-only and CUDA-enabled builds:
|
||||
|
||||
- **CPU Binary** (~500MB): Ships with the installer by default
|
||||
- **CUDA Binary** (~3GB): Downloaded on-demand for GPU users
|
||||
|
||||
## Problem Solved
|
||||
|
||||
Previously, bundling PyTorch with CUDA support created a 3GB server binary, which:
|
||||
- Made the installer too large (failed CI builds with WiX)
|
||||
- Forced all users to download CUDA libraries even without NVIDIA GPUs
|
||||
- Created poor user experience
|
||||
|
||||
## Solution
|
||||
|
||||
### Build Process
|
||||
|
||||
**Two separate binaries are built:**
|
||||
|
||||
1. **voicebox-server.exe** (CPU)
|
||||
- Built with: `pip install torch --index-url https://download.pytorch.org/whl/cpu`
|
||||
- Size: ~500MB
|
||||
- Works on all Windows machines
|
||||
- Included in the installer by default
|
||||
|
||||
2. **voicebox-server-cuda.exe** (CUDA)
|
||||
- Built with: `pip install torch --index-url https://download.pytorch.org/whl/cu121`
|
||||
- Size: ~3GB
|
||||
- Requires NVIDIA GPU + drivers
|
||||
- Uploaded as separate GitHub Release asset
|
||||
|
||||
### User Experience
|
||||
|
||||
**First Launch:**
|
||||
1. User installs app (~500MB download)
|
||||
2. App starts with CPU server
|
||||
3. If NVIDIA GPU detected:
|
||||
- Show notification: "Download CUDA support for 4-5x faster inference?"
|
||||
- User clicks "Download"
|
||||
- Download voicebox-server-cuda.exe from GitHub (~3GB)
|
||||
- Save to `%APPDATA%/voicebox/binaries/`
|
||||
- Restart server with CUDA version
|
||||
|
||||
**Settings Panel:**
|
||||
- Toggle between CPU/CUDA modes
|
||||
- Download CUDA if not already installed
|
||||
- Show current inference backend
|
||||
|
||||
### Build Scripts
|
||||
|
||||
**Windows:**
|
||||
```bash
|
||||
cd backend
|
||||
|
||||
# Build CPU only
|
||||
build_cpu.bat
|
||||
|
||||
# Build CUDA only
|
||||
build_cuda.bat
|
||||
|
||||
# Build both
|
||||
build_both.bat
|
||||
```
|
||||
|
||||
**Unix (macOS/Linux):**
|
||||
```bash
|
||||
cd backend
|
||||
|
||||
# Build CPU only
|
||||
./build_cpu.sh
|
||||
```
|
||||
|
||||
### CI/CD Workflow
|
||||
|
||||
**GitHub Actions (.github/workflows/release.yml):**
|
||||
|
||||
1. Install CPU PyTorch
|
||||
2. Build CPU server → Copy to Tauri binaries
|
||||
3. Install CUDA PyTorch
|
||||
4. Build CUDA server → Save for upload
|
||||
5. Build Tauri app (bundles CPU server)
|
||||
6. Upload CUDA server as separate release asset
|
||||
|
||||
### File Structure
|
||||
|
||||
```
|
||||
Release Assets:
|
||||
├── Voicebox_0.1.12_x64_en-US.msi (~500MB - includes CPU server)
|
||||
├── voicebox-server-cuda-x86_64-pc-windows-msvc.exe (~3GB - optional download)
|
||||
└── latest.json (updater manifest)
|
||||
```
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Modified Files
|
||||
|
||||
1. **backend/build_binary.py**
|
||||
- Added `variant` parameter ('cpu' or 'cuda')
|
||||
- Outputs different binary names based on variant
|
||||
|
||||
2. **backend/build_cpu.bat** (new)
|
||||
- Installs CPU PyTorch
|
||||
- Builds CPU binary
|
||||
- Restores CUDA PyTorch for dev
|
||||
|
||||
3. **backend/build_cuda.bat** (new)
|
||||
- Ensures CUDA PyTorch is installed
|
||||
- Builds CUDA binary
|
||||
|
||||
4. **.github/workflows/release.yml**
|
||||
- Build CPU binary first (for installer)
|
||||
- Build CUDA binary second (for upload)
|
||||
- Upload CUDA binary as additional release asset
|
||||
- Updated release notes to explain GPU acceleration
|
||||
|
||||
### Future Frontend Work
|
||||
|
||||
**TODO: Implement CUDA download in the app**
|
||||
|
||||
Location: `tauri/src/`
|
||||
|
||||
Features needed:
|
||||
1. GPU detection on startup
|
||||
2. Download manager for CUDA binary
|
||||
3. Server binary path switcher
|
||||
4. Settings UI for CPU/CUDA toggle
|
||||
5. Progress indicator for 3GB download
|
||||
|
||||
API endpoints needed (already exist):
|
||||
- `/health` - Shows GPU availability
|
||||
- Server restart mechanism
|
||||
|
||||
## Benefits
|
||||
|
||||
✓ **Smaller installer**: ~500MB instead of 3GB
|
||||
✓ **Faster CI builds**: WiX can handle 500MB easily
|
||||
✓ **User choice**: CPU users don't download unnecessary files
|
||||
✓ **Better UX**: Optional performance upgrade for GPU users
|
||||
✓ **Cost savings**: Reduced bandwidth for users without GPUs
|
||||
|
||||
## Testing
|
||||
|
||||
**Test CPU build:**
|
||||
```bash
|
||||
cd backend
|
||||
python build_binary.py cpu
|
||||
./dist/voicebox-server.exe --version
|
||||
```
|
||||
|
||||
**Test CUDA build:**
|
||||
```bash
|
||||
cd backend
|
||||
python build_binary.py cuda
|
||||
./dist/voicebox-server-cuda.exe --version
|
||||
```
|
||||
|
||||
**Verify size:**
|
||||
```bash
|
||||
ls -lh backend/dist/
|
||||
# Should see:
|
||||
# voicebox-server.exe ~500MB
|
||||
# voicebox-server-cuda.exe ~3GB
|
||||
```
|
||||
|
||||
**Test server startup:**
|
||||
```bash
|
||||
# CPU version
|
||||
./backend/dist/voicebox-server.exe
|
||||
# Check logs: Should show CPU inference
|
||||
|
||||
# CUDA version (requires NVIDIA GPU)
|
||||
./backend/dist/voicebox-server-cuda.exe
|
||||
# Check logs: Should show CUDA inference
|
||||
```
|
||||
@@ -1,122 +0,0 @@
|
||||
# GitHub 2GB Release Asset Limit Issue
|
||||
|
||||
## Problem
|
||||
|
||||
The CUDA server binary upload fails in CI with:
|
||||
```
|
||||
Error: File size (2543828017) is greater than 2 GiB
|
||||
```
|
||||
|
||||
GitHub release assets have a hard limit of 2GB per file. Our CUDA binary is ~2.5GB, which exceeds this limit.
|
||||
|
||||
## Background
|
||||
|
||||
The dual-server binary system (see `dual-server-binaries.md`) creates two binaries:
|
||||
- **CPU binary**: ~500MB ✅ Works fine
|
||||
- **CUDA binary**: ~2.5GB ❌ Exceeds GitHub limit
|
||||
|
||||
## Attempted Solution: Compression
|
||||
|
||||
We're testing 7z compression with maximum settings to see if we can squeeze the CUDA binary under 2GB.
|
||||
|
||||
### Test Script
|
||||
|
||||
Run `backend/test_cuda_compression.py` to test compression locally:
|
||||
|
||||
```bash
|
||||
cd backend
|
||||
python test_cuda_compression.py
|
||||
```
|
||||
|
||||
This will:
|
||||
1. Find the CUDA binary in `dist/`
|
||||
2. Compress it with 7z (maximum compression)
|
||||
3. Report if the compressed size fits under 2GB
|
||||
|
||||
### Expected Compression
|
||||
|
||||
PyTorch CUDA binaries typically compress well since they contain:
|
||||
- Repeated patterns in neural network weights
|
||||
- Debug symbols and metadata
|
||||
- Redundant CUDA libraries
|
||||
|
||||
Estimated compression: 30-40% reduction
|
||||
- Original: ~2.5GB
|
||||
- Target: <2GB
|
||||
- Required compression: >20%
|
||||
|
||||
## Fallback: External Hosting
|
||||
|
||||
If compression doesn't work, we'll need to host the CUDA binary externally:
|
||||
|
||||
### Option 1: AWS S3
|
||||
```yaml
|
||||
- name: Upload CUDA binary to S3
|
||||
run: |
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
s3://voicebox-releases/cuda-binaries/${{ github.ref_name }}/
|
||||
```
|
||||
|
||||
### Option 2: Azure Blob Storage
|
||||
```yaml
|
||||
- name: Upload to Azure Blob
|
||||
run: |
|
||||
az storage blob upload \
|
||||
--account-name voiceboxreleases \
|
||||
--container-name cuda-binaries \
|
||||
--file backend/cuda-release/voicebox-server-cuda-*.exe
|
||||
```
|
||||
|
||||
### Option 3: GitHub Packages (Container Registry)
|
||||
Package as a container image, though this adds complexity for desktop app distribution.
|
||||
|
||||
## Implementation Plan
|
||||
|
||||
1. **Test compression locally** ← Current step
|
||||
2. **If compression works (<2GB)**:
|
||||
- Update CI to compress before upload
|
||||
- Update app to handle .7z downloads
|
||||
- Add extraction step in download manager
|
||||
|
||||
3. **If compression fails (≥2GB)**:
|
||||
- Set up external storage (likely S3)
|
||||
- Update CI to upload to S3
|
||||
- Provide download URL in release notes
|
||||
- Update app download manager to fetch from S3
|
||||
|
||||
## CI Workflow Changes (if compression works)
|
||||
|
||||
```yaml
|
||||
- name: Compress CUDA binary (Windows only)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
shell: bash
|
||||
run: |
|
||||
cd backend/cuda-release
|
||||
7z a -t7z -m0=lzma2 -mx=9 -mfb=64 -md=32m -ms=on \
|
||||
voicebox-server-cuda-x86_64-pc-windows-msvc.7z \
|
||||
voicebox-server-cuda-*.exe
|
||||
|
||||
- name: Upload compressed CUDA server (Windows only)
|
||||
if: matrix.platform == 'windows-latest'
|
||||
uses: softprops/action-gh-release@v1
|
||||
with:
|
||||
files: backend/cuda-release/*.7z
|
||||
```
|
||||
|
||||
## User Experience Impact
|
||||
|
||||
### With Compression
|
||||
- Download: `voicebox-server-cuda-*.7z` (~1.5-1.8GB)
|
||||
- App extracts automatically
|
||||
- One extra step but manageable
|
||||
|
||||
### With External Hosting
|
||||
- Download from S3/Azure URL
|
||||
- No GitHub release asset dependency
|
||||
- Potentially faster download speeds (CDN)
|
||||
|
||||
## Status
|
||||
|
||||
🔄 **Testing compression locally to determine viability**
|
||||
|
||||
Results pending from local test run.
|
||||
@@ -5,7 +5,7 @@ description: "Welcome to Voicebox - the open-source voice synthesis studio"
|
||||
|
||||
## What is Voicebox?
|
||||
|
||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as the **Ollama for voice** — download models, clone voices, and generate speech entirely on your machine.
|
||||
Voicebox is a **local-first voice cloning studio** with DAW-like features for professional voice synthesis. Think of it as a **local, free and open-source alternative to ElevenLabs** — download models, clone voices, and generate speech entirely on your machine.
|
||||
|
||||
<Frame>
|
||||
<img src="/images/app-screenshot-1.webp" alt="Voicebox App Screenshot" />
|
||||
|
||||
@@ -0,0 +1,964 @@
|
||||
# TTS Provider Architecture
|
||||
|
||||
**Status:** Planned for v0.1.13
|
||||
**Created:** 2025-01-31
|
||||
**Problem:** GitHub 2GB release limit + poor UX for frequent updates requiring 2.4GB re-downloads
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Split the monolithic backend into modular components:
|
||||
|
||||
1. **Main App** (~150-200MB): Tauri + FastAPI backend + Whisper + UI/profiles/history
|
||||
2. **TTS Providers** (downloadable plugins): Separate executables for model inference
|
||||
|
||||
This architecture solves:
|
||||
|
||||
- ✅ GitHub 2GB release artifact limit
|
||||
- ✅ Frequent app updates without re-downloading large python binaries
|
||||
- ✅ User choice of compute backend (CPU/GPU/Cloud)
|
||||
- ✅ External provider support (OpenAI, custom servers)
|
||||
- ✅ Future extensibility
|
||||
|
||||
---
|
||||
|
||||
## Architecture Diagram
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────┐
|
||||
│ Voicebox App (Tauri + Backend) ~150MB │
|
||||
│ ├─ UI Layer (React) │
|
||||
│ ├─ Backend (FastAPI) │
|
||||
│ │ ├─ Voice Profiles │
|
||||
│ │ ├─ Generation History │
|
||||
│ │ ├─ Audio Editing / Stories │
|
||||
│ │ └─ Provider Manager ◄──────────────┐ │
|
||||
│ └─ Whisper (bundled, tiny ~50MB) │ │
|
||||
└─────────────────────────────────────────┼────────────────┘
|
||||
│
|
||||
HTTP/IPC │
|
||||
│
|
||||
┌────────────────────────────────┼─────────────────┐
|
||||
│ │ │
|
||||
▼ ▼ ▼
|
||||
┌─────────────────┐ ┌─────────────────┐ ┌──────────────────┐
|
||||
│ TTS Provider: │ │ TTS Provider: │ │ TTS Provider: │
|
||||
│ PyTorch CPU │ │ PyTorch CUDA │ │ MLX (Apple) │
|
||||
│ │ │ │ │ │
|
||||
│ ~300MB │ │ ~2.4GB │ │ ~800MB │
|
||||
│ │ │ │ │ │
|
||||
│ Local inference │ │ GPU inference │ │ Metal inference │
|
||||
└─────────────────┘ └─────────────────┘ └──────────────────┘
|
||||
│ │ │
|
||||
└────────────────────────┴─────────────────────┘
|
||||
│
|
||||
┌─────────────▼──────────────┐
|
||||
│ Future Providers: │
|
||||
│ • Remote Server │
|
||||
│ • OpenAI API │
|
||||
│ • ElevenLabs │
|
||||
│ • Custom Docker Container │
|
||||
└────────────────────────────┘
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Problem Statement
|
||||
|
||||
### Current Architecture Issues
|
||||
|
||||
**Monolithic Binary:**
|
||||
|
||||
- CPU version: ~295MB
|
||||
- CUDA version: ~2.37GB
|
||||
- GitHub releases: 2GB file size limit (BLOCKED)
|
||||
- Updates require re-downloading entire binary
|
||||
- Poor UX: update app → restart → download CUDA update → restart again
|
||||
|
||||
**User Pain Points:**
|
||||
|
||||
1. Cannot release CUDA version on GitHub (over 2GB)
|
||||
2. Every app update forces 2.4GB re-download for GPU users
|
||||
3. No flexibility (can't use OpenAI, remote servers, etc.)
|
||||
4. Wastes bandwidth for small bug fixes
|
||||
|
||||
---
|
||||
|
||||
## Solution: Pluggable TTS Providers
|
||||
|
||||
### Component Breakdown
|
||||
|
||||
#### 1. Main App (voicebox.exe / .app / .AppImage)
|
||||
|
||||
**Size:** ~100-150MB
|
||||
|
||||
**Includes:**
|
||||
|
||||
- Tauri runtime + React UI
|
||||
- FastAPI backend (pure Python, no PyTorch)
|
||||
- Whisper model (tiny, ~50MB)
|
||||
- SQLite database
|
||||
- Profile/history/audio editing logic
|
||||
- Provider management system
|
||||
|
||||
**Does NOT include:**
|
||||
|
||||
- PyTorch (CPU or CUDA)
|
||||
- TTS models (Qwen3-TTS)
|
||||
- Heavy ML dependencies
|
||||
|
||||
**Updates frequently:** UI fixes, feature additions, non-ML changes
|
||||
|
||||
---
|
||||
|
||||
#### 2. TTS Provider: PyTorch CPU
|
||||
|
||||
**Binary:** `tts-provider-pytorch-cpu.exe`
|
||||
**Size:** ~200MB
|
||||
|
||||
**Includes:**
|
||||
|
||||
- PyTorch CPU build
|
||||
- Qwen3-TTS package
|
||||
- Transformers
|
||||
- No CUDA libraries
|
||||
|
||||
**Download source:** Cloudflare R2
|
||||
**Updates rarely:** Only when model code changes
|
||||
|
||||
---
|
||||
|
||||
#### 3. TTS Provider: PyTorch CUDA
|
||||
|
||||
**Binary:** `tts-provider-pytorch-cuda.exe`
|
||||
**Size:** ~2.4GB
|
||||
|
||||
**Includes:**
|
||||
|
||||
- PyTorch CUDA build (cu121)
|
||||
- Qwen3-TTS package
|
||||
- CUDA runtime, cuDNN, cuBLAS
|
||||
- Transformers
|
||||
|
||||
**Download source:** Cloudflare R2
|
||||
**Platform:** Windows + Linux (NVIDIA GPU)
|
||||
**Updates rarely:** Only when model code or CUDA version changes
|
||||
|
||||
---
|
||||
|
||||
#### 4. TTS Provider: MLX
|
||||
|
||||
**Binary:** `tts-provider-mlx`
|
||||
**Size:** ~150MB
|
||||
|
||||
**Includes:**
|
||||
|
||||
- MLX framework
|
||||
- MLX-optimized Qwen3-TTS
|
||||
- Metal acceleration
|
||||
|
||||
**Platform:** macOS only (Apple Silicon)
|
||||
**Download source:** Cloudflare R2
|
||||
|
||||
---
|
||||
|
||||
#### 5. TTS Provider: Remote
|
||||
|
||||
**Binary:** None (built-in config)
|
||||
**Size:** 0MB
|
||||
|
||||
**How it works:**
|
||||
|
||||
- User provides URL to their own TTS server
|
||||
- Backend proxies requests to that server
|
||||
- Implements API spec from `EXTERNAL_PROVIDERS.md`
|
||||
|
||||
**Use cases:**
|
||||
|
||||
- AMD GPU users running their own server
|
||||
- Team deployments with shared GPU server
|
||||
- Cloud hosting (Modal, RunPod, Replicate)
|
||||
|
||||
---
|
||||
|
||||
#### 6. TTS Provider: OpenAI
|
||||
|
||||
**Binary:** None (API wrapper)
|
||||
**Size:** 0MB
|
||||
|
||||
**How it works:**
|
||||
|
||||
- User provides OpenAI API key
|
||||
- Backend wraps OpenAI Audio API
|
||||
- Voice profiles map to OpenAI voices
|
||||
|
||||
**Benefits:**
|
||||
|
||||
- Zero local compute
|
||||
- Pay-per-use
|
||||
- Instant setup
|
||||
|
||||
---
|
||||
|
||||
## Communication Protocol
|
||||
|
||||
### Provider API Specification
|
||||
|
||||
All TTS providers must implement these endpoints:
|
||||
|
||||
#### POST /tts/generate
|
||||
|
||||
Generate speech from text.
|
||||
|
||||
**Request:**
|
||||
|
||||
```json
|
||||
{
|
||||
"text": "Hello world!",
|
||||
"voice_prompt": {
|
||||
/* voice prompt object */
|
||||
},
|
||||
"language": "en",
|
||||
"seed": 12345,
|
||||
"model_size": "1.7B"
|
||||
}
|
||||
```
|
||||
|
||||
**Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"audio": "base64-encoded-audio",
|
||||
"sample_rate": 24000,
|
||||
"duration": 2.5
|
||||
}
|
||||
```
|
||||
|
||||
#### POST /tts/create_voice_prompt
|
||||
|
||||
Create voice prompt from reference audio.
|
||||
|
||||
**Request:** (multipart/form-data)
|
||||
|
||||
- `audio`: Audio file
|
||||
- `reference_text`: Transcript
|
||||
|
||||
**Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"voice_prompt": {
|
||||
/* serialized prompt */
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### GET /tts/health
|
||||
|
||||
Health check.
|
||||
|
||||
**Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"status": "healthy",
|
||||
"provider": "pytorch-cuda",
|
||||
"version": "1.0.0",
|
||||
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"device": "cuda:0"
|
||||
}
|
||||
```
|
||||
|
||||
#### GET /tts/status
|
||||
|
||||
Model status.
|
||||
|
||||
**Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"model_loaded": true,
|
||||
"model_size": "1.7B",
|
||||
"available_sizes": ["0.6B", "1.7B"],
|
||||
"gpu_available": true,
|
||||
"vram_used_mb": 1234
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Backend Implementation
|
||||
|
||||
### Provider Manager
|
||||
|
||||
**File:** `backend/providers/__init__.py`
|
||||
|
||||
```python
|
||||
class ProviderManager:
|
||||
"""Manages TTS provider lifecycle."""
|
||||
|
||||
def __init__(self):
|
||||
self.active_provider: Optional[Provider] = None
|
||||
self.config = load_provider_config()
|
||||
|
||||
async def start_provider(self, provider_type: str) -> str:
|
||||
"""Start a TTS provider process."""
|
||||
if provider_type == "pytorch-cpu":
|
||||
return await self._start_local_provider("tts-provider-pytorch-cpu.exe")
|
||||
elif provider_type == "pytorch-cuda":
|
||||
return await self._start_local_provider("tts-provider-pytorch-cuda.exe")
|
||||
elif provider_type == "mlx":
|
||||
return await self._start_local_provider("tts-provider-mlx")
|
||||
elif provider_type == "remote":
|
||||
return self.config["remote_url"]
|
||||
elif provider_type == "openai":
|
||||
return None # No subprocess, API wrapper
|
||||
|
||||
async def _start_local_provider(self, binary_name: str) -> str:
|
||||
"""Start local provider subprocess."""
|
||||
provider_path = get_provider_binary_path(binary_name)
|
||||
|
||||
if not provider_path.exists():
|
||||
raise ProviderNotInstalledException(binary_name)
|
||||
|
||||
# Start subprocess on random port
|
||||
port = get_free_port()
|
||||
process = subprocess.Popen([
|
||||
str(provider_path),
|
||||
"--port", str(port),
|
||||
"--data-dir", str(config.get_data_dir())
|
||||
])
|
||||
|
||||
# Wait for provider to be ready
|
||||
await wait_for_provider_health(f"http://localhost:{port}")
|
||||
|
||||
self.active_provider = Provider(process, port)
|
||||
return f"http://localhost:{port}"
|
||||
|
||||
async def stop_provider(self):
|
||||
"""Stop active provider."""
|
||||
if self.active_provider:
|
||||
self.active_provider.process.terminate()
|
||||
self.active_provider = None
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Provider Abstraction
|
||||
|
||||
**File:** `backend/providers/base.py`
|
||||
|
||||
```python
|
||||
class TTSProvider(ABC):
|
||||
"""Abstract base for TTS providers."""
|
||||
|
||||
@abstractmethod
|
||||
async def generate(
|
||||
self,
|
||||
text: str,
|
||||
voice_prompt: dict,
|
||||
language: str,
|
||||
seed: Optional[int]
|
||||
) -> tuple[np.ndarray, int]:
|
||||
"""Generate speech audio."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
async def create_voice_prompt(
|
||||
self,
|
||||
audio_path: str,
|
||||
reference_text: str
|
||||
) -> dict:
|
||||
"""Create voice prompt from reference audio."""
|
||||
pass
|
||||
```
|
||||
|
||||
**File:** `backend/providers/local.py`
|
||||
|
||||
```python
|
||||
class LocalProvider(TTSProvider):
|
||||
"""Provider that communicates with local subprocess via HTTP."""
|
||||
|
||||
def __init__(self, base_url: str):
|
||||
self.base_url = base_url
|
||||
self.client = httpx.AsyncClient()
|
||||
|
||||
async def generate(self, text, voice_prompt, language, seed):
|
||||
response = await self.client.post(
|
||||
f"{self.base_url}/tts/generate",
|
||||
json={
|
||||
"text": text,
|
||||
"voice_prompt": voice_prompt,
|
||||
"language": language,
|
||||
"seed": seed
|
||||
}
|
||||
)
|
||||
data = response.json()
|
||||
audio = np.frombuffer(base64.b64decode(data["audio"]), dtype=np.float32)
|
||||
return audio, data["sample_rate"]
|
||||
```
|
||||
|
||||
**File:** `backend/providers/openai.py`
|
||||
|
||||
```python
|
||||
class OpenAIProvider(TTSProvider):
|
||||
"""Provider that wraps OpenAI Audio API."""
|
||||
|
||||
def __init__(self, api_key: str):
|
||||
self.client = OpenAI(api_key=api_key)
|
||||
|
||||
async def generate(self, text, voice_prompt, language, seed):
|
||||
# Map voice_prompt to OpenAI voice name
|
||||
voice = map_profile_to_openai_voice(voice_prompt)
|
||||
|
||||
response = await self.client.audio.speech.create(
|
||||
model="tts-1",
|
||||
voice=voice,
|
||||
input=text
|
||||
)
|
||||
|
||||
# Convert to numpy array
|
||||
audio_data = response.content
|
||||
audio, sr = load_audio_from_bytes(audio_data)
|
||||
return audio, sr
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Provider Installation
|
||||
|
||||
### Download Manager
|
||||
|
||||
**File:** `backend/providers/installer.py`
|
||||
|
||||
```python
|
||||
class ProviderInstaller:
|
||||
"""Handles provider download and installation."""
|
||||
|
||||
async def download_provider(self, provider_type: str):
|
||||
"""Download provider binary from R2."""
|
||||
|
||||
binary_name = {
|
||||
"pytorch-cpu": "tts-provider-pytorch-cpu.exe",
|
||||
"pytorch-cuda": "tts-provider-pytorch-cuda.exe",
|
||||
"mlx": "tts-provider-mlx"
|
||||
}[provider_type]
|
||||
|
||||
download_url = f"https://downloads.voicebox.sh/providers/v{PROVIDER_VERSION}/{binary_name}"
|
||||
|
||||
# Download with progress tracking (reuse existing SSE system)
|
||||
await download_with_progress(
|
||||
url=download_url,
|
||||
destination=get_provider_install_path(binary_name),
|
||||
progress_key=f"provider-{provider_type}"
|
||||
)
|
||||
```
|
||||
|
||||
**Provider Storage Location:**
|
||||
|
||||
- Windows: `%APPDATA%/voicebox/providers/`
|
||||
- macOS: `~/Library/Application Support/voicebox/providers/`
|
||||
- Linux: `~/.local/share/voicebox/providers/`
|
||||
|
||||
---
|
||||
|
||||
## Frontend Implementation
|
||||
|
||||
### Provider Settings UI
|
||||
|
||||
**Component:** `app/src/components/ServerSettings/ProviderSettings.tsx`
|
||||
|
||||
```tsx
|
||||
export function ProviderSettings() {
|
||||
const [selectedProvider, setSelectedProvider] =
|
||||
useState<ProviderType>("auto");
|
||||
const {data: installedProviders} = useQuery({
|
||||
queryKey: ["providers", "installed"],
|
||||
queryFn: () => apiClient.getInstalledProviders(),
|
||||
});
|
||||
|
||||
return (
|
||||
<Card>
|
||||
<CardHeader>
|
||||
<CardTitle>TTS Provider</CardTitle>
|
||||
<CardDescription>Choose how Voicebox generates speech</CardDescription>
|
||||
</CardHeader>
|
||||
<CardContent>
|
||||
<RadioGroup
|
||||
value={selectedProvider}
|
||||
onValueChange={setSelectedProvider}
|
||||
>
|
||||
{/* Auto-detect */}
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem value="auto" id="auto" />
|
||||
<Label htmlFor="auto">
|
||||
<div className="font-medium">Auto-detect (Recommended)</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
Automatically choose the best available provider
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
|
||||
{/* PyTorch CUDA */}
|
||||
<div className="flex items-center justify-between">
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem
|
||||
value="pytorch-cuda"
|
||||
id="cuda"
|
||||
disabled={!gpuAvailable}
|
||||
/>
|
||||
<Label htmlFor="cuda">
|
||||
<div className="font-medium">PyTorch CUDA (NVIDIA GPU)</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
4-5x faster inference on NVIDIA GPUs
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
{!installedProviders?.includes("pytorch-cuda") && gpuAvailable && (
|
||||
<Button
|
||||
onClick={() => downloadProvider("pytorch-cuda")}
|
||||
size="sm"
|
||||
>
|
||||
Download (2.4GB)
|
||||
</Button>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* PyTorch CPU */}
|
||||
<div className="flex items-center justify-between">
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem value="pytorch-cpu" id="cpu" />
|
||||
<Label htmlFor="cpu">
|
||||
<div className="font-medium">PyTorch CPU</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
Works on any system, slower inference
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
{!installedProviders?.includes("pytorch-cpu") && (
|
||||
<Button onClick={() => downloadProvider("pytorch-cpu")} size="sm">
|
||||
Download (300MB)
|
||||
</Button>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* MLX (macOS only) */}
|
||||
{isMacOS && (
|
||||
<div className="flex items-center justify-between">
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem value="mlx" id="mlx" />
|
||||
<Label htmlFor="mlx">
|
||||
<div className="font-medium">MLX (Apple Silicon)</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
Optimized for M1/M2/M3 chips
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
{!installedProviders?.includes("mlx") && (
|
||||
<Button onClick={() => downloadProvider("mlx")} size="sm">
|
||||
Download (800MB)
|
||||
</Button>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Remote */}
|
||||
<div className="space-y-2">
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem value="remote" id="remote" />
|
||||
<Label htmlFor="remote">
|
||||
<div className="font-medium">Remote Server</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
Connect to your own TTS server
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
{selectedProvider === "remote" && (
|
||||
<Input placeholder="http://your-server:8000" className="ml-6" />
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* OpenAI */}
|
||||
<div className="space-y-2">
|
||||
<div className="flex items-center space-x-2">
|
||||
<RadioGroupItem value="openai" id="openai" />
|
||||
<Label htmlFor="openai">
|
||||
<div className="font-medium">OpenAI API</div>
|
||||
<div className="text-sm text-muted-foreground">
|
||||
Use OpenAI's TTS API (requires API key)
|
||||
</div>
|
||||
</Label>
|
||||
</div>
|
||||
{selectedProvider === "openai" && (
|
||||
<Input type="password" placeholder="sk-..." className="ml-6" />
|
||||
)}
|
||||
</div>
|
||||
</RadioGroup>
|
||||
</CardContent>
|
||||
</Card>
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## File Structure
|
||||
|
||||
```
|
||||
voicebox/
|
||||
├── backend/
|
||||
│ ├── main.py # Main FastAPI app (no TTS code)
|
||||
│ ├── providers/
|
||||
│ │ ├── __init__.py # ProviderManager
|
||||
│ │ ├── base.py # TTSProvider ABC
|
||||
│ │ ├── local.py # LocalProvider (subprocess)
|
||||
│ │ ├── remote.py # RemoteProvider (HTTP)
|
||||
│ │ ├── openai.py # OpenAIProvider (API wrapper)
|
||||
│ │ └── installer.py # Provider download logic
|
||||
│ ├── profiles.py # Voice profile management
|
||||
│ ├── history.py # Generation history
|
||||
│ ├── transcribe.py # Whisper (still bundled)
|
||||
│ └── ... (other backend modules)
|
||||
│
|
||||
├── providers/
|
||||
│ ├── pytorch-cpu/
|
||||
│ │ ├── main.py # FastAPI server for TTS
|
||||
│ │ ├── tts_backend.py # PyTorch TTS logic
|
||||
│ │ ├── requirements.txt # torch (CPU), qwen-tts, transformers
|
||||
│ │ └── build.spec # PyInstaller spec
|
||||
│ │
|
||||
│ ├── pytorch-cuda/
|
||||
│ │ ├── main.py # FastAPI server for TTS
|
||||
│ │ ├── tts_backend.py # PyTorch TTS logic
|
||||
│ │ ├── requirements.txt # torch+cu121, qwen-tts, transformers
|
||||
│ │ └── build.spec # PyInstaller spec
|
||||
│ │
|
||||
│ └── mlx/
|
||||
│ ├── main.py # FastAPI server for TTS
|
||||
│ ├── mlx_backend.py # MLX TTS logic
|
||||
│ ├── requirements.txt # mlx, qwen-tts-mlx
|
||||
│ └── build.spec # PyInstaller spec
|
||||
│
|
||||
├── app/ # Frontend (Tauri + React)
|
||||
│ └── src/
|
||||
│ └── components/
|
||||
│ └── ServerSettings/
|
||||
│ └── ProviderSettings.tsx
|
||||
│
|
||||
└── tauri/
|
||||
└── src-tauri/
|
||||
└── tauri.conf.json # No externalBin for providers
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Migration Path
|
||||
|
||||
### Phase 1: Refactor Backend (No User Changes)
|
||||
|
||||
**Goal:** Abstract TTS behind provider interface
|
||||
|
||||
1. Create `backend/providers/` module structure
|
||||
2. Implement `TTSProvider` abstract base class
|
||||
3. Create `LocalProvider` wrapper for current PyTorch code
|
||||
4. Modify `backend/tts.py` to use provider abstraction
|
||||
5. Keep PyTorch bundled in main app
|
||||
|
||||
**Result:** Code is prepared, but user experience unchanged
|
||||
|
||||
---
|
||||
|
||||
### Phase 2: Build Provider Binaries
|
||||
|
||||
**Goal:** Create standalone TTS provider executables
|
||||
|
||||
1. Create separate PyInstaller specs for each provider
|
||||
2. Build provider executables:
|
||||
- `tts-provider-pytorch-cpu.exe` (~300MB)
|
||||
- `tts-provider-pytorch-cuda.exe` (~2.4GB)
|
||||
- `tts-provider-mlx` (~800MB, macOS)
|
||||
3. Test subprocess communication
|
||||
4. Upload providers to Cloudflare R2
|
||||
|
||||
**Result:** Provider binaries exist but aren't used yet
|
||||
|
||||
---
|
||||
|
||||
### Phase 3: Remove PyTorch from Main App
|
||||
|
||||
**Goal:** Split main app from providers
|
||||
|
||||
1. Exclude PyTorch/Qwen3-TTS from main app PyInstaller spec
|
||||
2. Main app now requires provider download
|
||||
3. Update GitHub CI to build multiple artifacts:
|
||||
- `voicebox-{version}-{platform}.exe` (~150MB)
|
||||
- `tts-provider-pytorch-cpu-{version}.exe`
|
||||
- `tts-provider-pytorch-cuda-{version}.exe`
|
||||
- `tts-provider-mlx-{version}` (macOS)
|
||||
|
||||
**Result:** Main app is small, providers downloaded separately
|
||||
|
||||
---
|
||||
|
||||
### Phase 4: Add Provider UI
|
||||
|
||||
**Goal:** User-facing provider management
|
||||
|
||||
1. Create Provider Settings page
|
||||
2. Implement provider download UI
|
||||
3. Add provider status indicators
|
||||
4. Show active provider in UI
|
||||
|
||||
**Result:** Users can choose and download providers
|
||||
|
||||
---
|
||||
|
||||
### Phase 5: External Providers
|
||||
|
||||
**Goal:** Enable remote and cloud providers
|
||||
|
||||
1. Implement `RemoteProvider` (HTTP client)
|
||||
2. Implement `OpenAIProvider` (API wrapper)
|
||||
3. Add provider configuration UI (URLs, API keys)
|
||||
4. Document external provider API spec
|
||||
|
||||
**Result:** Full provider ecosystem
|
||||
|
||||
---
|
||||
|
||||
## Provider Versioning
|
||||
|
||||
### Independent Versioning
|
||||
|
||||
Providers have their own version numbers, independent of the main app:
|
||||
|
||||
- **App version:** `v0.2.0` (frequent updates)
|
||||
- **Provider version:** `v1.0.0` (rare updates)
|
||||
|
||||
### Compatibility Matrix
|
||||
|
||||
**Example:**
|
||||
|
||||
| App Version | Min Provider Version | Max Provider Version |
|
||||
| ----------- | -------------------- | -------------------- |
|
||||
| v0.2.0 | v1.0.0 | v1.x.x |
|
||||
| v0.3.0 | v1.0.0 | v1.x.x |
|
||||
| v0.4.0 | v1.2.0 | v1.x.x |
|
||||
| v1.0.0 | v2.0.0 | v2.x.x |
|
||||
|
||||
**Backend checks compatibility:**
|
||||
|
||||
```python
|
||||
async def check_provider_compatibility(provider_version: str) -> bool:
|
||||
"""Check if provider version is compatible with current app."""
|
||||
min_version = "1.0.0"
|
||||
max_version = "1.999.999"
|
||||
return min_version <= provider_version < max_version
|
||||
```
|
||||
|
||||
**UI shows warning if incompatible:**
|
||||
|
||||
```
|
||||
⚠️ Provider version 0.9.0 is outdated. Update to v1.0.0+
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## User Flows
|
||||
|
||||
### First-Time Setup
|
||||
|
||||
1. User downloads and installs Voicebox (~150MB)
|
||||
2. App launches → detects no TTS provider installed
|
||||
3. Shows setup wizard:
|
||||
|
||||
```
|
||||
Choose your TTS provider:
|
||||
|
||||
[ ] PyTorch CUDA (2.4GB) [Download]
|
||||
✓ Fastest on NVIDIA GPUs
|
||||
✗ Requires NVIDIA GPU
|
||||
|
||||
[●] PyTorch CPU (300MB) [Download]
|
||||
✓ Works on any system
|
||||
✗ Slower inference
|
||||
|
||||
[ ] MLX (800MB) [Download]
|
||||
✓ Fast on Apple Silicon
|
||||
✗ macOS only (M1/M2/M3)
|
||||
|
||||
[ ] Remote Server
|
||||
URL: ___________________
|
||||
|
||||
[ ] OpenAI API
|
||||
API Key: ________________
|
||||
```
|
||||
|
||||
4. User selects provider → downloads with progress bar
|
||||
5. Provider installs to AppData/Application Support
|
||||
6. App starts provider → ready to use
|
||||
|
||||
---
|
||||
|
||||
### App Update Flow (No Provider Change)
|
||||
|
||||
**Scenario:** Bug fix in UI, no backend changes
|
||||
|
||||
1. User gets update notification: "Voicebox v0.2.1 available"
|
||||
2. Downloads update (~150MB, not 2.4GB!)
|
||||
3. Installs and restarts
|
||||
4. **Provider stays the same** (no re-download needed)
|
||||
5. App starts using existing provider
|
||||
|
||||
**User experience:** Fast updates, no multi-GB downloads
|
||||
|
||||
---
|
||||
|
||||
### Provider Update Flow
|
||||
|
||||
**Scenario:** New Qwen3-TTS model version released
|
||||
|
||||
1. User opens Settings → Provider tab
|
||||
2. Sees notification: "Provider update available (v1.1.0)"
|
||||
3. Clicks "Update Provider"
|
||||
4. Downloads new provider binary
|
||||
5. Old provider binary is replaced
|
||||
6. Restart app to use new provider
|
||||
|
||||
**Frequency:** Rare (only when TTS model/backend changes)
|
||||
|
||||
---
|
||||
|
||||
### Switching Providers
|
||||
|
||||
**Scenario:** User upgrades to NVIDIA GPU
|
||||
|
||||
1. User goes to Settings → Provider
|
||||
2. Selects "PyTorch CUDA"
|
||||
3. Clicks "Download" → downloads 2.4GB
|
||||
4. Download completes → restarts app
|
||||
5. App now uses CUDA provider
|
||||
|
||||
---
|
||||
|
||||
## Benefits
|
||||
|
||||
| Benefit | Details |
|
||||
| ----------------------------- | --------------------------------------------------------- |
|
||||
| **GitHub Releases Work** | Main app ~150MB << 2GB limit |
|
||||
| **Fast Updates** | UI/feature updates don't require re-downloading providers |
|
||||
| **User Choice** | CPU, CUDA, MLX, OpenAI, remote server |
|
||||
| **External Provider Support** | Users can run their own TTS servers |
|
||||
| **Bandwidth Savings** | Only download provider once, app updates are small |
|
||||
| **Future-Proof** | Easy to add new providers (ElevenLabs, custom models) |
|
||||
| **Team Deployments** | Multiple users share one remote provider |
|
||||
| **Cloud-Ready** | Works with Modal, Replicate, RunPod, etc. |
|
||||
|
||||
---
|
||||
|
||||
## Open Questions
|
||||
|
||||
### 1. Provider Versioning
|
||||
|
||||
**Question:** Should providers have independent versions or match app version?
|
||||
|
||||
**Options:**
|
||||
|
||||
- A. Independent (providers: v1.x, app: v0.2.x)
|
||||
- B. Matched (both use v0.2.x)
|
||||
|
||||
**Recommendation:** Independent versioning with compatibility matrix
|
||||
|
||||
---
|
||||
|
||||
### 2. Auto-Update Providers
|
||||
|
||||
**Question:** Should providers auto-update separately from app?
|
||||
|
||||
**Options:**
|
||||
|
||||
- A. Manual updates only (user clicks "Update Provider")
|
||||
- B. Optional auto-update (user can enable)
|
||||
- C. Always auto-update
|
||||
|
||||
**Recommendation:** Optional auto-update (default off)
|
||||
|
||||
---
|
||||
|
||||
### 3. Provider Discovery
|
||||
|
||||
**Question:** How does app find installed providers?
|
||||
|
||||
**Options:**
|
||||
|
||||
- A. Check standard paths in AppData/Application Support
|
||||
- B. Registry (Windows) / plist (macOS)
|
||||
- C. Config file with provider locations
|
||||
|
||||
**Recommendation:** Standard paths + config fallback
|
||||
|
||||
---
|
||||
|
||||
### 4. Fallback Behavior
|
||||
|
||||
**Question:** What if no provider is installed?
|
||||
|
||||
**Options:**
|
||||
|
||||
- A. Show setup wizard on first launch
|
||||
- B. Block app until provider installed
|
||||
- C. Allow app to run in "demo mode" (transcription only)
|
||||
|
||||
**Recommendation:** Setup wizard on first launch
|
||||
|
||||
---
|
||||
|
||||
### 5. Provider Auto-Start
|
||||
|
||||
**Question:** Should provider start automatically with app?
|
||||
|
||||
**Options:**
|
||||
|
||||
- A. Always start selected provider on app launch
|
||||
- B. Start on-demand (when user generates speech)
|
||||
- C. User preference
|
||||
|
||||
**Recommendation:** Auto-start (configurable in settings)
|
||||
|
||||
---
|
||||
|
||||
## Future Enhancements
|
||||
|
||||
- [ ] **Provider Marketplace:** Built-in directory of community providers
|
||||
- [ ] **Multi-Provider Support:** Use different providers per voice/language
|
||||
- [ ] **Provider Health Monitoring:** Automatic failover if provider crashes
|
||||
- [ ] **Cost Tracking:** Monitor API usage for OpenAI/cloud providers
|
||||
- [ ] **Performance Metrics:** Latency, throughput, VRAM usage dashboards
|
||||
- [ ] **Docker Providers:** Run providers in Docker containers
|
||||
- [ ] **Provider Plugins:** Load custom providers from user scripts
|
||||
|
||||
---
|
||||
|
||||
## Related Documents
|
||||
|
||||
- [EXTERNAL_PROVIDERS.md](./EXTERNAL_PROVIDERS.md) - External provider support plan
|
||||
- [OPENAI_SUPPORT.md](./OPENAI_SUPPORT.md) - OpenAI API compatibility
|
||||
- [github-2gb-limit-issue.md](../github-2gb-limit-issue.md) - Original problem
|
||||
- [r2-setup.md](../r2-setup.md) - Cloudflare R2 configuration
|
||||
|
||||
---
|
||||
|
||||
## Contributing
|
||||
|
||||
If you want to build a custom TTS provider:
|
||||
|
||||
1. Implement the provider API spec (see above)
|
||||
2. Test with Voicebox locally
|
||||
3. Package as executable (PyInstaller, Docker, etc.)
|
||||
4. Share in GitHub Discussions
|
||||
|
||||
**Questions?**
|
||||
|
||||
- GitHub Issues: [voicebox/issues](https://github.com/jamiepine/voicebox/issues)
|
||||
- Discord: Coming soon
|
||||
@@ -1,274 +0,0 @@
|
||||
# Cloudflare R2 Setup Guide
|
||||
|
||||
## Overview
|
||||
|
||||
The CUDA binary (2.4GB) is hosted on Cloudflare R2 at `downloads.voicebox.sh` instead of GitHub Releases (which has a 2GB limit).
|
||||
|
||||
## R2 Bucket Configuration
|
||||
|
||||
✅ **Completed:**
|
||||
- Bucket created: `voicebox`
|
||||
- Custom domain configured: `downloads.voicebox.sh`
|
||||
|
||||
## GitHub Secrets Required
|
||||
|
||||
Add these secrets to your GitHub repository:
|
||||
|
||||
### 1. R2_ACCESS_KEY_ID
|
||||
|
||||
Your Cloudflare R2 API Access Key ID
|
||||
|
||||
**How to get it:**
|
||||
1. Go to Cloudflare Dashboard → R2
|
||||
2. Click "Manage R2 API Tokens"
|
||||
3. Create API Token with "Object Read & Write" permissions
|
||||
4. Copy the "Access Key ID"
|
||||
|
||||
**Add to GitHub:**
|
||||
```
|
||||
Repository Settings → Secrets and variables → Actions → New repository secret
|
||||
Name: R2_ACCESS_KEY_ID
|
||||
Value: <your-access-key-id>
|
||||
```
|
||||
|
||||
### 2. R2_SECRET_ACCESS_KEY
|
||||
|
||||
Your Cloudflare R2 Secret Access Key
|
||||
|
||||
**How to get it:**
|
||||
- Same process as above
|
||||
- Copy the "Secret Access Key" (shown only once!)
|
||||
- Store it securely
|
||||
|
||||
**Add to GitHub:**
|
||||
```
|
||||
Name: R2_SECRET_ACCESS_KEY
|
||||
Value: <your-secret-access-key>
|
||||
```
|
||||
|
||||
### 3. R2_ENDPOINT
|
||||
|
||||
Your Cloudflare R2 endpoint URL
|
||||
|
||||
**Format:**
|
||||
```
|
||||
https://<account-id>.r2.cloudflarestorage.com
|
||||
```
|
||||
|
||||
**How to find your account ID:**
|
||||
- Cloudflare Dashboard → R2
|
||||
- Look at the URL or bucket settings
|
||||
- Should be a string of letters/numbers
|
||||
|
||||
**Add to GitHub:**
|
||||
```
|
||||
Name: R2_ENDPOINT
|
||||
Value: https://<your-account-id>.r2.cloudflarestorage.com
|
||||
```
|
||||
|
||||
## Bucket Structure
|
||||
|
||||
After CI uploads, the bucket will have this structure:
|
||||
|
||||
```
|
||||
voicebox/
|
||||
└── cuda/
|
||||
├── v0.1.12/
|
||||
│ └── voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
├── v0.1.13/
|
||||
│ └── voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
└── v0.2.0/
|
||||
└── voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
```
|
||||
|
||||
## Public Access
|
||||
|
||||
Files are uploaded with `--acl public-read`, making them accessible at:
|
||||
|
||||
```
|
||||
https://downloads.voicebox.sh/cuda/v{VERSION}/voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
```
|
||||
|
||||
**Example:**
|
||||
```
|
||||
https://downloads.voicebox.sh/cuda/v0.1.12/voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
```
|
||||
|
||||
## Testing the Setup
|
||||
|
||||
### Local Test Upload
|
||||
|
||||
Before running the CI, test uploading locally:
|
||||
|
||||
```bash
|
||||
# Set environment variables
|
||||
export AWS_ACCESS_KEY_ID="your-r2-access-key-id"
|
||||
export AWS_SECRET_ACCESS_KEY="your-r2-secret-access-key"
|
||||
export R2_ENDPOINT="https://your-account-id.r2.cloudflarestorage.com"
|
||||
|
||||
# Install AWS CLI
|
||||
pip install awscli
|
||||
|
||||
# Test upload (use a small test file first)
|
||||
echo "test" > test.txt
|
||||
aws s3 cp test.txt \
|
||||
s3://voicebox/test/test.txt \
|
||||
--endpoint-url $R2_ENDPOINT \
|
||||
--acl public-read
|
||||
|
||||
# Verify it's accessible
|
||||
curl https://downloads.voicebox.sh/test/test.txt
|
||||
|
||||
# If successful, try the actual CUDA binary
|
||||
aws s3 cp backend/dist/voicebox-server-cuda.exe \
|
||||
s3://voicebox/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe \
|
||||
--endpoint-url $R2_ENDPOINT \
|
||||
--acl public-read
|
||||
```
|
||||
|
||||
### Verify Upload
|
||||
|
||||
Check if the file is accessible:
|
||||
|
||||
```bash
|
||||
curl -I https://downloads.voicebox.sh/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe
|
||||
```
|
||||
|
||||
Should return:
|
||||
```
|
||||
HTTP/2 200
|
||||
content-length: 2545086396
|
||||
content-type: application/x-msdownload
|
||||
...
|
||||
```
|
||||
|
||||
## CI Workflow
|
||||
|
||||
The workflow now:
|
||||
|
||||
1. **Builds CPU binary** → Includes in installer
|
||||
2. **Builds CUDA binary** → Uploads to R2
|
||||
3. **Release notes** → Include R2 download link
|
||||
|
||||
### CI Steps (Windows)
|
||||
|
||||
```yaml
|
||||
- name: Build CUDA Python server (Windows only)
|
||||
# Builds the CUDA binary
|
||||
|
||||
- name: Upload CUDA server to Cloudflare R2 (Windows only)
|
||||
env:
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
|
||||
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
|
||||
run: |
|
||||
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
|
||||
s3://voicebox/cuda/${VERSION}/... \
|
||||
--endpoint-url $R2_ENDPOINT \
|
||||
--acl public-read
|
||||
```
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
Monitor your R2 usage:
|
||||
|
||||
**Cloudflare Dashboard → R2 → voicebox → Metrics**
|
||||
|
||||
Expected costs (per month):
|
||||
- Storage: 2.4GB × $0.015/GB = **$0.036**
|
||||
- Egress: **$0.00** (free!)
|
||||
- Class A ops: ~100 × $4.50/million = **$0.00**
|
||||
|
||||
**Total: ~$0.04/month** (essentially free!)
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Upload fails: "Access Denied"
|
||||
|
||||
**Solution:** Check API token permissions
|
||||
- Must have "Object Read & Write" on the bucket
|
||||
- Regenerate token if needed
|
||||
|
||||
### File not accessible at downloads.voicebox.sh
|
||||
|
||||
**Solution:** Check custom domain configuration
|
||||
- R2 Dashboard → Bucket → Settings → Custom Domains
|
||||
- Ensure `downloads.voicebox.sh` is properly configured
|
||||
- DNS may take time to propagate
|
||||
|
||||
### "endpoint-url" not recognized
|
||||
|
||||
**Solution:** Make sure AWS CLI is updated
|
||||
```bash
|
||||
pip install --upgrade awscli
|
||||
```
|
||||
|
||||
### File uploaded but wrong permissions
|
||||
|
||||
**Solution:** Re-upload with `--acl public-read`
|
||||
```bash
|
||||
aws s3 cp ... --acl public-read
|
||||
```
|
||||
|
||||
Or set bucket default permissions in R2 Dashboard.
|
||||
|
||||
## Security Notes
|
||||
|
||||
### API Token Permissions
|
||||
|
||||
✅ **Recommended:**
|
||||
- Object Read & Write only
|
||||
- No admin permissions needed
|
||||
- Scoped to `voicebox` bucket only
|
||||
|
||||
❌ **Avoid:**
|
||||
- Account-wide permissions
|
||||
- Account admin access
|
||||
- Worker edit permissions
|
||||
|
||||
### Secret Rotation
|
||||
|
||||
Rotate API tokens every 6-12 months:
|
||||
1. Create new API token
|
||||
2. Update GitHub secrets
|
||||
3. Verify CI still works
|
||||
4. Delete old token
|
||||
|
||||
## Maintenance
|
||||
|
||||
### Cleaning Old Versions
|
||||
|
||||
Optional: Delete old CUDA binaries to save storage costs
|
||||
|
||||
```bash
|
||||
# List all versions
|
||||
aws s3 ls s3://voicebox/cuda/ \
|
||||
--endpoint-url $R2_ENDPOINT
|
||||
|
||||
# Delete old version
|
||||
aws s3 rm s3://voicebox/cuda/v0.1.0/ \
|
||||
--recursive \
|
||||
--endpoint-url $R2_ENDPOINT
|
||||
```
|
||||
|
||||
### Monitoring
|
||||
|
||||
Set up Cloudflare notifications:
|
||||
- Storage approaching limits
|
||||
- Unusual traffic patterns
|
||||
- High operation counts
|
||||
|
||||
## Next Steps
|
||||
|
||||
1. ✅ Bucket configured
|
||||
2. ⏳ Add GitHub secrets (R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, R2_ENDPOINT)
|
||||
3. ⏳ Test local upload
|
||||
4. ⏳ Push branch and create test release
|
||||
5. ⏳ Verify CUDA binary accessible from downloads.voicebox.sh
|
||||
6. ⏳ Implement frontend download manager
|
||||
|
||||
---
|
||||
|
||||
**Status**: Ready for testing
|
||||
**Cost**: ~$0.04/month
|
||||
**Bandwidth**: Free (unlimited)
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@voicebox/landing",
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"description": "Landing page for voicebox.sh",
|
||||
"scripts": {
|
||||
"dev": "bun --bun next dev --turbo",
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import type { Metadata } from 'next';
|
||||
import { Inter } from 'next/font/google';
|
||||
import './globals.css';
|
||||
import { Banner } from '@/components/Banner';
|
||||
import { Footer } from '@/components/Footer';
|
||||
import { Header } from '@/components/Header';
|
||||
|
||||
@@ -31,6 +32,7 @@ export default function RootLayout({ children }: { children: React.ReactNode })
|
||||
<html lang="en" suppressHydrationWarning className="dark">
|
||||
<body className={inter.variable}>
|
||||
<div className="relative min-h-screen bg-background font-sans flex flex-col">
|
||||
<Banner />
|
||||
<Header />
|
||||
<main className="container mx-auto px-4 sm:px-6 md:px-4 flex-1 py-4 sm:py-6 md:py-0">
|
||||
{children}
|
||||
|
||||
@@ -239,8 +239,9 @@ export default function Home() {
|
||||
<div className="space-y-6 text-lg text-foreground/80 text-center">
|
||||
<p>
|
||||
Voicebox is a <strong>local-first voice cloning studio</strong> with DAW-like features
|
||||
for professional voice synthesis. Think of it as the <strong>Ollama for voice</strong>{' '}
|
||||
— download models, clone voices, and generate speech entirely on your machine.
|
||||
for professional voice synthesis. Think of it as a{' '}
|
||||
<strong>local, free and open-source alternative to ElevenLabs</strong> — download
|
||||
models, clone voices, and generate speech entirely on your machine.
|
||||
</p>
|
||||
<p>
|
||||
Unlike cloud services that lock your voice data behind subscriptions, Voicebox gives
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
import { ArrowRight } from 'lucide-react';
|
||||
|
||||
export function Banner() {
|
||||
return (
|
||||
<div className="bg-primary/[0.06] border-b border-border backdrop-blur-sm">
|
||||
<div className="container mx-auto px-4">
|
||||
<div className="flex items-center justify-center h-10 text-sm">
|
||||
<a
|
||||
href="https://spacebot.sh"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="flex items-center gap-2 text-muted-foreground hover:text-foreground transition-colors group"
|
||||
>
|
||||
<span>
|
||||
Also by the creator of Voicebox:{' '}
|
||||
<strong className="text-foreground/90">Spacebot</strong>, an AI agent OS for teams.
|
||||
Connect Discord, Slack, or Telegram in one click.
|
||||
</span>
|
||||
<ArrowRight className="h-3.5 w-3.5 transition-transform group-hover:translate-x-0.5" />
|
||||
</a>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "voicebox",
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"private": true,
|
||||
"workspaces": [
|
||||
"app",
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
uvicorn
|
||||
fastapi
|
||||
sqlalchemy
|
||||
torch
|
||||
torchvision
|
||||
soundfile
|
||||
librosa
|
||||
python-multipart
|
||||
huggingface_hub
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@voicebox/tauri",
|
||||
"private": true,
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "voicebox"
|
||||
version = "0.1.12"
|
||||
version = "0.1.13"
|
||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||
authors = ["you"]
|
||||
license = ""
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
use crate::audio_capture::AudioCaptureState;
|
||||
|
||||
pub async fn start_capture(
|
||||
state: &AudioCaptureState,
|
||||
max_duration_secs: u32,
|
||||
) -> Result<(), String> {
|
||||
todo!("implement Linux audio capture")
|
||||
}
|
||||
|
||||
pub async fn stop_capture(state: &AudioCaptureState) -> Result<String, String> {
|
||||
todo!("implement Linux audio capture stop")
|
||||
}
|
||||
|
||||
pub fn is_supported() -> bool {
|
||||
false
|
||||
}
|
||||
@@ -2,11 +2,15 @@
|
||||
mod macos;
|
||||
#[cfg(target_os = "windows")]
|
||||
mod windows;
|
||||
#[cfg(target_os = "linux")]
|
||||
mod linux;
|
||||
|
||||
#[cfg(target_os = "macos")]
|
||||
pub use macos::*;
|
||||
#[cfg(target_os = "windows")]
|
||||
pub use windows::*;
|
||||
#[cfg(target_os = "linux")]
|
||||
pub use linux::*;
|
||||
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"$schema": "https://schema.tauri.app/config/2",
|
||||
"productName": "Voicebox",
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"identifier": "sh.voicebox.app",
|
||||
"build": {
|
||||
"beforeDevCommand": "bun run dev",
|
||||
@@ -12,7 +12,7 @@
|
||||
"bundle": {
|
||||
"active": true,
|
||||
"targets": "all",
|
||||
"createUpdaterArtifacts": true,
|
||||
"createUpdaterArtifacts": false,
|
||||
"externalBin": ["binaries/voicebox-server"],
|
||||
"icon": [
|
||||
"icons/32x32.png",
|
||||
|
||||
@@ -2,29 +2,25 @@ import type { PlatformFilesystem, FileFilter } from '@/platform/types';
|
||||
|
||||
export const tauriFilesystem: PlatformFilesystem = {
|
||||
async saveFile(filename: string, blob: Blob, filters?: FileFilter[]) {
|
||||
try {
|
||||
const { save } = await import('@tauri-apps/plugin-dialog');
|
||||
const filePath = await save({
|
||||
defaultPath: filename,
|
||||
filters: filters || [],
|
||||
});
|
||||
const { save } = await import('@tauri-apps/plugin-dialog');
|
||||
const { writeFile } = await import('@tauri-apps/plugin-fs');
|
||||
|
||||
if (filePath) {
|
||||
const { writeBinaryFile } = await import('@tauri-apps/plugin-fs');
|
||||
const arrayBuffer = await blob.arrayBuffer();
|
||||
await writeBinaryFile(filePath, new Uint8Array(arrayBuffer));
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Failed to use Tauri dialog, falling back to browser download:', error);
|
||||
// Fall back to browser download if Tauri dialog fails
|
||||
const url = window.URL.createObjectURL(blob);
|
||||
const a = document.createElement('a');
|
||||
a.href = url;
|
||||
a.download = filename;
|
||||
document.body.appendChild(a);
|
||||
a.click();
|
||||
window.URL.revokeObjectURL(url);
|
||||
document.body.removeChild(a);
|
||||
const filePath = await save({
|
||||
defaultPath: filename,
|
||||
filters: filters || [],
|
||||
});
|
||||
|
||||
if (!filePath) return; // User cancelled the dialog
|
||||
|
||||
const resolvedPath = typeof filePath === 'string'
|
||||
? filePath
|
||||
: (filePath as { path: string }).path;
|
||||
|
||||
if (!resolvedPath) {
|
||||
throw new Error('Failed to resolve save path from dialog');
|
||||
}
|
||||
|
||||
const arrayBuffer = await blob.arrayBuffer();
|
||||
await writeFile(resolvedPath, new Uint8Array(arrayBuffer));
|
||||
},
|
||||
};
|
||||
|
||||
@@ -35,15 +35,5 @@ export default defineConfig({
|
||||
minify: !process.env.TAURI_DEBUG,
|
||||
sourcemap: !!process.env.TAURI_DEBUG,
|
||||
outDir: 'dist',
|
||||
rollupOptions: {
|
||||
external: [
|
||||
'@tauri-apps/api',
|
||||
'@tauri-apps/plugin-dialog',
|
||||
'@tauri-apps/plugin-fs',
|
||||
'@tauri-apps/plugin-process',
|
||||
'@tauri-apps/plugin-shell',
|
||||
'@tauri-apps/plugin-updater',
|
||||
],
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
"""Test CUDA detection in voicebox backend"""
|
||||
import sys
|
||||
import torch
|
||||
|
||||
print("=" * 60)
|
||||
print("PyTorch CUDA Detection Test")
|
||||
print("=" * 60)
|
||||
|
||||
# Basic torch info
|
||||
print(f"\nPyTorch version: {torch.__version__}")
|
||||
print(f"CUDA available: {torch.cuda.is_available()}")
|
||||
|
||||
if torch.cuda.is_available():
|
||||
print(f"CUDA version: {torch.version.cuda}")
|
||||
print(f"GPU count: {torch.cuda.device_count()}")
|
||||
print(f"Current GPU: {torch.cuda.current_device()}")
|
||||
print(f"GPU name: {torch.cuda.get_device_name(0)}")
|
||||
print(f"GPU memory: {torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB")
|
||||
else:
|
||||
print("\nNo CUDA available - would run on CPU")
|
||||
|
||||
# Test backend device selection
|
||||
print("\n" + "=" * 60)
|
||||
print("Backend Device Selection")
|
||||
print("=" * 60)
|
||||
|
||||
# Simulate the _get_device method from pytorch_backend.py
|
||||
def _get_device() -> str:
|
||||
"""Get the best available device."""
|
||||
if torch.cuda.is_available():
|
||||
return "cuda"
|
||||
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
|
||||
# MPS can have issues, use CPU for stability
|
||||
return "cpu"
|
||||
return "cpu"
|
||||
|
||||
selected_device = _get_device()
|
||||
print(f"\nSelected device: {selected_device}")
|
||||
print(f"Would use dtype: {'torch.bfloat16' if selected_device != 'cpu' else 'torch.float32'}")
|
||||
|
||||
# Test actual tensor creation on device
|
||||
print("\n" + "=" * 60)
|
||||
print("Testing Tensor Creation on Device")
|
||||
print("=" * 60)
|
||||
|
||||
try:
|
||||
test_tensor = torch.randn(1000, 1000).to(selected_device)
|
||||
print(f"\n[OK] Successfully created tensor on {selected_device}")
|
||||
print(f" Tensor device: {test_tensor.device}")
|
||||
print(f" Tensor dtype: {test_tensor.dtype}")
|
||||
|
||||
# Test computation
|
||||
result = test_tensor @ test_tensor.T
|
||||
print(f"[OK] Successfully performed computation on {selected_device}")
|
||||
|
||||
if selected_device == "cuda":
|
||||
print(f"\nCUDA memory allocated: {torch.cuda.memory_allocated() / 1024**2:.2f} MB")
|
||||
print(f"CUDA memory reserved: {torch.cuda.memory_reserved() / 1024**2:.2f} MB")
|
||||
|
||||
except Exception as e:
|
||||
print(f"\n[ERROR] {e}")
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("Summary")
|
||||
print("=" * 60)
|
||||
|
||||
if selected_device == "cuda":
|
||||
print("\n[SUCCESS] CUDA IS WORKING!")
|
||||
print(" The backend will use your NVIDIA GPU for inference")
|
||||
print(f" GPU: {torch.cuda.get_device_name(0)}")
|
||||
print(f" This will be significantly faster than CPU")
|
||||
else:
|
||||
print("\n[FAIL] CUDA is not available")
|
||||
print(" The backend will use CPU for inference")
|
||||
print(" This will be slower than GPU")
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
+2
-1
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@voicebox/web",
|
||||
"private": true,
|
||||
"version": "0.1.12",
|
||||
"version": "0.1.13",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
@@ -21,6 +21,7 @@
|
||||
"@types/react-dom": "^18.3.0",
|
||||
"@typescript-eslint/eslint-plugin": "^7.0.0",
|
||||
"@typescript-eslint/parser": "^7.0.0",
|
||||
"@tailwindcss/vite": "^4.0.0",
|
||||
"@vitejs/plugin-react": "^4.3.0",
|
||||
"eslint": "^8.57.0",
|
||||
"eslint-plugin-react-hooks": "^4.6.0",
|
||||
|
||||
+2
-1
@@ -1,9 +1,10 @@
|
||||
import path from 'node:path';
|
||||
import react from '@vitejs/plugin-react';
|
||||
import tailwindcss from '@tailwindcss/vite';
|
||||
import { defineConfig } from 'vite';
|
||||
|
||||
export default defineConfig({
|
||||
plugins: [react()],
|
||||
plugins: [react(), tailwindcss()],
|
||||
resolve: {
|
||||
alias: {
|
||||
'@': path.resolve(__dirname, '../app/src'),
|
||||
|
||||
Reference in New Issue
Block a user