mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 15:15:27 -07:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1b66a528d1 | ||
|
|
bef4092e6e | ||
|
|
9654f7b642 |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[bumpversion]
|
[bumpversion]
|
||||||
current_version = 0.1.10
|
current_version = 0.1.11
|
||||||
commit = True
|
commit = True
|
||||||
tag = True
|
tag = True
|
||||||
tag_name = v{new_version}
|
tag_name = v{new_version}
|
||||||
|
|||||||
@@ -68,6 +68,7 @@ Unlike cloud services that lock your voice data behind subscriptions, Voicebox g
|
|||||||
- **Model flexibility** — currently powered by Qwen3-TTS, with support for XTTS, Bark, and other models coming soon
|
- **Model flexibility** — currently powered by Qwen3-TTS, with support for XTTS, Bark, and other models coming soon
|
||||||
- **API-first** — use the desktop app or integrate voice synthesis into your own projects
|
- **API-first** — use the desktop app or integrate voice synthesis into your own projects
|
||||||
- **Native performance** — built with Tauri (Rust), not Electron
|
- **Native performance** — built with Tauri (Rust), not Electron
|
||||||
|
- **Super fast on Mac** — MLX backend with native Metal acceleration for 4-5x faster inference on Apple Silicon
|
||||||
|
|
||||||
Download a voice model, clone any voice from a few seconds of audio, and compose multi-voice projects with studio-grade editing tools. No Python install required, no cloud dependency, no limits.
|
Download a voice model, clone any voice from a few seconds of audio, and compose multi-voice projects with studio-grade editing tools. No Python install required, no cloud dependency, no limits.
|
||||||
|
|
||||||
@@ -97,6 +98,7 @@ Powered by Alibaba's **Qwen3-TTS** — a breakthrough model that achieves near-p
|
|||||||
- **Instant cloning** — Upload a sample, get a voice profile
|
- **Instant cloning** — Upload a sample, get a voice profile
|
||||||
- **High fidelity** — Natural prosody, emotion, and cadence
|
- **High fidelity** — Natural prosody, emotion, and cadence
|
||||||
- **Multi-language** — English, Chinese, and more coming
|
- **Multi-language** — English, Chinese, and more coming
|
||||||
|
- **Lightning fast on Mac** — MLX backend leverages Apple Silicon's Neural Engine for super fast generation
|
||||||
|
|
||||||
### Voice Profile Management
|
### Voice Profile Management
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/app",
|
"name": "@voicebox/app",
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
import { Download, Edit, Mic, Trash2 } from 'lucide-react';
|
import { Download, Edit, Mic, Trash2 } from 'lucide-react';
|
||||||
import { useState } from 'react';
|
import { useState } from 'react';
|
||||||
import { useServerStore } from '@/stores/serverStore';
|
|
||||||
import { Badge } from '@/components/ui/badge';
|
import { Badge } from '@/components/ui/badge';
|
||||||
import { Button } from '@/components/ui/button';
|
import { Button } from '@/components/ui/button';
|
||||||
import { Card, CardContent, CardHeader, CardTitle } from '@/components/ui/card';
|
import { Card, CardContent, CardHeader, CardTitle } from '@/components/ui/card';
|
||||||
@@ -16,6 +15,7 @@ import {
|
|||||||
import type { VoiceProfileResponse } from '@/lib/api/types';
|
import type { VoiceProfileResponse } from '@/lib/api/types';
|
||||||
import { useDeleteProfile, useExportProfile } from '@/lib/hooks/useProfiles';
|
import { useDeleteProfile, useExportProfile } from '@/lib/hooks/useProfiles';
|
||||||
import { cn } from '@/lib/utils/cn';
|
import { cn } from '@/lib/utils/cn';
|
||||||
|
import { useServerStore } from '@/stores/serverStore';
|
||||||
import { useUIStore } from '@/stores/uiStore';
|
import { useUIStore } from '@/stores/uiStore';
|
||||||
|
|
||||||
interface ProfileCardProps {
|
interface ProfileCardProps {
|
||||||
@@ -35,9 +35,7 @@ export function ProfileCard({ profile }: ProfileCardProps) {
|
|||||||
|
|
||||||
const isSelected = selectedProfileId === profile.id;
|
const isSelected = selectedProfileId === profile.id;
|
||||||
|
|
||||||
const avatarUrl = profile.avatar_path
|
const avatarUrl = profile.avatar_path ? `${serverUrl}/profiles/${profile.id}/avatar` : null;
|
||||||
? `${serverUrl}/profiles/${profile.id}/avatar`
|
|
||||||
: null;
|
|
||||||
|
|
||||||
const handleSelect = () => {
|
const handleSelect = () => {
|
||||||
setSelectedProfileId(isSelected ? null : profile.id);
|
setSelectedProfileId(isSelected ? null : profile.id);
|
||||||
@@ -81,7 +79,7 @@ export function ProfileCard({ profile }: ProfileCardProps) {
|
|||||||
alt={`${profile.name} avatar`}
|
alt={`${profile.name} avatar`}
|
||||||
className={cn(
|
className={cn(
|
||||||
'h-full w-full object-cover transition-all duration-200',
|
'h-full w-full object-cover transition-all duration-200',
|
||||||
!isSelected && 'grayscale'
|
!isSelected && 'grayscale',
|
||||||
)}
|
)}
|
||||||
onError={() => setAvatarError(true)}
|
onError={() => setAvatarError(true)}
|
||||||
/>
|
/>
|
||||||
|
|||||||
@@ -45,8 +45,8 @@ import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
|||||||
import { useTranscription } from '@/lib/hooks/useTranscription';
|
import { useTranscription } from '@/lib/hooks/useTranscription';
|
||||||
import { isTauri } from '@/lib/tauri';
|
import { isTauri } from '@/lib/tauri';
|
||||||
import { formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
import { formatAudioDuration, getAudioDuration } from '@/lib/utils/audio';
|
||||||
import { type ProfileFormDraft, useUIStore } from '@/stores/uiStore';
|
|
||||||
import { useServerStore } from '@/stores/serverStore';
|
import { useServerStore } from '@/stores/serverStore';
|
||||||
|
import { type ProfileFormDraft, useUIStore } from '@/stores/uiStore';
|
||||||
import { AudioSampleRecording } from './AudioSampleRecording';
|
import { AudioSampleRecording } from './AudioSampleRecording';
|
||||||
import { AudioSampleSystem } from './AudioSampleSystem';
|
import { AudioSampleSystem } from './AudioSampleSystem';
|
||||||
import { AudioSampleUpload } from './AudioSampleUpload';
|
import { AudioSampleUpload } from './AudioSampleUpload';
|
||||||
@@ -427,7 +427,8 @@ export function ProfileForm() {
|
|||||||
} catch (avatarError) {
|
} catch (avatarError) {
|
||||||
toast({
|
toast({
|
||||||
title: 'Avatar upload failed',
|
title: 'Avatar upload failed',
|
||||||
description: avatarError instanceof Error ? avatarError.message : 'Failed to upload avatar',
|
description:
|
||||||
|
avatarError instanceof Error ? avatarError.message : 'Failed to upload avatar',
|
||||||
variant: 'destructive',
|
variant: 'destructive',
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -520,7 +521,8 @@ export function ProfileForm() {
|
|||||||
} catch (avatarError) {
|
} catch (avatarError) {
|
||||||
toast({
|
toast({
|
||||||
title: 'Avatar upload failed',
|
title: 'Avatar upload failed',
|
||||||
description: avatarError instanceof Error ? avatarError.message : 'Failed to upload avatar',
|
description:
|
||||||
|
avatarError instanceof Error ? avatarError.message : 'Failed to upload avatar',
|
||||||
variant: 'destructive',
|
variant: 'destructive',
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -1,3 +1,3 @@
|
|||||||
# Backend package
|
# Backend package
|
||||||
|
|
||||||
__version__ = "0.1.10"
|
__version__ = "0.1.11"
|
||||||
|
|||||||
@@ -341,21 +341,34 @@ class MLXSTTBackend:
|
|||||||
def _load_model_sync(self, model_size: str):
|
def _load_model_sync(self, model_size: str):
|
||||||
"""Synchronous model loading."""
|
"""Synchronous model loading."""
|
||||||
try:
|
try:
|
||||||
from mlx_audio.asr import load
|
# IMPORTANT: Set up progress tracking BEFORE importing mlx_audio
|
||||||
|
# This ensures tqdm is patched before any HuggingFace Hub imports
|
||||||
# MLX Whisper model naming
|
|
||||||
model_name = f"mlx-community/whisper-{model_size}"
|
|
||||||
|
|
||||||
# Set up progress tracking
|
|
||||||
progress_manager = get_progress_manager()
|
progress_manager = get_progress_manager()
|
||||||
progress_model_name = f"whisper-{model_size}"
|
progress_model_name = f"whisper-{model_size}"
|
||||||
|
|
||||||
|
# Set up progress callback and tracker
|
||||||
|
progress_callback = create_hf_progress_callback(progress_model_name, progress_manager)
|
||||||
|
tracker = HFProgressTracker(progress_callback)
|
||||||
|
|
||||||
|
# Patch tqdm BEFORE importing mlx_audio
|
||||||
|
# This is critical because mlx_audio imports huggingface_hub which imports tqdm
|
||||||
|
print("[DEBUG] Starting tqdm patch BEFORE mlx_audio import")
|
||||||
|
tracker_context = tracker.patch_download()
|
||||||
|
tracker_context.__enter__()
|
||||||
|
print("[DEBUG] tqdm patched, now importing mlx_audio")
|
||||||
|
|
||||||
|
# NOW import mlx_audio - it will use our patched tqdm
|
||||||
|
from mlx_audio.stt import load
|
||||||
|
|
||||||
|
# MLX Whisper uses the standard OpenAI models
|
||||||
|
model_name = f"openai/whisper-{model_size}"
|
||||||
|
|
||||||
# Start tracking download task
|
# Start tracking download task
|
||||||
task_manager = get_task_manager()
|
task_manager = get_task_manager()
|
||||||
task_manager.start_download(progress_model_name)
|
task_manager.start_download(progress_model_name)
|
||||||
|
|
||||||
print(f"Loading MLX Whisper model {model_size}...")
|
print(f"Loading MLX Whisper model {model_size}...")
|
||||||
|
|
||||||
# Initialize progress state
|
# Initialize progress state
|
||||||
progress_manager.update_progress(
|
progress_manager.update_progress(
|
||||||
model_name=progress_model_name,
|
model_name=progress_model_name,
|
||||||
@@ -364,14 +377,13 @@ class MLXSTTBackend:
|
|||||||
filename="",
|
filename="",
|
||||||
status="downloading",
|
status="downloading",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Set up progress callback
|
# Load the model (tqdm is already patched from above)
|
||||||
progress_callback = create_hf_progress_callback(progress_model_name, progress_manager)
|
try:
|
||||||
tracker = HFProgressTracker(progress_callback)
|
|
||||||
|
|
||||||
# Use progress tracker during download
|
|
||||||
with tracker.patch_download():
|
|
||||||
self.model = load(model_name)
|
self.model = load(model_name)
|
||||||
|
finally:
|
||||||
|
# Exit the patch context
|
||||||
|
tracker_context.__exit__(None, None, None)
|
||||||
|
|
||||||
self.model_size = model_size
|
self.model_size = model_size
|
||||||
|
|
||||||
@@ -412,34 +424,35 @@ class MLXSTTBackend:
|
|||||||
) -> str:
|
) -> str:
|
||||||
"""
|
"""
|
||||||
Transcribe audio to text.
|
Transcribe audio to text.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
audio_path: Path to audio file
|
audio_path: Path to audio file
|
||||||
language: Optional language hint (en or zh)
|
language: Optional language hint (en or zh)
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Transcribed text
|
Transcribed text
|
||||||
"""
|
"""
|
||||||
await self.load_model_async(None)
|
await self.load_model_async(None)
|
||||||
|
|
||||||
def _transcribe_sync():
|
def _transcribe_sync():
|
||||||
"""Run synchronous transcription in thread pool."""
|
"""Run synchronous transcription in thread pool."""
|
||||||
# Load audio
|
# MLX Whisper transcription using generate method
|
||||||
audio, sr = load_audio(audio_path, sample_rate=16000)
|
# The generate method accepts audio path directly
|
||||||
|
decode_options = {}
|
||||||
# MLX Whisper transcription
|
if language:
|
||||||
# The API may vary - check mlx-audio documentation
|
decode_options["language"] = language
|
||||||
# For now, assuming similar API to PyTorch Whisper
|
|
||||||
result = self.model.transcribe(audio, language=language)
|
result = self.model.generate(str(audio_path), **decode_options)
|
||||||
|
|
||||||
# Extract text from result (format may vary)
|
# Extract text from result
|
||||||
if isinstance(result, str):
|
if isinstance(result, str):
|
||||||
return result.strip()
|
return result.strip()
|
||||||
elif isinstance(result, dict):
|
elif isinstance(result, dict):
|
||||||
return result.get("text", "").strip()
|
return result.get("text", "").strip()
|
||||||
|
elif hasattr(result, "text"):
|
||||||
|
return result.text.strip()
|
||||||
else:
|
else:
|
||||||
# Try to get text attribute
|
|
||||||
return str(result).strip()
|
return str(result).strip()
|
||||||
|
|
||||||
# Run blocking transcription in thread pool
|
# Run blocking transcription in thread pool
|
||||||
return await asyncio.to_thread(_transcribe_sync)
|
return await asyncio.to_thread(_transcribe_sync)
|
||||||
|
|||||||
@@ -85,21 +85,31 @@ class PyTorchTTSBackend:
|
|||||||
def _load_model_sync(self, model_size: str):
|
def _load_model_sync(self, model_size: str):
|
||||||
"""Synchronous model loading."""
|
"""Synchronous model loading."""
|
||||||
try:
|
try:
|
||||||
from qwen_tts import Qwen3TTSModel
|
# IMPORTANT: Set up progress tracking BEFORE importing qwen_tts
|
||||||
|
# This ensures tqdm is patched before any HuggingFace Hub imports
|
||||||
# Get model path (local or HuggingFace Hub ID)
|
|
||||||
model_path = self._get_model_path(model_size)
|
|
||||||
|
|
||||||
# Set up progress tracking
|
|
||||||
progress_manager = get_progress_manager()
|
progress_manager = get_progress_manager()
|
||||||
model_name = f"qwen-tts-{model_size}"
|
model_name = f"qwen-tts-{model_size}"
|
||||||
|
|
||||||
|
# Set up progress callback and tracker
|
||||||
|
progress_callback = create_hf_progress_callback(model_name, progress_manager)
|
||||||
|
tracker = HFProgressTracker(progress_callback)
|
||||||
|
|
||||||
|
# Patch tqdm BEFORE importing qwen_tts
|
||||||
|
tracker_context = tracker.patch_download()
|
||||||
|
tracker_context.__enter__()
|
||||||
|
|
||||||
|
# NOW import qwen_tts - it will use our patched tqdm
|
||||||
|
from qwen_tts import Qwen3TTSModel
|
||||||
|
|
||||||
|
# Get model path (local or HuggingFace Hub ID)
|
||||||
|
model_path = self._get_model_path(model_size)
|
||||||
|
|
||||||
print(f"Loading TTS model {model_size} on {self.device}...")
|
print(f"Loading TTS model {model_size} on {self.device}...")
|
||||||
|
|
||||||
# Start tracking download task
|
# Start tracking download task
|
||||||
task_manager = get_task_manager()
|
task_manager = get_task_manager()
|
||||||
task_manager.start_download(model_name)
|
task_manager.start_download(model_name)
|
||||||
|
|
||||||
# Initialize progress state to show download has started
|
# Initialize progress state to show download has started
|
||||||
progress_manager.update_progress(
|
progress_manager.update_progress(
|
||||||
model_name=model_name,
|
model_name=model_name,
|
||||||
@@ -108,19 +118,17 @@ class PyTorchTTSBackend:
|
|||||||
filename="",
|
filename="",
|
||||||
status="downloading",
|
status="downloading",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Set up progress callback
|
# Load the model (tqdm is already patched from above)
|
||||||
progress_callback = create_hf_progress_callback(model_name, progress_manager)
|
try:
|
||||||
tracker = HFProgressTracker(progress_callback)
|
|
||||||
|
|
||||||
# Use progress tracker during download
|
|
||||||
with tracker.patch_download():
|
|
||||||
# Load the model - downloads will happen automatically with progress tracking
|
|
||||||
self.model = Qwen3TTSModel.from_pretrained(
|
self.model = Qwen3TTSModel.from_pretrained(
|
||||||
model_path,
|
model_path,
|
||||||
device_map=self.device,
|
device_map=self.device,
|
||||||
torch_dtype=torch.float32 if self.device == "cpu" else torch.bfloat16,
|
torch_dtype=torch.float32 if self.device == "cpu" else torch.bfloat16,
|
||||||
)
|
)
|
||||||
|
finally:
|
||||||
|
# Exit the patch context
|
||||||
|
tracker_context.__exit__(None, None, None)
|
||||||
|
|
||||||
# Mark as complete
|
# Mark as complete
|
||||||
progress_manager.mark_complete(model_name)
|
progress_manager.mark_complete(model_name)
|
||||||
@@ -314,40 +322,61 @@ class PyTorchSTTBackend:
|
|||||||
async def load_model_async(self, model_size: Optional[str] = None):
|
async def load_model_async(self, model_size: Optional[str] = None):
|
||||||
"""
|
"""
|
||||||
Lazy load the Whisper model.
|
Lazy load the Whisper model.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
model_size: Model size (tiny, base, small, medium, large)
|
model_size: Model size (tiny, base, small, medium, large)
|
||||||
"""
|
"""
|
||||||
|
print(f"[DEBUG] load_model_async called with size: {model_size}")
|
||||||
if model_size is None:
|
if model_size is None:
|
||||||
model_size = self.model_size
|
model_size = self.model_size
|
||||||
|
|
||||||
|
print(f"[DEBUG] Model already loaded? {self.model is not None}, current size: {self.model_size}, requested: {model_size}")
|
||||||
if self.model is not None and self.model_size == model_size:
|
if self.model is not None and self.model_size == model_size:
|
||||||
|
print(f"[DEBUG] Early return - model already loaded")
|
||||||
return
|
return
|
||||||
|
|
||||||
|
print(f"[DEBUG] Calling asyncio.to_thread for _load_model_sync")
|
||||||
# Run blocking load in thread pool
|
# Run blocking load in thread pool
|
||||||
await asyncio.to_thread(self._load_model_sync, model_size)
|
await asyncio.to_thread(self._load_model_sync, model_size)
|
||||||
|
print(f"[DEBUG] asyncio.to_thread completed")
|
||||||
|
|
||||||
# Alias for compatibility
|
# Alias for compatibility
|
||||||
load_model = load_model_async
|
load_model = load_model_async
|
||||||
|
|
||||||
def _load_model_sync(self, model_size: str):
|
def _load_model_sync(self, model_size: str):
|
||||||
"""Synchronous model loading."""
|
"""Synchronous model loading."""
|
||||||
|
print(f"[DEBUG] _load_model_sync called for Whisper {model_size}")
|
||||||
try:
|
try:
|
||||||
from transformers import WhisperProcessor, WhisperForConditionalGeneration
|
# IMPORTANT: Set up progress tracking BEFORE importing transformers
|
||||||
|
# This ensures tqdm is patched before any HuggingFace Hub imports
|
||||||
model_name = f"openai/whisper-{model_size}"
|
|
||||||
|
|
||||||
# Set up progress tracking
|
|
||||||
progress_manager = get_progress_manager()
|
progress_manager = get_progress_manager()
|
||||||
progress_model_name = f"whisper-{model_size}"
|
progress_model_name = f"whisper-{model_size}"
|
||||||
|
|
||||||
|
# Set up progress callback and tracker
|
||||||
|
progress_callback = create_hf_progress_callback(progress_model_name, progress_manager)
|
||||||
|
tracker = HFProgressTracker(progress_callback)
|
||||||
|
|
||||||
|
# Patch tqdm BEFORE importing transformers
|
||||||
|
print("[DEBUG] Starting tqdm patch BEFORE transformers import")
|
||||||
|
tracker_context = tracker.patch_download()
|
||||||
|
tracker_context.__enter__()
|
||||||
|
print("[DEBUG] tqdm patched, now importing transformers")
|
||||||
|
|
||||||
|
# NOW import transformers - it will use our patched tqdm
|
||||||
|
from transformers import WhisperProcessor, WhisperForConditionalGeneration
|
||||||
|
|
||||||
|
model_name = f"openai/whisper-{model_size}"
|
||||||
|
print(f"[DEBUG] Model name: {model_name}")
|
||||||
|
|
||||||
# Start tracking download task
|
# Start tracking download task
|
||||||
task_manager = get_task_manager()
|
task_manager = get_task_manager()
|
||||||
task_manager.start_download(progress_model_name)
|
task_manager.start_download(progress_model_name)
|
||||||
|
print(f"[DEBUG] Task manager started download")
|
||||||
|
|
||||||
print(f"Loading Whisper model {model_size} on {self.device}...")
|
print(f"Loading Whisper model {model_size} on {self.device}...")
|
||||||
|
|
||||||
# Initialize progress state to show download has started
|
# Initialize progress state to show download has started
|
||||||
|
print(f"[DEBUG] Calling update_progress...")
|
||||||
progress_manager.update_progress(
|
progress_manager.update_progress(
|
||||||
model_name=progress_model_name,
|
model_name=progress_model_name,
|
||||||
current=0,
|
current=0,
|
||||||
@@ -355,15 +384,15 @@ class PyTorchSTTBackend:
|
|||||||
filename="",
|
filename="",
|
||||||
status="downloading",
|
status="downloading",
|
||||||
)
|
)
|
||||||
|
print(f"[DEBUG] update_progress called, listeners: {len(progress_manager._listeners.get(progress_model_name, []))}")
|
||||||
# Set up progress callback
|
|
||||||
progress_callback = create_hf_progress_callback(progress_model_name, progress_manager)
|
# Load models (tqdm is already patched from above)
|
||||||
tracker = HFProgressTracker(progress_callback)
|
try:
|
||||||
|
|
||||||
# Use progress tracker during download
|
|
||||||
with tracker.patch_download():
|
|
||||||
self.processor = WhisperProcessor.from_pretrained(model_name)
|
self.processor = WhisperProcessor.from_pretrained(model_name)
|
||||||
self.model = WhisperForConditionalGeneration.from_pretrained(model_name)
|
self.model = WhisperForConditionalGeneration.from_pretrained(model_name)
|
||||||
|
finally:
|
||||||
|
# Exit the patch context
|
||||||
|
tracker_context.__exit__(None, None, None)
|
||||||
|
|
||||||
self.model.to(self.device)
|
self.model.to(self.device)
|
||||||
self.model_size = model_size
|
self.model_size = model_size
|
||||||
|
|||||||
@@ -80,7 +80,7 @@ def build_server():
|
|||||||
'--hidden-import', 'mlx.nn',
|
'--hidden-import', 'mlx.nn',
|
||||||
'--hidden-import', 'mlx_audio',
|
'--hidden-import', 'mlx_audio',
|
||||||
'--hidden-import', 'mlx_audio.tts',
|
'--hidden-import', 'mlx_audio.tts',
|
||||||
'--hidden-import', 'mlx_audio.asr',
|
'--hidden-import', 'mlx_audio.stt',
|
||||||
'--collect-submodules', 'mlx',
|
'--collect-submodules', 'mlx',
|
||||||
'--collect-submodules', 'mlx_audio',
|
'--collect-submodules', 'mlx_audio',
|
||||||
# Collect MLX data files including Metal shader libraries (.metallib)
|
# Collect MLX data files including Metal shader libraries (.metallib)
|
||||||
|
|||||||
+5
-1
@@ -1393,7 +1393,11 @@ async def trigger_model_download(request: models.ModelDownloadRequest):
|
|||||||
async def download_in_background():
|
async def download_in_background():
|
||||||
"""Download model in background without blocking the HTTP request."""
|
"""Download model in background without blocking the HTTP request."""
|
||||||
try:
|
try:
|
||||||
await asyncio.to_thread(config["load_func"])
|
# Call the load function (which may be async)
|
||||||
|
result = config["load_func"]()
|
||||||
|
# If it's a coroutine, await it
|
||||||
|
if asyncio.iscoroutine(result):
|
||||||
|
await result
|
||||||
task_manager.complete_download(request.model_name)
|
task_manager.complete_download(request.model_name)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
task_manager.error_download(request.model_name, str(e))
|
task_manager.error_download(request.model_name, str(e))
|
||||||
|
|||||||
@@ -29,8 +29,9 @@ class HFProgressTracker:
|
|||||||
|
|
||||||
class TrackedTqdm(original_tqdm):
|
class TrackedTqdm(original_tqdm):
|
||||||
"""A tqdm subclass that reports progress to our tracker."""
|
"""A tqdm subclass that reports progress to our tracker."""
|
||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
|
print(f"[DEBUG TrackedTqdm] __init__ called with desc: {kwargs.get('desc', '')}")
|
||||||
# Extract filename from desc before passing to parent
|
# Extract filename from desc before passing to parent
|
||||||
desc = kwargs.get("desc", "")
|
desc = kwargs.get("desc", "")
|
||||||
if not desc and args:
|
if not desc and args:
|
||||||
@@ -79,8 +80,9 @@ class HFProgressTracker:
|
|||||||
}
|
}
|
||||||
|
|
||||||
def update(self, n=1):
|
def update(self, n=1):
|
||||||
|
print(f"[DEBUG TrackedTqdm] update called with n={n}")
|
||||||
result = super().update(n)
|
result = super().update(n)
|
||||||
|
|
||||||
# Report progress
|
# Report progress
|
||||||
with tracker._lock:
|
with tracker._lock:
|
||||||
if id(self) in tracker._active_tqdms:
|
if id(self) in tracker._active_tqdms:
|
||||||
@@ -118,11 +120,13 @@ class HFProgressTracker:
|
|||||||
@contextmanager
|
@contextmanager
|
||||||
def patch_download(self):
|
def patch_download(self):
|
||||||
"""Context manager to patch tqdm for progress tracking."""
|
"""Context manager to patch tqdm for progress tracking."""
|
||||||
|
print("[DEBUG HFProgressTracker] patch_download called")
|
||||||
try:
|
try:
|
||||||
import tqdm as tqdm_module
|
import tqdm as tqdm_module
|
||||||
|
|
||||||
# Store original tqdm class
|
# Store original tqdm class
|
||||||
self._original_tqdm_class = tqdm_module.tqdm
|
self._original_tqdm_class = tqdm_module.tqdm
|
||||||
|
print(f"[DEBUG HFProgressTracker] Original tqdm class: {self._original_tqdm_class}")
|
||||||
|
|
||||||
# Reset totals
|
# Reset totals
|
||||||
with self._lock:
|
with self._lock:
|
||||||
@@ -135,18 +139,22 @@ class HFProgressTracker:
|
|||||||
|
|
||||||
# Create our tracked tqdm class
|
# Create our tracked tqdm class
|
||||||
tracked_tqdm = self._create_tracked_tqdm_class()
|
tracked_tqdm = self._create_tracked_tqdm_class()
|
||||||
|
print(f"[DEBUG HFProgressTracker] Created TrackedTqdm class: {tracked_tqdm}")
|
||||||
|
|
||||||
# Patch tqdm.tqdm
|
# Patch tqdm.tqdm
|
||||||
tqdm_module.tqdm = tracked_tqdm
|
tqdm_module.tqdm = tracked_tqdm
|
||||||
|
print(f"[DEBUG HFProgressTracker] Patched tqdm.tqdm")
|
||||||
|
|
||||||
# Also patch tqdm.auto.tqdm if it exists (used by huggingface_hub)
|
# Also patch tqdm.auto.tqdm if it exists (used by huggingface_hub)
|
||||||
self._original_tqdm_auto = None
|
self._original_tqdm_auto = None
|
||||||
if hasattr(tqdm_module, "auto") and hasattr(tqdm_module.auto, "tqdm"):
|
if hasattr(tqdm_module, "auto") and hasattr(tqdm_module.auto, "tqdm"):
|
||||||
self._original_tqdm_auto = tqdm_module.auto.tqdm
|
self._original_tqdm_auto = tqdm_module.auto.tqdm
|
||||||
tqdm_module.auto.tqdm = tracked_tqdm
|
tqdm_module.auto.tqdm = tracked_tqdm
|
||||||
|
print(f"[DEBUG HFProgressTracker] Patched tqdm.auto.tqdm")
|
||||||
|
|
||||||
# Patch in sys.modules to catch already-imported references
|
# Patch in sys.modules to catch already-imported references
|
||||||
self._patched_modules = {}
|
self._patched_modules = {}
|
||||||
|
patched_count = 0
|
||||||
for module_name in list(sys.modules.keys()):
|
for module_name in list(sys.modules.keys()):
|
||||||
if "huggingface" in module_name or module_name.startswith("tqdm"):
|
if "huggingface" in module_name or module_name.startswith("tqdm"):
|
||||||
try:
|
try:
|
||||||
@@ -159,8 +167,11 @@ class HFProgressTracker:
|
|||||||
):
|
):
|
||||||
self._patched_modules[module_name] = attr
|
self._patched_modules[module_name] = attr
|
||||||
setattr(module, "tqdm", tracked_tqdm)
|
setattr(module, "tqdm", tracked_tqdm)
|
||||||
|
patched_count += 1
|
||||||
|
print(f"[DEBUG HFProgressTracker] Patched {module_name}.tqdm")
|
||||||
except (AttributeError, TypeError):
|
except (AttributeError, TypeError):
|
||||||
pass
|
pass
|
||||||
|
print(f"[DEBUG HFProgressTracker] Patched {patched_count} modules in sys.modules")
|
||||||
|
|
||||||
yield
|
yield
|
||||||
|
|
||||||
|
|||||||
@@ -65,7 +65,7 @@ class ProgressManager:
|
|||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Update progress for a model download.
|
Update progress for a model download.
|
||||||
|
|
||||||
Thread-safe: can be called from background threads.
|
Thread-safe: can be called from background threads.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
@@ -89,16 +89,26 @@ class ProgressManager:
|
|||||||
"status": status,
|
"status": status,
|
||||||
"timestamp": datetime.now().isoformat(),
|
"timestamp": datetime.now().isoformat(),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
print(f"[DEBUG] update_progress called: {model_name}, {progress_pct:.1f}%")
|
||||||
|
|
||||||
# Thread-safe update of progress dict
|
# Thread-safe update of progress dict
|
||||||
with self._lock:
|
with self._lock:
|
||||||
self._progress[model_name] = progress_data
|
self._progress[model_name] = progress_data
|
||||||
|
|
||||||
# Notify all listeners (thread-safe)
|
# Notify all listeners (thread-safe)
|
||||||
listener_count = len(self._listeners.get(model_name, []))
|
listener_count = len(self._listeners.get(model_name, []))
|
||||||
|
print(f"[DEBUG] Listener count for {model_name}: {listener_count}")
|
||||||
|
print(f"[DEBUG] All listeners: {list(self._listeners.keys())}")
|
||||||
|
print(f"[DEBUG] Main loop set: {self._main_loop is not None}")
|
||||||
|
if self._main_loop:
|
||||||
|
print(f"[DEBUG] Main loop running: {self._main_loop.is_running()}")
|
||||||
|
|
||||||
if listener_count > 0:
|
if listener_count > 0:
|
||||||
logger.debug(f"Notifying {listener_count} listeners for {model_name}: {progress_pct:.1f}% ({filename})")
|
logger.debug(f"Notifying {listener_count} listeners for {model_name}: {progress_pct:.1f}% ({filename})")
|
||||||
|
print(f"[DEBUG] About to notify listeners...")
|
||||||
self._notify_listeners_threadsafe(model_name, progress_data)
|
self._notify_listeners_threadsafe(model_name, progress_data)
|
||||||
|
print(f"[DEBUG] Notified listeners")
|
||||||
else:
|
else:
|
||||||
logger.debug(f"No listeners for {model_name}, progress update stored: {progress_pct:.1f}%")
|
logger.debug(f"No listeners for {model_name}, progress update stored: {progress_pct:.1f}%")
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ from PyInstaller.utils.hooks import collect_submodules
|
|||||||
from PyInstaller.utils.hooks import copy_metadata
|
from PyInstaller.utils.hooks import copy_metadata
|
||||||
|
|
||||||
datas = []
|
datas = []
|
||||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.asr']
|
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
||||||
datas += collect_data_files('qwen_tts')
|
datas += collect_data_files('qwen_tts')
|
||||||
datas += collect_data_files('mlx')
|
datas += collect_data_files('mlx')
|
||||||
datas += collect_data_files('mlx_audio')
|
datas += collect_data_files('mlx_audio')
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/landing",
|
"name": "@voicebox/landing",
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"description": "Landing page for voicebox.sh",
|
"description": "Landing page for voicebox.sh",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "bun --bun next dev --turbo",
|
"dev": "bun --bun next dev --turbo",
|
||||||
|
|||||||
+12
-10
@@ -5,7 +5,7 @@ import Image from 'next/image';
|
|||||||
import { useEffect, useState } from 'react';
|
import { useEffect, useState } from 'react';
|
||||||
import { AppleIcon, LinuxIcon, WindowsIcon } from '@/components/PlatformIcons';
|
import { AppleIcon, LinuxIcon, WindowsIcon } from '@/components/PlatformIcons';
|
||||||
import { Button } from '@/components/ui/button';
|
import { Button } from '@/components/ui/button';
|
||||||
import { Section, SectionTitle } from '@/components/ui/section';
|
import { Section } from '@/components/ui/section';
|
||||||
import { DOWNLOAD_LINKS, GITHUB_REPO } from '@/lib/constants';
|
import { DOWNLOAD_LINKS, GITHUB_REPO } from '@/lib/constants';
|
||||||
import type { DownloadLinks } from '@/lib/releases';
|
import type { DownloadLinks } from '@/lib/releases';
|
||||||
import { FeatureCard } from '../components/ui/feature-card';
|
import { FeatureCard } from '../components/ui/feature-card';
|
||||||
@@ -39,17 +39,19 @@ export default function Home() {
|
|||||||
"Powered by Alibaba's Qwen3-TTS model for exceptional voice quality and accuracy.",
|
"Powered by Alibaba's Qwen3-TTS model for exceptional voice quality and accuracy.",
|
||||||
icon: <Zap className="h-6 w-6" />,
|
icon: <Zap className="h-6 w-6" />,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
title: 'Stories Editor',
|
||||||
|
description:
|
||||||
|
'Create multi-voice narratives with a timeline-based editor. Arrange tracks, trim clips, and mix conversations.',
|
||||||
|
icon: <Code className="h-6 w-6" />,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
title: 'Multi-Sample Support',
|
title: 'Multi-Sample Support',
|
||||||
description:
|
description:
|
||||||
'Combine multiple voice samples for higher quality and more natural-sounding results.',
|
'Combine multiple voice samples for higher quality and more natural-sounding results.',
|
||||||
icon: <Code className="h-6 w-6" />,
|
icon: <Code className="h-6 w-6" />,
|
||||||
},
|
},
|
||||||
{
|
|
||||||
title: 'Smart Caching',
|
|
||||||
description: 'Instant re-generation with voice prompt caching. No need to reprocess samples.',
|
|
||||||
icon: <Zap className="h-6 w-6" />,
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
title: 'Local or Remote',
|
title: 'Local or Remote',
|
||||||
description:
|
description:
|
||||||
@@ -246,6 +248,10 @@ export default function Home() {
|
|||||||
model, clone any voice from a few seconds of audio, and compose multi-voice projects
|
model, clone any voice from a few seconds of audio, and compose multi-voice projects
|
||||||
with studio-grade editing tools.
|
with studio-grade editing tools.
|
||||||
</p>
|
</p>
|
||||||
|
<p>
|
||||||
|
Optimized for performance with <strong>Metal acceleration on Mac</strong> and{' '}
|
||||||
|
<strong>CUDA acceleration on Windows/Linux</strong> for fast, local inference.
|
||||||
|
</p>
|
||||||
<p className="text-foreground/60">No Python install required.</p>
|
<p className="text-foreground/60">No Python install required.</p>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
@@ -281,10 +287,6 @@ export default function Home() {
|
|||||||
|
|
||||||
{/* Features Section */}
|
{/* Features Section */}
|
||||||
<Section id="features">
|
<Section id="features">
|
||||||
<SectionTitle className="mb-4 text-center">Features</SectionTitle>
|
|
||||||
<p className="text-sm text-muted-foreground mb-8 text-center max-w-2xl mx-auto">
|
|
||||||
Everything you need for professional voice cloning in a desktop app.
|
|
||||||
</p>
|
|
||||||
<div className="grid grid-cols-1 md:grid-cols-2 lg:grid-cols-3 gap-4 sm:gap-6">
|
<div className="grid grid-cols-1 md:grid-cols-2 lg:grid-cols-3 gap-4 sm:gap-6">
|
||||||
{features.map((feature) => (
|
{features.map((feature) => (
|
||||||
<FeatureCard
|
<FeatureCard
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "voicebox",
|
"name": "voicebox",
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"private": true,
|
"private": true,
|
||||||
"workspaces": [
|
"workspaces": [
|
||||||
"app",
|
"app",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/tauri",
|
"name": "@voicebox/tauri",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Generated
+1
-1
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.1.9"
|
version = "0.1.11"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"core-foundation-sys",
|
"core-foundation-sys",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.1.10"
|
version = "0.1.11"
|
||||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||||
authors = ["you"]
|
authors = ["you"]
|
||||||
license = ""
|
license = ""
|
||||||
|
|||||||
Binary file not shown.
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://schema.tauri.app/config/2",
|
"$schema": "https://schema.tauri.app/config/2",
|
||||||
"productName": "Voicebox",
|
"productName": "Voicebox",
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"identifier": "sh.voicebox.app",
|
"identifier": "sh.voicebox.app",
|
||||||
"build": {
|
"build": {
|
||||||
"beforeDevCommand": "bun run dev",
|
"beforeDevCommand": "bun run dev",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/web",
|
"name": "@voicebox/web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.10",
|
"version": "0.1.11",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Reference in New Issue
Block a user