mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 07:05:14 -07:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
535cf362de | ||
|
|
f1116e05a6 | ||
|
|
f9aca9d418 | ||
|
|
d913a9ae2a | ||
|
|
36031a0df5 | ||
|
|
e59f86aa63 | ||
|
|
b59c0f44e5 | ||
|
|
f8c5e54962 | ||
|
|
8b67faf96d |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
[bumpversion]
|
[bumpversion]
|
||||||
current_version = 0.1.1
|
current_version = 0.1.2
|
||||||
commit = True
|
commit = True
|
||||||
tag = True
|
tag = True
|
||||||
tag_name = v{new_version}
|
tag_name = v{new_version}
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/app",
|
"name": "@voicebox/app",
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
|
|||||||
@@ -71,11 +71,23 @@ export function UpdateStatus() {
|
|||||||
|
|
||||||
{status.downloading && (
|
{status.downloading && (
|
||||||
<div className="space-y-2">
|
<div className="space-y-2">
|
||||||
<div className="flex items-center gap-2 text-sm">
|
<div className="flex items-center justify-between text-sm">
|
||||||
<Download className="h-4 w-4" />
|
<div className="flex items-center gap-2">
|
||||||
Downloading update...
|
<Download className="h-4 w-4" />
|
||||||
|
Downloading update...
|
||||||
|
</div>
|
||||||
|
{status.downloadProgress !== undefined && (
|
||||||
|
<span className="text-muted-foreground">
|
||||||
|
{status.downloadProgress}%
|
||||||
|
</span>
|
||||||
|
)}
|
||||||
</div>
|
</div>
|
||||||
<Progress />
|
<Progress value={status.downloadProgress} />
|
||||||
|
{status.downloadedBytes !== undefined && status.totalBytes !== undefined && status.totalBytes > 0 && (
|
||||||
|
<div className="text-xs text-muted-foreground">
|
||||||
|
{(status.downloadedBytes / 1024 / 1024).toFixed(1)} MB / {(status.totalBytes / 1024 / 1024).toFixed(1)} MB
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
</div>
|
</div>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
import { Mic, Square, Play, Pause } from 'lucide-react';
|
import { Mic, Pause, Play, Square } from 'lucide-react';
|
||||||
import { Button } from '@/components/ui/button';
|
import { Button } from '@/components/ui/button';
|
||||||
import { FormControl, FormItem, FormLabel, FormMessage } from '@/components/ui/form';
|
import { FormControl, FormItem, FormLabel, FormMessage } from '@/components/ui/form';
|
||||||
import { formatAudioDuration } from '@/lib/utils/audio';
|
import { formatAudioDuration } from '@/lib/utils/audio';
|
||||||
@@ -35,12 +35,7 @@ export function AudioSampleRecording({
|
|||||||
<div className="space-y-4">
|
<div className="space-y-4">
|
||||||
{!isRecording && !file && (
|
{!isRecording && !file && (
|
||||||
<div className="flex flex-col items-center justify-center gap-4 p-4 border-2 border-dashed rounded-lg min-h-[180px]">
|
<div className="flex flex-col items-center justify-center gap-4 p-4 border-2 border-dashed rounded-lg min-h-[180px]">
|
||||||
<Button
|
<Button type="button" onClick={onStart} size="lg" className="flex items-center gap-2">
|
||||||
type="button"
|
|
||||||
onClick={onStart}
|
|
||||||
size="lg"
|
|
||||||
className="flex items-center gap-2"
|
|
||||||
>
|
|
||||||
<Mic className="h-5 w-5" />
|
<Mic className="h-5 w-5" />
|
||||||
Start Recording
|
Start Recording
|
||||||
</Button>
|
</Button>
|
||||||
@@ -81,16 +76,9 @@ export function AudioSampleRecording({
|
|||||||
<Mic className="h-5 w-5 text-primary" />
|
<Mic className="h-5 w-5 text-primary" />
|
||||||
<span className="font-medium">Recording complete</span>
|
<span className="font-medium">Recording complete</span>
|
||||||
</div>
|
</div>
|
||||||
<p className="text-sm text-muted-foreground text-center">
|
<p className="text-sm text-muted-foreground text-center">File: {file.name}</p>
|
||||||
File: {file.name}
|
|
||||||
</p>
|
|
||||||
<div className="flex gap-2">
|
<div className="flex gap-2">
|
||||||
<Button
|
<Button type="button" size="icon" variant="outline" onClick={onPlayPause}>
|
||||||
type="button"
|
|
||||||
size="icon"
|
|
||||||
variant="outline"
|
|
||||||
onClick={onPlayPause}
|
|
||||||
>
|
|
||||||
{isPlaying ? <Pause className="h-4 w-4" /> : <Play className="h-4 w-4" />}
|
{isPlaying ? <Pause className="h-4 w-4" /> : <Play className="h-4 w-4" />}
|
||||||
</Button>
|
</Button>
|
||||||
<Button
|
<Button
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
import { Monitor, Square, Play, Pause, Mic } from 'lucide-react';
|
import { Mic, Monitor, Pause, Play, Square } from 'lucide-react';
|
||||||
import { Button } from '@/components/ui/button';
|
import { Button } from '@/components/ui/button';
|
||||||
import { FormControl, FormItem, FormLabel, FormMessage } from '@/components/ui/form';
|
import { FormControl, FormItem, FormLabel, FormMessage } from '@/components/ui/form';
|
||||||
import { formatAudioDuration } from '@/lib/utils/audio';
|
import { formatAudioDuration } from '@/lib/utils/audio';
|
||||||
@@ -35,12 +35,7 @@ export function AudioSampleSystem({
|
|||||||
<div className="space-y-4">
|
<div className="space-y-4">
|
||||||
{!isRecording && !file && (
|
{!isRecording && !file && (
|
||||||
<div className="flex flex-col items-center justify-center gap-4 p-4 border-2 border-dashed rounded-lg min-h-[180px]">
|
<div className="flex flex-col items-center justify-center gap-4 p-4 border-2 border-dashed rounded-lg min-h-[180px]">
|
||||||
<Button
|
<Button type="button" onClick={onStart} size="lg" className="flex items-center gap-2">
|
||||||
type="button"
|
|
||||||
onClick={onStart}
|
|
||||||
size="lg"
|
|
||||||
className="flex items-center gap-2"
|
|
||||||
>
|
|
||||||
<Monitor className="h-5 w-5" />
|
<Monitor className="h-5 w-5" />
|
||||||
Start Capture
|
Start Capture
|
||||||
</Button>
|
</Button>
|
||||||
@@ -81,16 +76,9 @@ export function AudioSampleSystem({
|
|||||||
<Monitor className="h-5 w-5 text-primary" />
|
<Monitor className="h-5 w-5 text-primary" />
|
||||||
<span className="font-medium">Capture complete</span>
|
<span className="font-medium">Capture complete</span>
|
||||||
</div>
|
</div>
|
||||||
<p className="text-sm text-muted-foreground text-center">
|
<p className="text-sm text-muted-foreground text-center">File: {file.name}</p>
|
||||||
File: {file.name}
|
|
||||||
</p>
|
|
||||||
<div className="flex gap-2">
|
<div className="flex gap-2">
|
||||||
<Button
|
<Button type="button" size="icon" variant="outline" onClick={onPlayPause}>
|
||||||
type="button"
|
|
||||||
size="icon"
|
|
||||||
variant="outline"
|
|
||||||
onClick={onPlayPause}
|
|
||||||
>
|
|
||||||
{isPlaying ? <Pause className="h-4 w-4" /> : <Play className="h-4 w-4" />}
|
{isPlaying ? <Pause className="h-4 w-4" /> : <Play className="h-4 w-4" />}
|
||||||
</Button>
|
</Button>
|
||||||
<Button
|
<Button
|
||||||
|
|||||||
@@ -22,15 +22,15 @@ import {
|
|||||||
import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs';
|
import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs';
|
||||||
import { Textarea } from '@/components/ui/textarea';
|
import { Textarea } from '@/components/ui/textarea';
|
||||||
import { useToast } from '@/components/ui/use-toast';
|
import { useToast } from '@/components/ui/use-toast';
|
||||||
import { useAudioRecording } from '@/lib/hooks/useAudioRecording';
|
|
||||||
import { useAudioPlayer } from '@/lib/hooks/useAudioPlayer';
|
import { useAudioPlayer } from '@/lib/hooks/useAudioPlayer';
|
||||||
|
import { useAudioRecording } from '@/lib/hooks/useAudioRecording';
|
||||||
import { useAddSample, useProfile } from '@/lib/hooks/useProfiles';
|
import { useAddSample, useProfile } from '@/lib/hooks/useProfiles';
|
||||||
import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
import { useSystemAudioCapture } from '@/lib/hooks/useSystemAudioCapture';
|
||||||
import { useTranscription } from '@/lib/hooks/useTranscription';
|
import { useTranscription } from '@/lib/hooks/useTranscription';
|
||||||
import { isTauri } from '@/lib/tauri';
|
import { isTauri } from '@/lib/tauri';
|
||||||
import { AudioSampleUpload } from './AudioSampleUpload';
|
|
||||||
import { AudioSampleRecording } from './AudioSampleRecording';
|
import { AudioSampleRecording } from './AudioSampleRecording';
|
||||||
import { AudioSampleSystem } from './AudioSampleSystem';
|
import { AudioSampleSystem } from './AudioSampleSystem';
|
||||||
|
import { AudioSampleUpload } from './AudioSampleUpload';
|
||||||
|
|
||||||
const sampleSchema = z.object({
|
const sampleSchema = z.object({
|
||||||
file: z.instanceof(File, { message: 'Please select an audio file' }),
|
file: z.instanceof(File, { message: 'Please select an audio file' }),
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
import { useEffect, useState } from 'react';
|
import { useCallback, useEffect, useState } from 'react';
|
||||||
import { check, type Update } from '@tauri-apps/plugin-updater';
|
import { check, type Update } from '@tauri-apps/plugin-updater';
|
||||||
import { relaunch } from '@tauri-apps/plugin-process';
|
import { relaunch } from '@tauri-apps/plugin-process';
|
||||||
|
|
||||||
@@ -9,6 +9,9 @@ export interface UpdateStatus {
|
|||||||
downloading: boolean;
|
downloading: boolean;
|
||||||
installing: boolean;
|
installing: boolean;
|
||||||
error?: string;
|
error?: string;
|
||||||
|
downloadProgress?: number; // 0-100 percentage
|
||||||
|
downloadedBytes?: number;
|
||||||
|
totalBytes?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
const isTauri = () => {
|
const isTauri = () => {
|
||||||
@@ -25,7 +28,7 @@ export function useAutoUpdater(checkOnMount = false) {
|
|||||||
|
|
||||||
const [update, setUpdate] = useState<Update | null>(null);
|
const [update, setUpdate] = useState<Update | null>(null);
|
||||||
|
|
||||||
const checkForUpdates = async () => {
|
const checkForUpdates = useCallback(async () => {
|
||||||
if (!isTauri()) {
|
if (!isTauri()) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -61,7 +64,7 @@ export function useAutoUpdater(checkOnMount = false) {
|
|||||||
error: error instanceof Error ? error.message : 'Failed to check for updates',
|
error: error instanceof Error ? error.message : 'Failed to check for updates',
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
};
|
}, []);
|
||||||
|
|
||||||
const downloadAndInstall = async () => {
|
const downloadAndInstall = async () => {
|
||||||
if (!update || !isTauri()) return;
|
if (!update || !isTauri()) return;
|
||||||
@@ -69,19 +72,40 @@ export function useAutoUpdater(checkOnMount = false) {
|
|||||||
try {
|
try {
|
||||||
setStatus((prev) => ({ ...prev, downloading: true, error: undefined }));
|
setStatus((prev) => ({ ...prev, downloading: true, error: undefined }));
|
||||||
|
|
||||||
|
let downloadedBytes = 0;
|
||||||
|
let totalBytes = 0;
|
||||||
|
|
||||||
await update.downloadAndInstall((event) => {
|
await update.downloadAndInstall((event) => {
|
||||||
switch (event.event) {
|
switch (event.event) {
|
||||||
case 'Started':
|
case 'Started':
|
||||||
setStatus((prev) => ({ ...prev, downloading: true }));
|
totalBytes = event.data.contentLength || 0;
|
||||||
|
downloadedBytes = 0;
|
||||||
|
setStatus((prev) => ({
|
||||||
|
...prev,
|
||||||
|
downloading: true,
|
||||||
|
totalBytes,
|
||||||
|
downloadedBytes: 0,
|
||||||
|
downloadProgress: 0
|
||||||
|
}));
|
||||||
break;
|
break;
|
||||||
case 'Progress':
|
case 'Progress': {
|
||||||
console.log(`Downloaded ${event.data.chunkLength} bytes`);
|
downloadedBytes += event.data.chunkLength;
|
||||||
|
const progress = totalBytes > 0
|
||||||
|
? Math.round((downloadedBytes / totalBytes) * 100)
|
||||||
|
: undefined;
|
||||||
|
setStatus((prev) => ({
|
||||||
|
...prev,
|
||||||
|
downloadedBytes,
|
||||||
|
downloadProgress: progress
|
||||||
|
}));
|
||||||
break;
|
break;
|
||||||
|
}
|
||||||
case 'Finished':
|
case 'Finished':
|
||||||
setStatus((prev) => ({
|
setStatus((prev) => ({
|
||||||
...prev,
|
...prev,
|
||||||
downloading: false,
|
downloading: false,
|
||||||
installing: true,
|
installing: true,
|
||||||
|
downloadProgress: 100
|
||||||
}));
|
}));
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -93,6 +117,9 @@ export function useAutoUpdater(checkOnMount = false) {
|
|||||||
...prev,
|
...prev,
|
||||||
downloading: false,
|
downloading: false,
|
||||||
installing: false,
|
installing: false,
|
||||||
|
downloadProgress: undefined,
|
||||||
|
downloadedBytes: undefined,
|
||||||
|
totalBytes: undefined,
|
||||||
error: error instanceof Error ? error.message : 'Failed to install update',
|
error: error instanceof Error ? error.message : 'Failed to install update',
|
||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
@@ -102,7 +129,7 @@ export function useAutoUpdater(checkOnMount = false) {
|
|||||||
if (checkOnMount && isTauri()) {
|
if (checkOnMount && isTauri()) {
|
||||||
checkForUpdates();
|
checkForUpdates();
|
||||||
}
|
}
|
||||||
}, [checkOnMount]);
|
}, [checkOnMount, checkForUpdates]);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
status,
|
status,
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
import { useState, useRef, useCallback, useEffect } from 'react';
|
import { useCallback, useEffect, useRef, useState } from 'react';
|
||||||
import { isTauri } from '@/lib/tauri';
|
import { isTauri } from '@/lib/tauri';
|
||||||
|
import { convertToWav } from '@/lib/utils/audio';
|
||||||
|
|
||||||
interface UseAudioRecordingOptions {
|
interface UseAudioRecordingOptions {
|
||||||
maxDurationSeconds?: number;
|
maxDurationSeconds?: number;
|
||||||
@@ -85,13 +86,26 @@ export function useAudioRecording({
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
mediaRecorder.onstop = () => {
|
mediaRecorder.onstop = async () => {
|
||||||
const blob = new Blob(chunksRef.current, { type: 'audio/webm' });
|
const webmBlob = new Blob(chunksRef.current, { type: 'audio/webm' });
|
||||||
// Pass the actual recorded duration
|
|
||||||
const recordedDuration = startTimeRef.current
|
// Convert to WAV format to avoid needing ffmpeg on backend
|
||||||
? (Date.now() - startTimeRef.current) / 1000
|
try {
|
||||||
: undefined;
|
const wavBlob = await convertToWav(webmBlob);
|
||||||
onRecordingComplete?.(blob, recordedDuration);
|
|
||||||
|
// Pass the actual recorded duration
|
||||||
|
const recordedDuration = startTimeRef.current
|
||||||
|
? (Date.now() - startTimeRef.current) / 1000
|
||||||
|
: undefined;
|
||||||
|
onRecordingComplete?.(wavBlob, recordedDuration);
|
||||||
|
} catch (err) {
|
||||||
|
console.error('Error converting audio to WAV:', err);
|
||||||
|
// Fallback to original blob if conversion fails
|
||||||
|
const recordedDuration = startTimeRef.current
|
||||||
|
? (Date.now() - startTimeRef.current) / 1000
|
||||||
|
: undefined;
|
||||||
|
onRecordingComplete?.(webmBlob, recordedDuration);
|
||||||
|
}
|
||||||
|
|
||||||
// Stop all tracks
|
// Stop all tracks
|
||||||
streamRef.current?.getTracks().forEach((track) => {
|
streamRef.current?.getTracks().forEach((track) => {
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ export function useSystemAudioCapture({
|
|||||||
const timerRef = useRef<number | null>(null);
|
const timerRef = useRef<number | null>(null);
|
||||||
const startTimeRef = useRef<number | null>(null);
|
const startTimeRef = useRef<number | null>(null);
|
||||||
const stopRecordingRef = useRef<(() => Promise<void>) | null>(null);
|
const stopRecordingRef = useRef<(() => Promise<void>) | null>(null);
|
||||||
|
const isRecordingRef = useRef(false);
|
||||||
|
|
||||||
// Check if system audio capture is supported
|
// Check if system audio capture is supported
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
@@ -62,6 +63,7 @@ export function useSystemAudioCapture({
|
|||||||
});
|
});
|
||||||
|
|
||||||
setIsRecording(true);
|
setIsRecording(true);
|
||||||
|
isRecordingRef.current = true;
|
||||||
startTimeRef.current = Date.now();
|
startTimeRef.current = Date.now();
|
||||||
|
|
||||||
// Start timer
|
// Start timer
|
||||||
@@ -93,6 +95,7 @@ export function useSystemAudioCapture({
|
|||||||
|
|
||||||
try {
|
try {
|
||||||
setIsRecording(false);
|
setIsRecording(false);
|
||||||
|
isRecordingRef.current = false;
|
||||||
|
|
||||||
if (timerRef.current !== null) {
|
if (timerRef.current !== null) {
|
||||||
clearInterval(timerRef.current);
|
clearInterval(timerRef.current);
|
||||||
@@ -130,32 +133,37 @@ export function useSystemAudioCapture({
|
|||||||
}, [stopRecording]);
|
}, [stopRecording]);
|
||||||
|
|
||||||
const cancelRecording = useCallback(async () => {
|
const cancelRecording = useCallback(async () => {
|
||||||
if (isRecording) {
|
if (isRecordingRef.current) {
|
||||||
await stopRecording();
|
await stopRecording();
|
||||||
}
|
}
|
||||||
|
|
||||||
setIsRecording(false);
|
setIsRecording(false);
|
||||||
|
isRecordingRef.current = false;
|
||||||
setDuration(0);
|
setDuration(0);
|
||||||
|
|
||||||
if (timerRef.current !== null) {
|
if (timerRef.current !== null) {
|
||||||
clearInterval(timerRef.current);
|
clearInterval(timerRef.current);
|
||||||
timerRef.current = null;
|
timerRef.current = null;
|
||||||
}
|
}
|
||||||
}, [isRecording, stopRecording]);
|
}, [stopRecording]);
|
||||||
|
|
||||||
// Cleanup on unmount
|
// Cleanup on unmount only
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
return () => {
|
return () => {
|
||||||
if (timerRef.current !== null) {
|
if (timerRef.current !== null) {
|
||||||
clearInterval(timerRef.current);
|
clearInterval(timerRef.current);
|
||||||
|
timerRef.current = null;
|
||||||
}
|
}
|
||||||
// Cancel recording on unmount if still recording
|
// Cancel recording on unmount if still recording
|
||||||
if (isRecording) {
|
if (isRecordingRef.current && isTauri()) {
|
||||||
void cancelRecording();
|
// Call stop directly without the callback to avoid stale closure
|
||||||
|
invoke('stop_system_audio_capture').catch((err) => {
|
||||||
|
console.error('Error stopping audio capture on unmount:', err);
|
||||||
|
});
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
// biome-ignore lint/correctness/useExhaustiveDependencies: cancelRecording is stable
|
// biome-ignore lint/correctness/useExhaustiveDependencies: Only run on unmount
|
||||||
}, [isRecording]);
|
}, []);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
isRecording,
|
isRecording,
|
||||||
|
|||||||
@@ -16,3 +16,104 @@ export function formatAudioDuration(seconds: number): string {
|
|||||||
const secs = Math.floor(seconds % 60);
|
const secs = Math.floor(seconds % 60);
|
||||||
return `${mins}:${secs.toString().padStart(2, '0')}`;
|
return `${mins}:${secs.toString().padStart(2, '0')}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Convert any audio blob to WAV format using Web Audio API.
|
||||||
|
* This ensures compatibility without requiring ffmpeg on the backend.
|
||||||
|
*/
|
||||||
|
export async function convertToWav(audioBlob: Blob): Promise<Blob> {
|
||||||
|
// Create audio context
|
||||||
|
const audioContext = new AudioContext();
|
||||||
|
|
||||||
|
// Read blob as array buffer
|
||||||
|
const arrayBuffer = await audioBlob.arrayBuffer();
|
||||||
|
|
||||||
|
// Decode audio data
|
||||||
|
const audioBuffer = await audioContext.decodeAudioData(arrayBuffer);
|
||||||
|
|
||||||
|
// Convert to WAV
|
||||||
|
const wavBlob = audioBufferToWav(audioBuffer);
|
||||||
|
|
||||||
|
// Close audio context to free resources
|
||||||
|
await audioContext.close();
|
||||||
|
|
||||||
|
return wavBlob;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Convert AudioBuffer to WAV blob.
|
||||||
|
*/
|
||||||
|
function audioBufferToWav(buffer: AudioBuffer): Blob {
|
||||||
|
const numberOfChannels = buffer.numberOfChannels;
|
||||||
|
const sampleRate = buffer.sampleRate;
|
||||||
|
const format = 1; // PCM
|
||||||
|
const bitDepth = 16;
|
||||||
|
|
||||||
|
const bytesPerSample = bitDepth / 8;
|
||||||
|
const blockAlign = numberOfChannels * bytesPerSample;
|
||||||
|
|
||||||
|
// Interleave channels
|
||||||
|
const interleaved = interleaveChannels(buffer);
|
||||||
|
|
||||||
|
// Create WAV file
|
||||||
|
const dataLength = interleaved.length * bytesPerSample;
|
||||||
|
const buffer2 = new ArrayBuffer(44 + dataLength);
|
||||||
|
const view = new DataView(buffer2);
|
||||||
|
|
||||||
|
// Write WAV header
|
||||||
|
writeString(view, 0, 'RIFF');
|
||||||
|
view.setUint32(4, 36 + dataLength, true);
|
||||||
|
writeString(view, 8, 'WAVE');
|
||||||
|
writeString(view, 12, 'fmt ');
|
||||||
|
view.setUint32(16, 16, true); // fmt chunk size
|
||||||
|
view.setUint16(20, format, true); // audio format (PCM)
|
||||||
|
view.setUint16(22, numberOfChannels, true);
|
||||||
|
view.setUint32(24, sampleRate, true);
|
||||||
|
view.setUint32(28, sampleRate * blockAlign, true); // byte rate
|
||||||
|
view.setUint16(32, blockAlign, true);
|
||||||
|
view.setUint16(34, bitDepth, true);
|
||||||
|
writeString(view, 36, 'data');
|
||||||
|
view.setUint32(40, dataLength, true);
|
||||||
|
|
||||||
|
// Write audio data
|
||||||
|
floatTo16BitPCM(view, 44, interleaved);
|
||||||
|
|
||||||
|
return new Blob([buffer2], { type: 'audio/wav' });
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Interleave multiple channels into a single array.
|
||||||
|
*/
|
||||||
|
function interleaveChannels(buffer: AudioBuffer): Float32Array {
|
||||||
|
const numberOfChannels = buffer.numberOfChannels;
|
||||||
|
const length = buffer.length;
|
||||||
|
const interleaved = new Float32Array(length * numberOfChannels);
|
||||||
|
|
||||||
|
for (let channel = 0; channel < numberOfChannels; channel++) {
|
||||||
|
const channelData = buffer.getChannelData(channel);
|
||||||
|
for (let i = 0; i < length; i++) {
|
||||||
|
interleaved[i * numberOfChannels + channel] = channelData[i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return interleaved;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Write string to DataView.
|
||||||
|
*/
|
||||||
|
function writeString(view: DataView, offset: number, string: string): void {
|
||||||
|
for (let i = 0; i < string.length; i++) {
|
||||||
|
view.setUint8(offset + i, string.charCodeAt(i));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Convert float32 audio data to 16-bit PCM.
|
||||||
|
*/
|
||||||
|
function floatTo16BitPCM(view: DataView, offset: number, input: Float32Array): void {
|
||||||
|
for (let i = 0; i < input.length; i++, offset += 2) {
|
||||||
|
const s = Math.max(-1, Math.min(1, input[i]));
|
||||||
|
view.setInt16(offset, s < 0 ? s * 0x8000 : s * 0x7fff, true);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+1
-1
@@ -47,7 +47,7 @@ app.add_middleware(
|
|||||||
@app.get("/")
|
@app.get("/")
|
||||||
async def root():
|
async def root():
|
||||||
"""Root endpoint."""
|
"""Root endpoint."""
|
||||||
return {"message": "voicebox API", "version": "0.1.1"}
|
return {"message": "voicebox API", "version": "0.1.2"}
|
||||||
|
|
||||||
|
|
||||||
@app.get("/health", response_model=models.HealthResponse)
|
@app.get("/health", response_model=models.HealthResponse)
|
||||||
|
|||||||
@@ -4,15 +4,16 @@ from PyInstaller.utils.hooks import collect_submodules
|
|||||||
from PyInstaller.utils.hooks import copy_metadata
|
from PyInstaller.utils.hooks import copy_metadata
|
||||||
|
|
||||||
datas = []
|
datas = []
|
||||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli']
|
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern']
|
||||||
datas += collect_data_files('qwen_tts')
|
datas += collect_data_files('qwen_tts')
|
||||||
datas += copy_metadata('qwen-tts')
|
datas += copy_metadata('qwen-tts')
|
||||||
hiddenimports += collect_submodules('qwen_tts')
|
hiddenimports += collect_submodules('qwen_tts')
|
||||||
|
hiddenimports += collect_submodules('jaraco')
|
||||||
|
|
||||||
|
|
||||||
a = Analysis(
|
a = Analysis(
|
||||||
['server.py'],
|
['server.py'],
|
||||||
pathex=['/Users/jamespine/Projects/voice/Qwen3-TTS'],
|
pathex=['C:\\Users\\ijame\\Projects\\voice\\Qwen3-TTS'],
|
||||||
binaries=[],
|
binaries=[],
|
||||||
datas=datas,
|
datas=datas,
|
||||||
hiddenimports=hiddenimports,
|
hiddenimports=hiddenimports,
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/landing",
|
"name": "@voicebox/landing",
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"description": "Landing page for voicebox.sh",
|
"description": "Landing page for voicebox.sh",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "bun --bun next dev --turbo",
|
"dev": "bun --bun next dev --turbo",
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "voicebox",
|
"name": "voicebox",
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"private": true,
|
"private": true,
|
||||||
"workspaces": [
|
"workspaces": [
|
||||||
"app",
|
"app",
|
||||||
@@ -12,7 +12,7 @@
|
|||||||
"dev": "cd tauri && bun run tauri dev",
|
"dev": "cd tauri && bun run tauri dev",
|
||||||
"dev:web": "cd web && bun run dev",
|
"dev:web": "cd web && bun run dev",
|
||||||
"dev:landing": "cd landing && bun run dev",
|
"dev:landing": "cd landing && bun run dev",
|
||||||
"dev:server": "source backend/venv/bin/activate && uvicorn backend.main:app --reload --port 8000",
|
"dev:server": "uvicorn backend.main:app --reload --port 17493",
|
||||||
"build": "cd tauri && bun run tauri build",
|
"build": "cd tauri && bun run tauri build",
|
||||||
"build:web": "cd web && bun run build",
|
"build:web": "cd web && bun run build",
|
||||||
"build:landing": "cd landing && bun run build",
|
"build:landing": "cd landing && bun run build",
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/tauri",
|
"name": "@voicebox/tauri",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Generated
+1
@@ -4490,6 +4490,7 @@ dependencies = [
|
|||||||
"coreaudio-sys",
|
"coreaudio-sys",
|
||||||
"hound",
|
"hound",
|
||||||
"objc",
|
"objc",
|
||||||
|
"scopeguard",
|
||||||
"screencapturekit",
|
"screencapturekit",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
[package]
|
[package]
|
||||||
name = "voicebox"
|
name = "voicebox"
|
||||||
version = "0.1.1"
|
version = "0.1.2"
|
||||||
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
description = "A production-quality desktop app for Qwen3-TTS voice cloning and generation"
|
||||||
authors = ["you"]
|
authors = ["you"]
|
||||||
license = ""
|
license = ""
|
||||||
@@ -22,6 +22,7 @@ serde_json = "1.0"
|
|||||||
tokio = { version = "1", features = ["full"] }
|
tokio = { version = "1", features = ["full"] }
|
||||||
hound = "3.5"
|
hound = "3.5"
|
||||||
base64 = "0.22"
|
base64 = "0.22"
|
||||||
|
scopeguard = "1.2.0"
|
||||||
|
|
||||||
[target.'cfg(target_os = "macos")'.dependencies]
|
[target.'cfg(target_os = "macos")'.dependencies]
|
||||||
screencapturekit = { version = "1", features = ["async"] }
|
screencapturekit = { version = "1", features = ["async"] }
|
||||||
@@ -31,7 +32,7 @@ core-foundation-sys = "0.8"
|
|||||||
|
|
||||||
[target.'cfg(target_os = "windows")'.dependencies]
|
[target.'cfg(target_os = "windows")'.dependencies]
|
||||||
wasapi = "0.22"
|
wasapi = "0.22"
|
||||||
windows = { version = "0.62", features = ["Win32_Foundation", "Win32_UI_WindowsAndMessaging"] }
|
windows = { version = "0.62", features = ["Win32_Foundation", "Win32_UI_WindowsAndMessaging", "Win32_System_Com"] }
|
||||||
|
|
||||||
[target.'cfg(not(any(target_os = "android", target_os = "ios")))'.dependencies]
|
[target.'cfg(not(any(target_os = "android", target_os = "ios")))'.dependencies]
|
||||||
tauri-plugin-updater = "2.0"
|
tauri-plugin-updater = "2.0"
|
||||||
|
|||||||
@@ -8,5 +8,7 @@
|
|||||||
<true/>
|
<true/>
|
||||||
<key>com.apple.security.cs.disable-library-validation</key>
|
<key>com.apple.security.cs.disable-library-validation</key>
|
||||||
<true/>
|
<true/>
|
||||||
|
<key>com.apple.security.device.audio-input</key>
|
||||||
|
<true/>
|
||||||
</dict>
|
</dict>
|
||||||
</plist>
|
</plist>
|
||||||
|
|||||||
Binary file not shown.
@@ -163,25 +163,67 @@ fn extract_audio_samples(sample_buffer: CMSampleBuffer) -> Result<Vec<f32>, Stri
|
|||||||
.audio_buffer_list()
|
.audio_buffer_list()
|
||||||
.ok_or_else(|| "Failed to get audio buffer list".to_string())?;
|
.ok_or_else(|| "Failed to get audio buffer list".to_string())?;
|
||||||
|
|
||||||
let mut samples = Vec::new();
|
let buffers: Vec<_> = audio_buffer_list.iter().collect();
|
||||||
|
let num_buffers = buffers.len();
|
||||||
|
|
||||||
// Iterate through audio buffers
|
if num_buffers == 0 {
|
||||||
for buffer in audio_buffer_list.iter() {
|
return Ok(Vec::new());
|
||||||
// Get raw bytes and interpret as f32 samples
|
}
|
||||||
|
|
||||||
|
// ScreenCaptureKit on macOS provides audio in Float32 format
|
||||||
|
// The audio can be either:
|
||||||
|
// - Interleaved (1 buffer with L,R,L,R,... samples)
|
||||||
|
// - Planar (2 buffers, one for L channel, one for R channel)
|
||||||
|
|
||||||
|
if num_buffers == 1 {
|
||||||
|
// Interleaved stereo or mono in a single buffer
|
||||||
|
let buffer = &buffers[0];
|
||||||
let data_bytes = buffer.data();
|
let data_bytes = buffer.data();
|
||||||
let num_samples = data_bytes.len() / std::mem::size_of::<f32>();
|
let num_samples = data_bytes.len() / std::mem::size_of::<f32>();
|
||||||
|
|
||||||
if num_samples > 0 {
|
if num_samples > 0 {
|
||||||
unsafe {
|
unsafe {
|
||||||
// Interpret bytes as f32 samples
|
|
||||||
let data_ptr = data_bytes.as_ptr() as *const f32;
|
let data_ptr = data_bytes.as_ptr() as *const f32;
|
||||||
let data = std::slice::from_raw_parts(data_ptr, num_samples);
|
let data = std::slice::from_raw_parts(data_ptr, num_samples);
|
||||||
samples.extend_from_slice(data);
|
return Ok(data.to_vec());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
} else {
|
||||||
|
// Planar format - separate buffer for each channel
|
||||||
|
// We need to interleave them: L0, R0, L1, R1, ...
|
||||||
|
let mut channel_data: Vec<Vec<f32>> = Vec::new();
|
||||||
|
let mut max_samples = 0;
|
||||||
|
|
||||||
|
for buffer in &buffers {
|
||||||
|
let data_bytes = buffer.data();
|
||||||
|
let num_samples = data_bytes.len() / std::mem::size_of::<f32>();
|
||||||
|
|
||||||
|
if num_samples > 0 {
|
||||||
|
unsafe {
|
||||||
|
let data_ptr = data_bytes.as_ptr() as *const f32;
|
||||||
|
let data = std::slice::from_raw_parts(data_ptr, num_samples);
|
||||||
|
channel_data.push(data.to_vec());
|
||||||
|
max_samples = max_samples.max(num_samples);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Interleave the channels
|
||||||
|
let mut interleaved = Vec::with_capacity(max_samples * num_buffers);
|
||||||
|
for i in 0..max_samples {
|
||||||
|
for channel in &channel_data {
|
||||||
|
if i < channel.len() {
|
||||||
|
interleaved.push(channel[i]);
|
||||||
|
} else {
|
||||||
|
interleaved.push(0.0); // Pad with silence if needed
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return Ok(interleaved);
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(samples)
|
Ok(Vec::new())
|
||||||
}
|
}
|
||||||
|
|
||||||
fn samples_to_wav(samples: &[f32], sample_rate: u32, channels: u16) -> Result<Vec<u8>, String> {
|
fn samples_to_wav(samples: &[f32], sample_rate: u32, channels: u16) -> Result<Vec<u8>, String> {
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ pub struct AudioCaptureState {
|
|||||||
pub sample_rate: Arc<Mutex<u32>>,
|
pub sample_rate: Arc<Mutex<u32>>,
|
||||||
pub channels: Arc<Mutex<u16>>,
|
pub channels: Arc<Mutex<u16>>,
|
||||||
pub stop_tx: Arc<Mutex<Option<tokio::sync::mpsc::Sender<()>>>>,
|
pub stop_tx: Arc<Mutex<Option<tokio::sync::mpsc::Sender<()>>>>,
|
||||||
|
pub error: Arc<Mutex<Option<String>>>,
|
||||||
#[cfg(target_os = "macos")]
|
#[cfg(target_os = "macos")]
|
||||||
pub stream: Arc<Mutex<Option<SCStream>>>,
|
pub stream: Arc<Mutex<Option<SCStream>>>,
|
||||||
}
|
}
|
||||||
@@ -29,6 +30,7 @@ impl AudioCaptureState {
|
|||||||
sample_rate: Arc::new(Mutex::new(44100)),
|
sample_rate: Arc::new(Mutex::new(44100)),
|
||||||
channels: Arc::new(Mutex::new(2)),
|
channels: Arc::new(Mutex::new(2)),
|
||||||
stop_tx: Arc::new(Mutex::new(None)),
|
stop_tx: Arc::new(Mutex::new(None)),
|
||||||
|
error: Arc::new(Mutex::new(None)),
|
||||||
#[cfg(target_os = "macos")]
|
#[cfg(target_os = "macos")]
|
||||||
stream: Arc::new(Mutex::new(None)),
|
stream: Arc::new(Mutex::new(None)),
|
||||||
}
|
}
|
||||||
@@ -36,5 +38,6 @@ impl AudioCaptureState {
|
|||||||
|
|
||||||
pub fn reset(&self) {
|
pub fn reset(&self) {
|
||||||
*self.samples.lock().unwrap() = Vec::new();
|
*self.samples.lock().unwrap() = Vec::new();
|
||||||
|
*self.error.lock().unwrap() = None;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,8 +5,8 @@ use std::io::Cursor;
|
|||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use std::sync::atomic::{AtomicBool, Ordering};
|
use std::sync::atomic::{AtomicBool, Ordering};
|
||||||
use std::thread;
|
use std::thread;
|
||||||
use std::time::Duration;
|
|
||||||
use wasapi::*;
|
use wasapi::*;
|
||||||
|
use windows::Win32::System::Com::{CoInitializeEx, CoUninitialize, COINIT_MULTITHREADED};
|
||||||
|
|
||||||
pub async fn start_capture(
|
pub async fn start_capture(
|
||||||
state: &AudioCaptureState,
|
state: &AudioCaptureState,
|
||||||
@@ -19,6 +19,7 @@ pub async fn start_capture(
|
|||||||
let sample_rate_arc = state.sample_rate.clone();
|
let sample_rate_arc = state.sample_rate.clone();
|
||||||
let channels_arc = state.channels.clone();
|
let channels_arc = state.channels.clone();
|
||||||
let stop_tx = state.stop_tx.clone();
|
let stop_tx = state.stop_tx.clone();
|
||||||
|
let error_arc = state.error.clone();
|
||||||
|
|
||||||
// Use AtomicBool for stop signal (works with non-Send types)
|
// Use AtomicBool for stop signal (works with non-Send types)
|
||||||
let stop_flag = Arc::new(AtomicBool::new(false));
|
let stop_flag = Arc::new(AtomicBool::new(false));
|
||||||
@@ -36,13 +37,29 @@ pub async fn start_capture(
|
|||||||
// Spawn capture task on a dedicated thread (WASAPI COM objects are not Send)
|
// Spawn capture task on a dedicated thread (WASAPI COM objects are not Send)
|
||||||
// All WASAPI objects must be created and used on the same thread
|
// All WASAPI objects must be created and used on the same thread
|
||||||
thread::spawn(move || {
|
thread::spawn(move || {
|
||||||
|
// Initialize COM for this thread
|
||||||
|
unsafe {
|
||||||
|
let hr = CoInitializeEx(None, COINIT_MULTITHREADED);
|
||||||
|
if hr.is_err() {
|
||||||
|
eprintln!("Failed to initialize COM: {:?}", hr);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Ensure COM is uninitialized when thread exits
|
||||||
|
let _com_guard = scopeguard::guard((), |_| unsafe {
|
||||||
|
CoUninitialize();
|
||||||
|
});
|
||||||
|
|
||||||
// Initialize WASAPI on this thread
|
// Initialize WASAPI on this thread
|
||||||
let device = match DeviceEnumerator::new()
|
let device = match DeviceEnumerator::new()
|
||||||
.and_then(|enumerator| enumerator.get_default_device(&Direction::Render))
|
.and_then(|enumerator| enumerator.get_default_device(&Direction::Render))
|
||||||
{
|
{
|
||||||
Ok(d) => d,
|
Ok(d) => d,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("Failed to get audio device: {}", e);
|
let error_msg = format!("Failed to get audio device: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -50,7 +67,9 @@ pub async fn start_capture(
|
|||||||
let mut audio_client = match device.get_iaudioclient() {
|
let mut audio_client = match device.get_iaudioclient() {
|
||||||
Ok(client) => client,
|
Ok(client) => client,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("Failed to get audio client: {}", e);
|
let error_msg = format!("Failed to get audio client: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -58,7 +77,9 @@ pub async fn start_capture(
|
|||||||
let mix_format = match audio_client.get_mixformat() {
|
let mix_format = match audio_client.get_mixformat() {
|
||||||
Ok(format) => format,
|
Ok(format) => format,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("Failed to get mix format: {}", e);
|
let error_msg = format!("Failed to get mix format: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -69,27 +90,53 @@ pub async fn start_capture(
|
|||||||
*sample_rate_arc.lock().unwrap() = mix_format.get_samplespersec();
|
*sample_rate_arc.lock().unwrap() = mix_format.get_samplespersec();
|
||||||
*channels_arc.lock().unwrap() = mix_format.get_nchannels();
|
*channels_arc.lock().unwrap() = mix_format.get_nchannels();
|
||||||
|
|
||||||
|
// Get device period
|
||||||
|
let (_def_period, min_period) = match audio_client.get_device_period() {
|
||||||
|
Ok(periods) => periods,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("Failed to get device period: {}", e);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
// Initialize audio client for loopback with StreamMode
|
// Initialize audio client for loopback with StreamMode
|
||||||
|
// For loopback mode: get Render device, initialize with Capture direction
|
||||||
|
// This triggers AUDCLNT_STREAMFLAGS_LOOPBACK in the wasapi crate
|
||||||
let stream_mode = StreamMode::EventsShared {
|
let stream_mode = StreamMode::EventsShared {
|
||||||
autoconvert: false,
|
autoconvert: true, // Enable automatic format conversion
|
||||||
buffer_duration_hns: 0, // 0 = use default buffer size
|
buffer_duration_hns: min_period, // Use minimum period
|
||||||
};
|
};
|
||||||
|
|
||||||
if let Err(e) = audio_client.initialize_client(&mix_format, &Direction::Capture, &stream_mode) {
|
if let Err(e) = audio_client.initialize_client(&mix_format, &Direction::Capture, &stream_mode) {
|
||||||
eprintln!("Failed to initialize audio client: {}", e);
|
let error_msg = format!("Failed to initialize audio client: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Set up event handle for EventsShared mode
|
||||||
|
let h_event = match audio_client.set_get_eventhandle() {
|
||||||
|
Ok(event) => event,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("Failed to set event handle: {}", e);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
let capture_client = match audio_client.get_audiocaptureclient() {
|
let capture_client = match audio_client.get_audiocaptureclient() {
|
||||||
Ok(client) => client,
|
Ok(client) => client,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("Failed to get capture client: {}", e);
|
let error_msg = format!("Failed to get capture client: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
if let Err(e) = audio_client.start_stream() {
|
if let Err(e) = audio_client.start_stream() {
|
||||||
eprintln!("Failed to start stream: {}", e);
|
let error_msg = format!("Failed to start stream: {}", e);
|
||||||
|
eprintln!("{}", error_msg);
|
||||||
|
*error_arc.lock().unwrap() = Some(error_msg);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -145,8 +192,10 @@ pub async fn start_capture(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Sleep briefly to avoid busy-waiting
|
// Wait for event signal (with timeout to allow checking stop flag)
|
||||||
thread::sleep(Duration::from_millis(10));
|
if h_event.wait_for_event(100).is_err() {
|
||||||
|
// Timeout is expected - just continue to check stop flag
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Stop the stream when done
|
// Stop the stream when done
|
||||||
@@ -176,13 +225,18 @@ pub async fn stop_capture(state: &AudioCaptureState) -> Result<String, String> {
|
|||||||
// Wait a bit for capture to stop
|
// Wait a bit for capture to stop
|
||||||
tokio::time::sleep(tokio::time::Duration::from_millis(500)).await;
|
tokio::time::sleep(tokio::time::Duration::from_millis(500)).await;
|
||||||
|
|
||||||
|
// Check if there was an error during capture
|
||||||
|
if let Some(error) = state.error.lock().unwrap().as_ref() {
|
||||||
|
return Err(error.clone());
|
||||||
|
}
|
||||||
|
|
||||||
// Get samples
|
// Get samples
|
||||||
let samples = state.samples.lock().unwrap().clone();
|
let samples = state.samples.lock().unwrap().clone();
|
||||||
let sample_rate = *state.sample_rate.lock().unwrap();
|
let sample_rate = *state.sample_rate.lock().unwrap();
|
||||||
let channels = *state.channels.lock().unwrap();
|
let channels = *state.channels.lock().unwrap();
|
||||||
|
|
||||||
if samples.is_empty() {
|
if samples.is_empty() {
|
||||||
return Err("No audio samples captured".to_string());
|
return Err("No audio samples captured. Make sure audio is playing on your system during recording.".to_string());
|
||||||
}
|
}
|
||||||
|
|
||||||
// Convert to WAV
|
// Convert to WAV
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod audio_capture;
|
||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://schema.tauri.app/config/2",
|
"$schema": "https://schema.tauri.app/config/2",
|
||||||
"productName": "Voicebox",
|
"productName": "Voicebox",
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"identifier": "sh.voicebox.app",
|
"identifier": "sh.voicebox.app",
|
||||||
"build": {
|
"build": {
|
||||||
"beforeDevCommand": "bun run dev",
|
"beforeDevCommand": "bun run dev",
|
||||||
|
|||||||
@@ -0,0 +1,59 @@
|
|||||||
|
// NOTE: This test requires system audio to be playing during execution.
|
||||||
|
// To run this test successfully:
|
||||||
|
// 1. Start playing audio (music, video, etc.)
|
||||||
|
// 2. Run: cargo test --test audio_capture_test -- --nocapture
|
||||||
|
// 3. The test will capture audio for 5 seconds and verify the output
|
||||||
|
|
||||||
|
use voicebox::audio_capture::{AudioCaptureState, start_capture, stop_capture};
|
||||||
|
use base64::Engine;
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_system_audio_capture() {
|
||||||
|
// Create AudioCaptureState
|
||||||
|
let state = AudioCaptureState::new();
|
||||||
|
|
||||||
|
println!("Starting system audio capture with 5 second max duration...");
|
||||||
|
|
||||||
|
// Start capture with 5 second max duration
|
||||||
|
let result = start_capture(&state, 5).await;
|
||||||
|
|
||||||
|
if let Err(e) = result {
|
||||||
|
panic!("Failed to start capture: {}", e);
|
||||||
|
}
|
||||||
|
|
||||||
|
println!("Capture started, waiting 5 seconds...");
|
||||||
|
|
||||||
|
// Wait 5 seconds for capture to complete
|
||||||
|
tokio::time::sleep(tokio::time::Duration::from_secs(5)).await;
|
||||||
|
|
||||||
|
println!("Stopping capture...");
|
||||||
|
|
||||||
|
// Stop capture and get the result
|
||||||
|
let audio_data = stop_capture(&state).await;
|
||||||
|
|
||||||
|
match audio_data {
|
||||||
|
Ok(base64_wav) => {
|
||||||
|
println!("Capture stopped successfully");
|
||||||
|
|
||||||
|
// Validate the returned base64 WAV data
|
||||||
|
println!("Validating base64 WAV data...");
|
||||||
|
|
||||||
|
// Decode base64 to bytes
|
||||||
|
let decoded_bytes = base64::engine::general_purpose::STANDARD
|
||||||
|
.decode(&base64_wav)
|
||||||
|
.expect("Failed to decode base64 data");
|
||||||
|
|
||||||
|
// Verify bytes array is not empty
|
||||||
|
assert!(!decoded_bytes.is_empty(), "Decoded bytes array is empty");
|
||||||
|
|
||||||
|
// Confirm data has content (length > 0)
|
||||||
|
println!("WAV data length: {} bytes", decoded_bytes.len());
|
||||||
|
assert!(decoded_bytes.len() > 0, "WAV data has no content");
|
||||||
|
|
||||||
|
println!("✓ Test passed: Audio capture produced valid WAV data");
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
panic!("Failed to stop capture or get audio data: {}", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@voicebox/web",
|
"name": "@voicebox/web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.1.1",
|
"version": "0.1.2",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite",
|
"dev": "vite",
|
||||||
|
|||||||
Reference in New Issue
Block a user