From 83906c6c4b80fcc0dcbb4d64ed8ae1dcdd90dc06 Mon Sep 17 00:00:00 2001 From: Jamie Pine Date: Sat, 31 Jan 2026 00:47:18 -0800 Subject: [PATCH] Fix R2 bucket name: voicebox (not voicebox-releases) --- .github/workflows/release.yml | 2 +- backend/test_r2_upload.sh | 4 +- backend/voicebox-server-cuda.spec | 48 +++++++++++++++++++ backend/voicebox-server.spec | 6 +-- docs/r2-setup.md | 18 ++++---- tauri/src-tauri/Cargo.lock | 2 +- test_cuda.py | 77 +++++++++++++++++++++++++++++++ 7 files changed, 139 insertions(+), 18 deletions(-) create mode 100644 backend/voicebox-server-cuda.spec create mode 100644 test_cuda.py diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index ac839ebc..a302755e 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -205,7 +205,7 @@ jobs: # Upload to R2 aws s3 cp backend/cuda-release/voicebox-server-cuda-${PLATFORM}.exe \ - s3://voicebox-releases/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe \ + s3://voicebox/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe \ --endpoint-url $R2_ENDPOINT \ --acl public-read diff --git a/backend/test_r2_upload.sh b/backend/test_r2_upload.sh index d166e148..9c638253 100644 --- a/backend/test_r2_upload.sh +++ b/backend/test_r2_upload.sh @@ -48,7 +48,7 @@ echo "Test upload configuration:" echo " Version: $VERSION" echo " Platform: $PLATFORM" echo " Endpoint: $R2_ENDPOINT" -echo " Bucket: voicebox-releases" +echo " Bucket: voicebox" echo " Path: cuda/$VERSION/$FILENAME" echo "" @@ -63,7 +63,7 @@ echo "" echo "Uploading to R2..." aws s3 cp "$CUDA_BINARY" \ - "s3://voicebox-releases/cuda/${VERSION}/${FILENAME}" \ + "s3://voicebox/cuda/${VERSION}/${FILENAME}" \ --endpoint-url "$R2_ENDPOINT" \ --acl public-read diff --git a/backend/voicebox-server-cuda.spec b/backend/voicebox-server-cuda.spec new file mode 100644 index 00000000..54ce30e5 --- /dev/null +++ b/backend/voicebox-server-cuda.spec @@ -0,0 +1,48 @@ +# -*- mode: python ; coding: utf-8 -*- +from PyInstaller.utils.hooks import collect_data_files +from PyInstaller.utils.hooks import collect_submodules +from PyInstaller.utils.hooks import copy_metadata + +datas = [] +hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern'] +datas += collect_data_files('qwen_tts') +datas += copy_metadata('qwen-tts') +hiddenimports += collect_submodules('qwen_tts') +hiddenimports += collect_submodules('jaraco') + + +a = Analysis( + ['server.py'], + pathex=[], + binaries=[], + datas=datas, + hiddenimports=hiddenimports, + hookspath=[], + hooksconfig={}, + runtime_hooks=[], + excludes=[], + noarchive=False, + optimize=0, +) +pyz = PYZ(a.pure) + +exe = EXE( + pyz, + a.scripts, + a.binaries, + a.datas, + [], + name='voicebox-server-cuda', + debug=False, + bootloader_ignore_signals=False, + strip=False, + upx=True, + upx_exclude=[], + runtime_tmpdir=None, + console=True, + disable_windowed_traceback=False, + argv_emulation=False, + target_arch=None, + codesign_identity=None, + entitlements_file=None, +) diff --git a/backend/voicebox-server.spec b/backend/voicebox-server.spec index 5d6bb317..a81af989 100644 --- a/backend/voicebox-server.spec +++ b/backend/voicebox-server.spec @@ -4,15 +4,11 @@ from PyInstaller.utils.hooks import collect_submodules from PyInstaller.utils.hooks import copy_metadata datas = [] -hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt'] +hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern'] datas += collect_data_files('qwen_tts') -datas += collect_data_files('mlx') -datas += collect_data_files('mlx_audio') datas += copy_metadata('qwen-tts') hiddenimports += collect_submodules('qwen_tts') hiddenimports += collect_submodules('jaraco') -hiddenimports += collect_submodules('mlx') -hiddenimports += collect_submodules('mlx_audio') a = Analysis( diff --git a/docs/r2-setup.md b/docs/r2-setup.md index 5d8c0698..3558a92e 100644 --- a/docs/r2-setup.md +++ b/docs/r2-setup.md @@ -7,7 +7,7 @@ The CUDA binary (2.4GB) is hosted on Cloudflare R2 at `downloads.voicebox.sh` in ## R2 Bucket Configuration ✅ **Completed:** -- Bucket created: `voicebox-releases` +- Bucket created: `voicebox` - Custom domain configured: `downloads.voicebox.sh` ## GitHub Secrets Required @@ -71,7 +71,7 @@ Value: https://.r2.cloudflarestorage.com After CI uploads, the bucket will have this structure: ``` -voicebox-releases/ +voicebox/ └── cuda/ ├── v0.1.12/ │ └── voicebox-server-cuda-x86_64-pc-windows-msvc.exe @@ -112,7 +112,7 @@ pip install awscli # Test upload (use a small test file first) echo "test" > test.txt aws s3 cp test.txt \ - s3://voicebox-releases/test/test.txt \ + s3://voicebox/test/test.txt \ --endpoint-url $R2_ENDPOINT \ --acl public-read @@ -121,7 +121,7 @@ curl https://downloads.voicebox.sh/test/test.txt # If successful, try the actual CUDA binary aws s3 cp backend/dist/voicebox-server-cuda.exe \ - s3://voicebox-releases/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe \ + s3://voicebox/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe \ --endpoint-url $R2_ENDPOINT \ --acl public-read ``` @@ -163,7 +163,7 @@ The workflow now: R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }} run: | aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \ - s3://voicebox-releases/cuda/${VERSION}/... \ + s3://voicebox/cuda/${VERSION}/... \ --endpoint-url $R2_ENDPOINT \ --acl public-read ``` @@ -172,7 +172,7 @@ The workflow now: Monitor your R2 usage: -**Cloudflare Dashboard → R2 → voicebox-releases → Metrics** +**Cloudflare Dashboard → R2 → voicebox → Metrics** Expected costs (per month): - Storage: 2.4GB × $0.015/GB = **$0.036** @@ -219,7 +219,7 @@ Or set bucket default permissions in R2 Dashboard. ✅ **Recommended:** - Object Read & Write only - No admin permissions needed -- Scoped to `voicebox-releases` bucket only +- Scoped to `voicebox` bucket only ❌ **Avoid:** - Account-wide permissions @@ -242,11 +242,11 @@ Optional: Delete old CUDA binaries to save storage costs ```bash # List all versions -aws s3 ls s3://voicebox-releases/cuda/ \ +aws s3 ls s3://voicebox/cuda/ \ --endpoint-url $R2_ENDPOINT # Delete old version -aws s3 rm s3://voicebox-releases/cuda/v0.1.0/ \ +aws s3 rm s3://voicebox/cuda/v0.1.0/ \ --recursive \ --endpoint-url $R2_ENDPOINT ``` diff --git a/tauri/src-tauri/Cargo.lock b/tauri/src-tauri/Cargo.lock index 4528097c..35b15188 100644 --- a/tauri/src-tauri/Cargo.lock +++ b/tauri/src-tauri/Cargo.lock @@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" [[package]] name = "voicebox" -version = "0.1.11" +version = "0.1.12" dependencies = [ "base64 0.22.1", "core-foundation-sys", diff --git a/test_cuda.py b/test_cuda.py new file mode 100644 index 00000000..3968a582 --- /dev/null +++ b/test_cuda.py @@ -0,0 +1,77 @@ +"""Test CUDA detection in voicebox backend""" +import sys +import torch + +print("=" * 60) +print("PyTorch CUDA Detection Test") +print("=" * 60) + +# Basic torch info +print(f"\nPyTorch version: {torch.__version__}") +print(f"CUDA available: {torch.cuda.is_available()}") + +if torch.cuda.is_available(): + print(f"CUDA version: {torch.version.cuda}") + print(f"GPU count: {torch.cuda.device_count()}") + print(f"Current GPU: {torch.cuda.current_device()}") + print(f"GPU name: {torch.cuda.get_device_name(0)}") + print(f"GPU memory: {torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB") +else: + print("\nNo CUDA available - would run on CPU") + +# Test backend device selection +print("\n" + "=" * 60) +print("Backend Device Selection") +print("=" * 60) + +# Simulate the _get_device method from pytorch_backend.py +def _get_device() -> str: + """Get the best available device.""" + if torch.cuda.is_available(): + return "cuda" + elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available(): + # MPS can have issues, use CPU for stability + return "cpu" + return "cpu" + +selected_device = _get_device() +print(f"\nSelected device: {selected_device}") +print(f"Would use dtype: {'torch.bfloat16' if selected_device != 'cpu' else 'torch.float32'}") + +# Test actual tensor creation on device +print("\n" + "=" * 60) +print("Testing Tensor Creation on Device") +print("=" * 60) + +try: + test_tensor = torch.randn(1000, 1000).to(selected_device) + print(f"\n[OK] Successfully created tensor on {selected_device}") + print(f" Tensor device: {test_tensor.device}") + print(f" Tensor dtype: {test_tensor.dtype}") + + # Test computation + result = test_tensor @ test_tensor.T + print(f"[OK] Successfully performed computation on {selected_device}") + + if selected_device == "cuda": + print(f"\nCUDA memory allocated: {torch.cuda.memory_allocated() / 1024**2:.2f} MB") + print(f"CUDA memory reserved: {torch.cuda.memory_reserved() / 1024**2:.2f} MB") + +except Exception as e: + print(f"\n[ERROR] {e}") + +print("\n" + "=" * 60) +print("Summary") +print("=" * 60) + +if selected_device == "cuda": + print("\n[SUCCESS] CUDA IS WORKING!") + print(" The backend will use your NVIDIA GPU for inference") + print(f" GPU: {torch.cuda.get_device_name(0)}") + print(f" This will be significantly faster than CPU") +else: + print("\n[FAIL] CUDA is not available") + print(" The backend will use CPU for inference") + print(" This will be slower than GPU") + +print("\n" + "=" * 60)