Fix R2 bucket name: voicebox (not voicebox-releases)

This commit is contained in:
Jamie Pine
2026-01-31 00:47:18 -08:00
parent 65f132e9c2
commit 83906c6c4b
7 changed files with 139 additions and 18 deletions
+1 -1
View File
@@ -205,7 +205,7 @@ jobs:
# Upload to R2
aws s3 cp backend/cuda-release/voicebox-server-cuda-${PLATFORM}.exe \
s3://voicebox-releases/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe \
s3://voicebox/cuda/${VERSION}/voicebox-server-cuda-${PLATFORM}.exe \
--endpoint-url $R2_ENDPOINT \
--acl public-read
+2 -2
View File
@@ -48,7 +48,7 @@ echo "Test upload configuration:"
echo " Version: $VERSION"
echo " Platform: $PLATFORM"
echo " Endpoint: $R2_ENDPOINT"
echo " Bucket: voicebox-releases"
echo " Bucket: voicebox"
echo " Path: cuda/$VERSION/$FILENAME"
echo ""
@@ -63,7 +63,7 @@ echo ""
echo "Uploading to R2..."
aws s3 cp "$CUDA_BINARY" \
"s3://voicebox-releases/cuda/${VERSION}/${FILENAME}" \
"s3://voicebox/cuda/${VERSION}/${FILENAME}" \
--endpoint-url "$R2_ENDPOINT" \
--acl public-read
+48
View File
@@ -0,0 +1,48 @@
# -*- mode: python ; coding: utf-8 -*-
from PyInstaller.utils.hooks import collect_data_files
from PyInstaller.utils.hooks import collect_submodules
from PyInstaller.utils.hooks import copy_metadata
datas = []
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern']
datas += collect_data_files('qwen_tts')
datas += copy_metadata('qwen-tts')
hiddenimports += collect_submodules('qwen_tts')
hiddenimports += collect_submodules('jaraco')
a = Analysis(
['server.py'],
pathex=[],
binaries=[],
datas=datas,
hiddenimports=hiddenimports,
hookspath=[],
hooksconfig={},
runtime_hooks=[],
excludes=[],
noarchive=False,
optimize=0,
)
pyz = PYZ(a.pure)
exe = EXE(
pyz,
a.scripts,
a.binaries,
a.datas,
[],
name='voicebox-server-cuda',
debug=False,
bootloader_ignore_signals=False,
strip=False,
upx=True,
upx_exclude=[],
runtime_tmpdir=None,
console=True,
disable_windowed_traceback=False,
argv_emulation=False,
target_arch=None,
codesign_identity=None,
entitlements_file=None,
)
+1 -5
View File
@@ -4,15 +4,11 @@ from PyInstaller.utils.hooks import collect_submodules
from PyInstaller.utils.hooks import copy_metadata
datas = []
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.profiles', 'backend.history', 'backend.tts', 'backend.transcribe', 'backend.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.utils.validation', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'librosa', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'pkg_resources.extern']
datas += collect_data_files('qwen_tts')
datas += collect_data_files('mlx')
datas += collect_data_files('mlx_audio')
datas += copy_metadata('qwen-tts')
hiddenimports += collect_submodules('qwen_tts')
hiddenimports += collect_submodules('jaraco')
hiddenimports += collect_submodules('mlx')
hiddenimports += collect_submodules('mlx_audio')
a = Analysis(
+9 -9
View File
@@ -7,7 +7,7 @@ The CUDA binary (2.4GB) is hosted on Cloudflare R2 at `downloads.voicebox.sh` in
## R2 Bucket Configuration
**Completed:**
- Bucket created: `voicebox-releases`
- Bucket created: `voicebox`
- Custom domain configured: `downloads.voicebox.sh`
## GitHub Secrets Required
@@ -71,7 +71,7 @@ Value: https://<your-account-id>.r2.cloudflarestorage.com
After CI uploads, the bucket will have this structure:
```
voicebox-releases/
voicebox/
└── cuda/
├── v0.1.12/
│ └── voicebox-server-cuda-x86_64-pc-windows-msvc.exe
@@ -112,7 +112,7 @@ pip install awscli
# Test upload (use a small test file first)
echo "test" > test.txt
aws s3 cp test.txt \
s3://voicebox-releases/test/test.txt \
s3://voicebox/test/test.txt \
--endpoint-url $R2_ENDPOINT \
--acl public-read
@@ -121,7 +121,7 @@ curl https://downloads.voicebox.sh/test/test.txt
# If successful, try the actual CUDA binary
aws s3 cp backend/dist/voicebox-server-cuda.exe \
s3://voicebox-releases/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe \
s3://voicebox/cuda/v0.1.12-test/voicebox-server-cuda-x86_64-pc-windows-msvc.exe \
--endpoint-url $R2_ENDPOINT \
--acl public-read
```
@@ -163,7 +163,7 @@ The workflow now:
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
run: |
aws s3 cp backend/cuda-release/voicebox-server-cuda-*.exe \
s3://voicebox-releases/cuda/${VERSION}/... \
s3://voicebox/cuda/${VERSION}/... \
--endpoint-url $R2_ENDPOINT \
--acl public-read
```
@@ -172,7 +172,7 @@ The workflow now:
Monitor your R2 usage:
**Cloudflare Dashboard → R2 → voicebox-releases → Metrics**
**Cloudflare Dashboard → R2 → voicebox → Metrics**
Expected costs (per month):
- Storage: 2.4GB × $0.015/GB = **$0.036**
@@ -219,7 +219,7 @@ Or set bucket default permissions in R2 Dashboard.
**Recommended:**
- Object Read & Write only
- No admin permissions needed
- Scoped to `voicebox-releases` bucket only
- Scoped to `voicebox` bucket only
**Avoid:**
- Account-wide permissions
@@ -242,11 +242,11 @@ Optional: Delete old CUDA binaries to save storage costs
```bash
# List all versions
aws s3 ls s3://voicebox-releases/cuda/ \
aws s3 ls s3://voicebox/cuda/ \
--endpoint-url $R2_ENDPOINT
# Delete old version
aws s3 rm s3://voicebox-releases/cuda/v0.1.0/ \
aws s3 rm s3://voicebox/cuda/v0.1.0/ \
--recursive \
--endpoint-url $R2_ENDPOINT
```
+1 -1
View File
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
[[package]]
name = "voicebox"
version = "0.1.11"
version = "0.1.12"
dependencies = [
"base64 0.22.1",
"core-foundation-sys",
+77
View File
@@ -0,0 +1,77 @@
"""Test CUDA detection in voicebox backend"""
import sys
import torch
print("=" * 60)
print("PyTorch CUDA Detection Test")
print("=" * 60)
# Basic torch info
print(f"\nPyTorch version: {torch.__version__}")
print(f"CUDA available: {torch.cuda.is_available()}")
if torch.cuda.is_available():
print(f"CUDA version: {torch.version.cuda}")
print(f"GPU count: {torch.cuda.device_count()}")
print(f"Current GPU: {torch.cuda.current_device()}")
print(f"GPU name: {torch.cuda.get_device_name(0)}")
print(f"GPU memory: {torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB")
else:
print("\nNo CUDA available - would run on CPU")
# Test backend device selection
print("\n" + "=" * 60)
print("Backend Device Selection")
print("=" * 60)
# Simulate the _get_device method from pytorch_backend.py
def _get_device() -> str:
"""Get the best available device."""
if torch.cuda.is_available():
return "cuda"
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
# MPS can have issues, use CPU for stability
return "cpu"
return "cpu"
selected_device = _get_device()
print(f"\nSelected device: {selected_device}")
print(f"Would use dtype: {'torch.bfloat16' if selected_device != 'cpu' else 'torch.float32'}")
# Test actual tensor creation on device
print("\n" + "=" * 60)
print("Testing Tensor Creation on Device")
print("=" * 60)
try:
test_tensor = torch.randn(1000, 1000).to(selected_device)
print(f"\n[OK] Successfully created tensor on {selected_device}")
print(f" Tensor device: {test_tensor.device}")
print(f" Tensor dtype: {test_tensor.dtype}")
# Test computation
result = test_tensor @ test_tensor.T
print(f"[OK] Successfully performed computation on {selected_device}")
if selected_device == "cuda":
print(f"\nCUDA memory allocated: {torch.cuda.memory_allocated() / 1024**2:.2f} MB")
print(f"CUDA memory reserved: {torch.cuda.memory_reserved() / 1024**2:.2f} MB")
except Exception as e:
print(f"\n[ERROR] {e}")
print("\n" + "=" * 60)
print("Summary")
print("=" * 60)
if selected_device == "cuda":
print("\n[SUCCESS] CUDA IS WORKING!")
print(" The backend will use your NVIDIA GPU for inference")
print(f" GPU: {torch.cuda.get_device_name(0)}")
print(f" This will be significantly faster than CPU")
else:
print("\n[FAIL] CUDA is not available")
print(" The backend will use CPU for inference")
print(" This will be slower than GPU")
print("\n" + "=" * 60)