feat: split CUDA backend into independently versioned server + libs archives

Switch CUDA builds from PyInstaller --onefile to --onedir and split the
output into two separately versioned archives:

1. Server core (~200-400MB) — versioned with the app, redownloaded on
   every app update
2. CUDA libs (~2GB) — versioned independently (cu126-v1), only
   redownloaded when the CUDA toolkit or torch version changes

This eliminates the ~2.4GB full redownload on every version bump.
After initial setup, most app updates only need ~200-400MB.

Closes #297
This commit is contained in:
James Pine
2026-03-17 04:04:17 -07:00
parent 2c1ee94891
commit 564d787927
6 changed files with 511 additions and 142 deletions
+15 -11
View File
@@ -200,33 +200,37 @@ jobs:
run: | run: |
python -c "import torch; print(f'CUDA available in build: {torch.cuda.is_available()}'); print(f'CUDA version: {torch.version.cuda}')" python -c "import torch; print(f'CUDA available in build: {torch.cuda.is_available()}'); print(f'CUDA version: {torch.version.cuda}')"
- name: Build CUDA server binary - name: Build CUDA server binary (onedir)
shell: bash shell: bash
working-directory: backend working-directory: backend
run: python build_binary.py --cuda run: python build_binary.py --cuda
- name: Split binary for GitHub Releases - name: Package into server core + CUDA libs archives
shell: bash shell: bash
run: | run: |
python scripts/split_binary.py \ python scripts/package_cuda.py \
backend/dist/voicebox-server-cuda.exe \ backend/dist/voicebox-server-cuda/ \
--output release-assets/ --output release-assets/ \
--cuda-libs-version cu126-v1 \
--torch-compat ">=2.6.0,<2.8.0"
- name: Upload split parts to GitHub Release - name: Upload archives to GitHub Release
if: startsWith(github.ref, 'refs/tags/') if: startsWith(github.ref, 'refs/tags/')
uses: softprops/action-gh-release@v1 uses: softprops/action-gh-release@v1
with: with:
files: | files: |
release-assets/voicebox-server-cuda.part*.exe release-assets/voicebox-server-cuda.tar.gz
release-assets/voicebox-server-cuda.sha256 release-assets/voicebox-server-cuda.tar.gz.sha256
release-assets/voicebox-server-cuda.manifest release-assets/cuda-libs-cu126-v1.tar.gz
release-assets/cuda-libs-cu126-v1.tar.gz.sha256
release-assets/cuda-libs.json
draft: true draft: true
env: env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Upload binary as workflow artifact - name: Upload onedir as workflow artifact
uses: actions/upload-artifact@v4 uses: actions/upload-artifact@v4
with: with:
name: voicebox-server-cuda-windows name: voicebox-server-cuda-windows
path: backend/dist/voicebox-server-cuda.exe path: backend/dist/voicebox-server-cuda/
retention-days: 7 retention-days: 7
+7 -1
View File
@@ -34,9 +34,15 @@ def build_server(cuda=False):
binary_name = "voicebox-server-cuda" if cuda else "voicebox-server" binary_name = "voicebox-server-cuda" if cuda else "voicebox-server"
# PyInstaller arguments # PyInstaller arguments
# CUDA builds use --onedir so we can split the output into two archives:
# 1. Server core (~200-400MB) — versioned with the app
# 2. CUDA libs (~2GB) — versioned independently (only redownloaded on
# CUDA toolkit / torch major version changes)
# CPU builds remain --onefile for simplicity.
pack_mode = "--onedir" if cuda else "--onefile"
args = [ args = [
"server.py", # Use server.py as entry point instead of main.py "server.py", # Use server.py as entry point instead of main.py
"--onefile", pack_mode,
"--name", "--name",
binary_name, binary_name,
] ]
+262 -119
View File
@@ -1,16 +1,22 @@
""" """
CUDA backend binary download, assembly, and verification. CUDA backend download, assembly, and verification.
Downloads split parts of the CUDA-enabled voicebox-server binary from Downloads two archives from GitHub Releases:
GitHub Releases, reassembles them, verifies integrity via SHA-256, 1. Server core (voicebox-server-cuda.tar.gz) — the exe + non-NVIDIA deps,
and places the binary in the app's data directory for use on next versioned with the app.
backend restart. 2. CUDA libs (cuda-libs-{version}.tar.gz) — NVIDIA runtime libraries,
versioned independently (only redownloaded on CUDA toolkit bump).
Both archives are extracted into {data_dir}/backends/cuda/ which forms the
complete PyInstaller --onedir directory structure that torch expects.
""" """
import hashlib import hashlib
import json
import logging import logging
import os import os
import sys import sys
import tarfile
from pathlib import Path from pathlib import Path
from typing import Optional from typing import Optional
@@ -24,6 +30,10 @@ GITHUB_RELEASES_URL = "https://github.com/jamiepine/voicebox/releases/download"
PROGRESS_KEY = "cuda-backend" PROGRESS_KEY = "cuda-backend"
# The current expected CUDA libs version. Bump this when we change the
# CUDA toolkit version or torch's CUDA dependency changes (e.g. cu126 -> cu128).
CUDA_LIBS_VERSION = "cu126-v1"
def get_backends_dir() -> Path: def get_backends_dir() -> Path:
"""Directory where downloaded backend binaries are stored.""" """Directory where downloaded backend binaries are stored."""
@@ -32,21 +42,46 @@ def get_backends_dir() -> Path:
return d return d
def get_cuda_binary_name() -> str: def get_cuda_dir() -> Path:
"""Platform-specific CUDA binary filename.""" """Directory where the CUDA backend (onedir) is extracted."""
d = get_backends_dir() / "cuda"
d.mkdir(parents=True, exist_ok=True)
return d
def get_cuda_exe_name() -> str:
"""Platform-specific CUDA executable filename."""
if sys.platform == "win32": if sys.platform == "win32":
return "voicebox-server-cuda.exe" return "voicebox-server-cuda.exe"
return "voicebox-server-cuda" return "voicebox-server-cuda"
def get_cuda_binary_path() -> Optional[Path]: def get_cuda_binary_path() -> Optional[Path]:
"""Return path to CUDA binary if it exists.""" """Return path to the CUDA executable if it exists inside the onedir."""
p = get_backends_dir() / get_cuda_binary_name() p = get_cuda_dir() / get_cuda_exe_name()
if p.exists(): if p.exists():
return p return p
return None return None
def get_cuda_libs_manifest_path() -> Path:
"""Path to the cuda-libs.json manifest inside the CUDA dir."""
return get_cuda_dir() / "cuda-libs.json"
def get_installed_cuda_libs_version() -> Optional[str]:
"""Read the installed CUDA libs version from cuda-libs.json, or None."""
manifest_path = get_cuda_libs_manifest_path()
if not manifest_path.exists():
return None
try:
data = json.loads(manifest_path.read_text())
return data.get("version")
except Exception as e:
logger.warning(f"Could not read cuda-libs.json: {e}")
return None
def is_cuda_active() -> bool: def is_cuda_active() -> bool:
"""Check if the current process is the CUDA binary. """Check if the current process is the CUDA binary.
@@ -60,25 +95,147 @@ def get_cuda_status() -> dict:
progress_manager = get_progress_manager() progress_manager = get_progress_manager()
cuda_path = get_cuda_binary_path() cuda_path = get_cuda_binary_path()
progress = progress_manager.get_progress(PROGRESS_KEY) progress = progress_manager.get_progress(PROGRESS_KEY)
cuda_libs_version = get_installed_cuda_libs_version()
return { return {
"available": cuda_path is not None, "available": cuda_path is not None,
"active": is_cuda_active(), "active": is_cuda_active(),
"binary_path": str(cuda_path) if cuda_path else None, "binary_path": str(cuda_path) if cuda_path else None,
"cuda_libs_version": cuda_libs_version,
"downloading": progress is not None and progress.get("status") == "downloading", "downloading": progress is not None and progress.get("status") == "downloading",
"download_progress": progress, "download_progress": progress,
} }
async def download_cuda_binary(version: Optional[str] = None): def _needs_server_download(version: Optional[str] = None) -> bool:
"""Download the CUDA backend binary from GitHub Releases. """Check if the server core archive needs to be (re)downloaded."""
cuda_path = get_cuda_binary_path()
if not cuda_path:
return True
# Check if the binary version matches the expected app version
installed = get_cuda_binary_version()
expected = version or __version__
if expected.startswith("v"):
expected = expected[1:]
return installed != expected
Downloads split parts listed in a manifest file, concatenates them,
and verifies the SHA-256 checksum for integrity. Atomic write def _needs_cuda_libs_download() -> bool:
(temp file -> rename). """Check if the CUDA libs archive needs to be (re)downloaded."""
installed = get_installed_cuda_libs_version()
if installed is None:
return True
return installed != CUDA_LIBS_VERSION
async def _download_and_extract_archive(
client,
url: str,
sha256_url: Optional[str],
dest_dir: Path,
label: str,
progress_offset: int,
total_size: int,
):
"""Download a .tar.gz archive and extract it into dest_dir.
Args: Args:
version: Version tag (e.g. "v0.2.0"). Defaults to current app version. client: httpx.AsyncClient
url: URL of the .tar.gz archive
sha256_url: URL of the .sha256 checksum file (optional)
dest_dir: Directory to extract into
label: Human-readable label for progress updates
progress_offset: Byte offset for progress reporting (when downloading
multiple archives sequentially)
total_size: Total bytes across all downloads (for progress bar)
"""
progress = get_progress_manager()
temp_path = dest_dir / f".download-{label.replace(' ', '-')}.tmp"
# Clean up leftover partial download
if temp_path.exists():
temp_path.unlink()
# Fetch expected checksum
expected_sha = None
if sha256_url:
try:
sha_resp = await client.get(sha256_url)
if sha_resp.status_code == 200:
expected_sha = sha_resp.text.strip().split()[0]
logger.info(f"{label}: expected SHA-256: {expected_sha[:16]}...")
except Exception as e:
logger.warning(f"{label}: could not fetch checksum — skipping verification: {e}")
# Stream download
downloaded = 0
async with client.stream("GET", url) as response:
response.raise_for_status()
with open(temp_path, "wb") as f:
async for chunk in response.aiter_bytes(chunk_size=1024 * 1024):
f.write(chunk)
downloaded += len(chunk)
progress.update_progress(
PROGRESS_KEY,
current=progress_offset + downloaded,
total=total_size,
filename=f"Downloading {label}",
status="downloading",
)
# Verify integrity
if expected_sha:
progress.update_progress(
PROGRESS_KEY,
current=progress_offset + downloaded,
total=total_size,
filename=f"Verifying {label}...",
status="downloading",
)
sha256 = hashlib.sha256()
with open(temp_path, "rb") as f:
while True:
data = f.read(1024 * 1024)
if not data:
break
sha256.update(data)
actual = sha256.hexdigest()
if actual != expected_sha:
temp_path.unlink()
raise ValueError(f"{label} integrity check failed: expected {expected_sha[:16]}..., got {actual[:16]}...")
logger.info(f"{label}: integrity verified")
# Extract (use data filter for path traversal protection on Python 3.12+)
progress.update_progress(
PROGRESS_KEY,
current=progress_offset + downloaded,
total=total_size,
filename=f"Extracting {label}...",
status="downloading",
)
with tarfile.open(temp_path, "r:gz") as tar:
if sys.version_info >= (3, 12):
tar.extractall(path=dest_dir, filter="data")
else:
tar.extractall(path=dest_dir)
temp_path.unlink()
logger.info(f"{label}: extracted to {dest_dir}")
return downloaded
async def download_cuda_binary(version: Optional[str] = None):
"""Download the CUDA backend (server core + CUDA libs if needed).
Downloads both archives from GitHub Releases, extracts them into
{data_dir}/backends/cuda/, and writes the cuda-libs.json manifest.
Only downloads what's needed:
- Server core: always redownloaded (versioned with app)
- CUDA libs: only if missing or version mismatch
Args:
version: Version tag (e.g. "v0.3.0"). Defaults to current app version.
""" """
import httpx import httpx
@@ -86,114 +243,91 @@ async def download_cuda_binary(version: Optional[str] = None):
version = f"v{__version__}" version = f"v{__version__}"
progress = get_progress_manager() progress = get_progress_manager()
binary_name = get_cuda_binary_name() cuda_dir = get_cuda_dir()
dest_dir = get_backends_dir()
final_path = dest_dir / binary_name
temp_path = dest_dir / f"{binary_name}.download"
# Clean up any leftover partial download need_server = _needs_server_download(version)
if temp_path.exists(): need_libs = _needs_cuda_libs_download()
temp_path.unlink()
logger.info(f"Starting CUDA backend download for {version}") if not need_server and not need_libs:
logger.info("CUDA backend is up to date, nothing to download")
return
logger.info(
f"Starting CUDA backend download for {version} "
f"(server={'yes' if need_server else 'cached'}, "
f"libs={'yes' if need_libs else 'cached'})"
)
progress.update_progress( progress.update_progress(
PROGRESS_KEY, current=0, total=0, PROGRESS_KEY,
filename="Fetching manifest...", status="downloading", current=0,
total=0,
filename="Preparing download...",
status="downloading",
) )
base_url = f"{GITHUB_RELEASES_URL}/{version}" base_url = f"{GITHUB_RELEASES_URL}/{version}"
stem = Path(binary_name).stem # voicebox-server-cuda server_archive = "voicebox-server-cuda.tar.gz"
libs_archive = f"cuda-libs-{CUDA_LIBS_VERSION}.tar.gz"
try: try:
async with httpx.AsyncClient(follow_redirects=True, timeout=30.0) as client: async with httpx.AsyncClient(follow_redirects=True, timeout=30.0) as client:
# Fetch the manifest (list of split part filenames) # Estimate total download size
manifest_url = f"{base_url}/{stem}.manifest"
manifest_resp = await client.get(manifest_url)
manifest_resp.raise_for_status()
parts = [p.strip() for p in manifest_resp.text.strip().splitlines() if p.strip()]
if not parts:
raise ValueError("Empty manifest — no split parts found")
logger.info(f"Found {len(parts)} split parts to download")
# Fetch expected checksum (optional — for integrity verification)
expected_sha = None
try:
sha_url = f"{base_url}/{stem}.sha256"
sha_resp = await client.get(sha_url)
if sha_resp.status_code == 200:
# Format: "sha256hex filename\n"
expected_sha = sha_resp.text.strip().split()[0]
logger.info(f"Expected SHA-256: {expected_sha[:16]}...")
except Exception as e:
logger.warning(f"Could not fetch checksum file — skipping verification: {e}")
# Get total size across all parts by issuing HEAD requests
total_size = 0 total_size = 0
for part_name in parts: if need_server:
try: try:
head_resp = await client.head(f"{base_url}/{part_name}") head = await client.head(f"{base_url}/{server_archive}")
content_length = int(head_resp.headers.get("content-length", 0)) total_size += int(head.headers.get("content-length", 0))
total_size += content_length
except Exception: except Exception:
pass pass
if need_libs:
try:
head = await client.head(f"{base_url}/{libs_archive}")
total_size += int(head.headers.get("content-length", 0))
except Exception:
pass
logger.info(f"Total download size: {total_size / 1024 / 1024:.1f} MB") logger.info(f"Total download size: {total_size / 1024 / 1024:.1f} MB")
# Download and concatenate parts offset = 0
total_downloaded = 0
with open(temp_path, "wb") as f:
for i, part_name in enumerate(parts):
part_url = f"{base_url}/{part_name}"
logger.info(f"Downloading part {i + 1}/{len(parts)}: {part_name}")
async with client.stream("GET", part_url) as response: # Download server core
response.raise_for_status() if need_server:
async for chunk in response.aiter_bytes(chunk_size=1024 * 1024): server_downloaded = await _download_and_extract_archive(
f.write(chunk) client,
total_downloaded += len(chunk) url=f"{base_url}/{server_archive}",
progress.update_progress( sha256_url=f"{base_url}/{server_archive}.sha256",
PROGRESS_KEY, current=total_downloaded, total=total_size, dest_dir=cuda_dir,
filename=f"Downloading CUDA backend ({i + 1}/{len(parts)})", label="CUDA server",
status="downloading", progress_offset=offset,
) total_size=total_size,
# Verify integrity if checksum was available
if expected_sha:
progress.update_progress(
PROGRESS_KEY, current=total_downloaded, total=total_downloaded,
filename="Verifying integrity...", status="downloading",
)
sha256 = hashlib.sha256()
with open(temp_path, "rb") as f:
while True:
chunk = f.read(1024 * 1024)
if not chunk:
break
sha256.update(chunk)
actual = sha256.hexdigest()
if actual != expected_sha:
raise ValueError(
f"Integrity check failed: expected {expected_sha[:16]}..., "
f"got {actual[:16]}..."
) )
logger.info(f"Integrity verified: {actual[:16]}...") offset += server_downloaded
# Atomic move into place (replace handles existing target on all platforms) # Make executable on Unix
temp_path.replace(final_path) exe_path = cuda_dir / get_cuda_exe_name()
if sys.platform != "win32" and exe_path.exists():
exe_path.chmod(0o755)
# Make executable on Unix # Download CUDA libs
if sys.platform != "win32": if need_libs:
final_path.chmod(0o755) await _download_and_extract_archive(
client,
url=f"{base_url}/{libs_archive}",
sha256_url=f"{base_url}/{libs_archive}.sha256",
dest_dir=cuda_dir,
label="CUDA libraries",
progress_offset=offset,
total_size=total_size,
)
logger.info(f"CUDA backend downloaded to {final_path}") # Write local cuda-libs.json manifest
manifest = {"version": CUDA_LIBS_VERSION}
get_cuda_libs_manifest_path().write_text(json.dumps(manifest, indent=2) + "\n")
logger.info(f"CUDA backend ready at {cuda_dir}")
progress.mark_complete(PROGRESS_KEY) progress.mark_complete(PROGRESS_KEY)
except Exception as e: except Exception as e:
# Clean up on failure
if temp_path.exists():
temp_path.unlink()
logger.error(f"CUDA backend download failed: {e}") logger.error(f"CUDA backend download failed: {e}")
progress.mark_error(PROGRESS_KEY, str(e)) progress.mark_error(PROGRESS_KEY, str(e))
raise raise
@@ -202,15 +336,19 @@ async def download_cuda_binary(version: Optional[str] = None):
def get_cuda_binary_version() -> Optional[str]: def get_cuda_binary_version() -> Optional[str]:
"""Get the version of the installed CUDA binary, or None if not installed.""" """Get the version of the installed CUDA binary, or None if not installed."""
import subprocess import subprocess
cuda_path = get_cuda_binary_path() cuda_path = get_cuda_binary_path()
if not cuda_path: if not cuda_path:
return None return None
try: try:
result = subprocess.run( result = subprocess.run(
[str(cuda_path), "--version"], [str(cuda_path), "--version"],
capture_output=True, text=True, timeout=30, capture_output=True,
text=True,
timeout=30,
cwd=str(cuda_path.parent), # Run from the onedir directory
) )
# Output format: "voicebox-server 0.2.0" # Output format: "voicebox-server 0.3.0"
for line in result.stdout.strip().splitlines(): for line in result.stdout.strip().splitlines():
if "voicebox-server" in line: if "voicebox-server" in line:
return line.split()[-1] return line.split()[-1]
@@ -222,26 +360,29 @@ def get_cuda_binary_version() -> Optional[str]:
async def check_and_update_cuda_binary(): async def check_and_update_cuda_binary():
"""Check if the CUDA binary is outdated and auto-download if so. """Check if the CUDA binary is outdated and auto-download if so.
Called on server startup. If a CUDA binary exists but its version Called on server startup. Checks both server version and CUDA libs
doesn't match the current app version, triggers a background download version. Downloads only what's needed.
of the updated CUDA binary. The download progress is visible to the
frontend via the existing SSE progress endpoint.
""" """
cuda_path = get_cuda_binary_path() cuda_path = get_cuda_binary_path()
if not cuda_path: if not cuda_path:
return # No CUDA binary installed, nothing to update return # No CUDA binary installed, nothing to update
cuda_version = get_cuda_binary_version() need_server = _needs_server_download()
current_version = __version__ need_libs = _needs_cuda_libs_download()
if cuda_version == current_version: if not need_server and not need_libs:
logger.info(f"CUDA binary is up to date (v{current_version})") logger.info(f"CUDA binary is up to date (server=v{__version__}, libs={get_installed_cuda_libs_version()})")
return return
logger.info( reasons = []
f"CUDA binary version mismatch: binary=v{cuda_version}, app=v{current_version}. " if need_server:
f"Auto-downloading updated CUDA backend..." cuda_version = get_cuda_binary_version()
) reasons.append(f"server v{cuda_version} != v{__version__}")
if need_libs:
installed_libs = get_installed_cuda_libs_version()
reasons.append(f"libs {installed_libs} != {CUDA_LIBS_VERSION}")
logger.info(f"CUDA backend needs update ({', '.join(reasons)}). Auto-downloading...")
try: try:
await download_cuda_binary() await download_cuda_binary()
@@ -250,10 +391,12 @@ async def check_and_update_cuda_binary():
async def delete_cuda_binary() -> bool: async def delete_cuda_binary() -> bool:
"""Delete the downloaded CUDA binary. Returns True if deleted.""" """Delete the downloaded CUDA backend directory. Returns True if deleted."""
path = get_cuda_binary_path() import shutil
if path and path.exists():
path.unlink() cuda_dir = get_cuda_dir()
logger.info(f"Deleted CUDA binary: {path}") if cuda_dir.exists() and any(cuda_dir.iterdir()):
shutil.rmtree(cuda_dir)
logger.info(f"Deleted CUDA backend directory: {cuda_dir}")
return True return True
return False return False
+204
View File
@@ -0,0 +1,204 @@
"""
Package the PyInstaller --onedir CUDA build into two archives.
Takes the PyInstaller --onedir output directory and splits it into:
1. voicebox-server-cuda.tar.gz — server core (exe + non-NVIDIA deps)
2. cuda-libs-cu126.tar.gz — NVIDIA runtime libraries only
3. cuda-libs.json — version manifest for the CUDA libs
Usage:
python scripts/package_cuda.py backend/dist/voicebox-server-cuda/
python scripts/package_cuda.py backend/dist/voicebox-server-cuda/ --output release-assets/
python scripts/package_cuda.py backend/dist/voicebox-server-cuda/ --cuda-libs-version cu126-v1
"""
import argparse
import hashlib
import json
import sys
import tarfile
from pathlib import Path
# Directories / prefixes that belong in the CUDA libs archive.
# PyInstaller --onedir puts NVIDIA packages in nvidia/ subdirectories
# (e.g. nvidia/cublas/lib/, nvidia/cudnn/lib/, etc.)
NVIDIA_PREFIXES = (
"nvidia/",
"nvidia\\",
)
# Individual DLL patterns that may end up at the top level on Windows
NVIDIA_DLL_PREFIXES = (
"cublas",
"cudart",
"cudnn",
"cufft",
"curand",
"cusolver",
"cusparse",
"nvjitlink",
"nvrtc",
)
def is_nvidia_file(rel_path: str) -> bool:
"""Check if a relative path belongs to the NVIDIA CUDA libs."""
rel_lower = rel_path.lower().replace("\\", "/")
# Files under nvidia/ subdirectory tree
if rel_lower.startswith("nvidia/"):
return True
# Top-level NVIDIA DLLs (Windows) — e.g. cublas64_12.dll
name = rel_lower.rsplit("/", 1)[-1]
for prefix in NVIDIA_DLL_PREFIXES:
if name.startswith(prefix) and (name.endswith(".dll") or name.endswith(".so")):
return True
return False
def sha256_file(path: Path) -> str:
"""Compute SHA-256 hex digest of a file."""
h = hashlib.sha256()
with open(path, "rb") as f:
while True:
chunk = f.read(1024 * 1024)
if not chunk:
break
h.update(chunk)
return h.hexdigest()
def package(
onedir_path: Path,
output_dir: Path,
cuda_libs_version: str,
torch_compat: str,
):
output_dir.mkdir(parents=True, exist_ok=True)
# Collect all files in the onedir output, split into core vs nvidia
core_files = []
nvidia_files = []
for item in sorted(onedir_path.rglob("*")):
if item.is_dir():
continue
rel = item.relative_to(onedir_path)
rel_str = str(rel)
if is_nvidia_file(rel_str):
nvidia_files.append((rel_str, item))
else:
core_files.append((rel_str, item))
core_size = sum(f.stat().st_size for _, f in core_files)
nvidia_size = sum(f.stat().st_size for _, f in nvidia_files)
print(f"Input directory: {onedir_path}")
print(f"Core files: {len(core_files)} ({core_size / (1024**2):.1f} MB)")
print(f"NVIDIA files: {len(nvidia_files)} ({nvidia_size / (1024**2):.1f} MB)")
if not nvidia_files:
print(
"WARNING: No NVIDIA files found! The CUDA libs archive will be empty.",
file=sys.stderr,
)
print(
"Make sure you built with --cuda and the NVIDIA packages are present.",
file=sys.stderr,
)
# Create server core archive
# Files are stored relative to the archive root (no parent directory prefix)
# so extracting to backends/cuda/ puts everything at the right level.
server_archive = output_dir / "voicebox-server-cuda.tar.gz"
print(f"\nCreating server core archive: {server_archive.name}")
with tarfile.open(server_archive, "w:gz") as tar:
for rel_str, full_path in core_files:
tar.add(full_path, arcname=rel_str)
server_sha = sha256_file(server_archive)
(output_dir / "voicebox-server-cuda.tar.gz.sha256").write_text(
f"{server_sha} voicebox-server-cuda.tar.gz\n"
)
print(f" Size: {server_archive.stat().st_size / (1024**2):.1f} MB")
print(f" SHA-256: {server_sha[:16]}...")
# Create CUDA libs archive
cuda_libs_archive = output_dir / f"cuda-libs-{cuda_libs_version}.tar.gz"
print(f"\nCreating CUDA libs archive: {cuda_libs_archive.name}")
with tarfile.open(cuda_libs_archive, "w:gz") as tar:
for rel_str, full_path in nvidia_files:
tar.add(full_path, arcname=rel_str)
cuda_sha = sha256_file(cuda_libs_archive)
(output_dir / f"cuda-libs-{cuda_libs_version}.tar.gz.sha256").write_text(
f"{cuda_sha} cuda-libs-{cuda_libs_version}.tar.gz\n"
)
print(f" Size: {cuda_libs_archive.stat().st_size / (1024**2):.1f} MB")
print(f" SHA-256: {cuda_sha[:16]}...")
# Write cuda-libs.json manifest
manifest = {
"version": cuda_libs_version,
"torch_compat": torch_compat,
"archive": cuda_libs_archive.name,
"sha256": cuda_sha,
}
manifest_path = output_dir / "cuda-libs.json"
manifest_path.write_text(json.dumps(manifest, indent=2) + "\n")
print(f"\nManifest: {manifest_path.name}")
print(json.dumps(manifest, indent=2))
# Summary
total_input = core_size + nvidia_size
total_output = server_archive.stat().st_size + cuda_libs_archive.stat().st_size
print(f"\nTotal input: {total_input / (1024**3):.2f} GB")
print(f"Total output: {total_output / (1024**3):.2f} GB (compressed)")
print(
f"Server core: {server_archive.stat().st_size / (1024**2):.1f} MB (redownloaded on app update)"
)
print(
f"CUDA libs: {cuda_libs_archive.stat().st_size / (1024**2):.1f} MB (cached until CUDA toolkit bump)"
)
def main():
parser = argparse.ArgumentParser(
description="Package PyInstaller --onedir CUDA build into server + CUDA libs archives"
)
parser.add_argument(
"input",
type=Path,
help="Path to PyInstaller --onedir output directory (e.g. backend/dist/voicebox-server-cuda/)",
)
parser.add_argument(
"--output",
type=Path,
default=None,
help="Output directory for archives (default: same as input parent)",
)
parser.add_argument(
"--cuda-libs-version",
type=str,
default="cu126-v1",
help="Version string for the CUDA libs archive (default: cu126-v1)",
)
parser.add_argument(
"--torch-compat",
type=str,
default=">=2.6.0,<2.8.0",
help="Torch version compatibility range (default: >=2.6.0,<2.8.0)",
)
args = parser.parse_args()
if not args.input.is_dir():
print(f"Error: {args.input} is not a directory", file=sys.stderr)
print("Expected a PyInstaller --onedir output directory.", file=sys.stderr)
sys.exit(1)
output_dir = args.output or args.input.parent
package(args.input, output_dir, args.cuda_libs_version, args.torch_compat)
if __name__ == "__main__":
main()
+7 -1
View File
@@ -1,6 +1,12 @@
""" """
Split a large binary into chunks for GitHub Releases (<2 GB each). Split a large binary into chunks for GitHub Releases (<2 GB each).
DEPRECATED: For CUDA builds, use scripts/package_cuda.py instead.
This script was used when the CUDA binary was built with --onefile and
needed to be split into parts for the 2GB GitHub Release asset limit.
With the switch to --onedir + dual archives (server core + CUDA libs),
package_cuda.py handles the packaging.
Usage: Usage:
python scripts/split_binary.py backend/dist/voicebox-server-cuda.exe python scripts/split_binary.py backend/dist/voicebox-server-cuda.exe
python scripts/split_binary.py backend/dist/voicebox-server-cuda.exe --chunk-size 1900000000 python scripts/split_binary.py backend/dist/voicebox-server-cuda.exe --chunk-size 1900000000
@@ -34,7 +40,7 @@ def split(input_path: Path, chunk_size: int, output_dir: Path):
part_index = len(parts) part_index = len(parts)
part_name = f"{input_path.stem}.part{part_index:02d}{input_path.suffix}" part_name = f"{input_path.stem}.part{part_index:02d}{input_path.suffix}"
part_path = output_dir / part_name part_path = output_dir / part_name
part_path.write_bytes(data[i:i + chunk_size]) part_path.write_bytes(data[i : i + chunk_size])
parts.append(part_name) parts.append(part_name)
# Write manifest (ordered list of part filenames) # Write manifest (ordered list of part filenames)
+16 -10
View File
@@ -197,22 +197,24 @@ async fn start_server(
println!("Data directory: {:?}", data_dir); println!("Data directory: {:?}", data_dir);
println!("Remote mode: {}", remote.unwrap_or(false)); println!("Remote mode: {}", remote.unwrap_or(false));
// Check for CUDA backend binary in data directory // Check for CUDA backend in data directory (onedir layout: backends/cuda/)
let cuda_binary = { let cuda_binary = {
let backends_dir = data_dir.join("backends"); let cuda_dir = data_dir.join("backends").join("cuda");
let cuda_name = if cfg!(windows) { let cuda_name = if cfg!(windows) {
"voicebox-server-cuda.exe" "voicebox-server-cuda.exe"
} else { } else {
"voicebox-server-cuda" "voicebox-server-cuda"
}; };
let path = backends_dir.join(cuda_name); let exe_path = cuda_dir.join(cuda_name);
if path.exists() { if exe_path.exists() {
println!("Found CUDA backend binary at {:?}", path); println!("Found CUDA backend at {:?}", cuda_dir);
// Version check: run --version and compare to app version // Version check: run --version from the onedir directory so
// PyInstaller can find its support files for the fast --version path
let app_version = app.config().version.clone().unwrap_or_default(); let app_version = app.config().version.clone().unwrap_or_default();
let version_ok = match std::process::Command::new(&path) let version_ok = match std::process::Command::new(&exe_path)
.arg("--version") .arg("--version")
.current_dir(&cuda_dir)
.output() .output()
{ {
Ok(output) => { Ok(output) => {
@@ -237,7 +239,7 @@ async fn start_server(
}; };
if version_ok { if version_ok {
Some(path) Some(exe_path)
} else { } else {
None None
} }
@@ -300,10 +302,14 @@ async fn start_server(
println!("Custom models directory: {}", dir); println!("Custom models directory: {}", dir);
} }
// If CUDA binary exists, launch it directly instead of the bundled sidecar // If CUDA binary exists, launch it from the onedir directory.
// .current_dir() is critical: PyInstaller onedir expects all DLLs and
// support files (nvidia/, _internal/, etc.) relative to the exe.
let spawn_result = if let Some(ref cuda_path) = cuda_binary { let spawn_result = if let Some(ref cuda_path) = cuda_binary {
println!("Launching CUDA backend: {:?}", cuda_path); let cuda_dir = cuda_path.parent().unwrap();
println!("Launching CUDA backend: {:?} (cwd: {:?})", cuda_path, cuda_dir);
let mut cmd = app.shell().command(cuda_path.to_str().unwrap()); let mut cmd = app.shell().command(cuda_path.to_str().unwrap());
cmd = cmd.current_dir(cuda_dir);
cmd = cmd.args(["--data-dir", &data_dir_str, "--port", &port_str, "--parent-pid", &parent_pid_str]); cmd = cmd.args(["--data-dir", &data_dir_str, "--port", &port_str, "--parent-pid", &parent_pid_str]);
if is_remote { if is_remote {
cmd = cmd.args(["--host", "0.0.0.0"]); cmd = cmd.args(["--host", "0.0.0.0"]);