From ab9a19790cf9ab22373588c34095f40d45b75dde Mon Sep 17 00:00:00 2001 From: harry Date: Fri, 18 Sep 2026 16:19:19 +0530 Subject: [PATCH] fix(docker): fix startup crash and permission errors in the container Four related fixes uncovered while getting Docker running on a Linux host without AVX-512 and with an NVIDIA GPU: - pedalboard>=0.9.21 ships a Linux wheel with AVX-512 instructions baked into its native extension, which SIGILLs (exit 132) on import on any CPU without AVX-512 support. Pin below the regression until upstream fixes it (spotify/pedalboard#454). - The rocminfo probe in app.py ran unconditionally on every build variant, always failing and logging on non-ROCm systems. Gate it on /dev/kfd actually being present. - sox (required by qwen-tts's X-vector extractor) and a C compiler (required by Triton to JIT-compile its CUDA driver shim on first use) were missing from the runtime image, causing hard failures once those code paths were actually exercised. - Named volumes and bind mounts (HF cache, app data, generated audio) are created root-owned by Docker on first use, but the app runs as the unprivileged voicebox user. chown the mount points in the entrypoint before dropping privileges so this self-heals on every start regardless of host UID. --- Dockerfile | 12 ++++++++++-- backend/app.py | 2 +- backend/requirements.txt | 6 +++++- 3 files changed, 16 insertions(+), 4 deletions(-) diff --git a/Dockerfile b/Dockerfile index 003ff4b1..adf8c7ea 100644 --- a/Dockerfile +++ b/Dockerfile @@ -84,12 +84,20 @@ RUN mkdir -p /app/data/generations \ WORKDIR /app -# Install only runtime system dependencies (gosu drops root in the entrypoint) +# Install only runtime system dependencies (gosu drops root in the entrypoint). +# sox is required by qwen-tts's speech_vq X-vector extractor, which shells +# out to it via the `sox` Python bindings for reference-audio normalization. +# gcc/g++ are required by Triton (used by some torch ops, e.g. +# bmm_outer_product) to JIT-compile its CUDA driver shim on first use -- +# without them, PyTorch's CUDA-capable build crashes with "Failed to find +# C compiler" even when a GPU is actually present. RUN apt-get update && apt-get install -y --no-install-recommends \ ffmpeg \ + sox \ + gcc \ + g++ \ curl \ gosu \ - sox \ && rm -rf /var/lib/apt/lists/* # Copy installed Python packages from builder stage diff --git a/backend/app.py b/backend/app.py index c7a4029b..3d892e3c 100644 --- a/backend/app.py +++ b/backend/app.py @@ -49,7 +49,7 @@ if not os.environ.get("HSA_OVERRIDE_GFX_VERSION"): # Only set HSA_OVERRIDE_GFX_VERSION for older GPUs that need it. # RDNA 3+ (gfx1100+) and RDNA 4 (gfx1200+) are natively supported by ROCm # and the override can cause suboptimal performance or errors. -if not os.environ.get("HSA_OVERRIDE_GFX_VERSION"): +if not os.environ.get("HSA_OVERRIDE_GFX_VERSION") and Path("/dev/kfd").exists(): try: result = subprocess.run( ["rocminfo"], diff --git a/backend/requirements.txt b/backend/requirements.txt index ddc9bb6c..92621822 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -58,7 +58,11 @@ librosa>=0.10.0 soundfile>=0.12.0 numpy>=1.24.0,<2.0 numba>=0.60.0,<0.61.0 -pedalboard>=0.9.0 +# pedalboard>=0.9.21 ships a Linux wheel with AVX-512 (zmm) instructions baked +# into its native extension, which SIGILLs on import on any CPU without +# AVX-512 (e.g. Zen 2/3, pre-Ice-Lake Intel). Pin below the regression until +# upstream fixes it: https://github.com/spotify/pedalboard/issues/454 +pedalboard>=0.9.0,<0.9.21 # HTTP client (for CUDA backend download) httpx>=0.27.0