mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-29 07:05:14 -07:00
test: Playwright E2E with real backend + fake TTS engine
e2e/ drives the web build (vite preview) against a per-worker uvicorn: each Playwright worker boots its own backend on its own port with a temp cwd, so SQLite and audio files are fully isolated and parallel- safe. The page fixture seeds the persisted voicebox-server store with the worker's URL before any script runs. VOICEBOX_FAKE_TTS=1 short-circuits get_tts_backend_for_engine to a backend that synthesizes a sine tone sized to the text — the real task queue, SSE progress, history rows, and audio serving all run, only inference is fake. backend/requirements-ci.txt is the slim dependency set validated to boot the app on a CPU-only runner. Specs: startup, settings layout, seeded profile in /voices, and generate-to-completed-audio through the full pipeline.
This commit is contained in:
@@ -12,6 +12,7 @@ and a model config registry that eliminates per-engine dispatch maps.
|
||||
# HF_HUB_OFFLINE=1 and on network failures.
|
||||
from ..utils import hf_offline_patch # noqa: F401
|
||||
|
||||
import os
|
||||
import threading
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Protocol, Optional, Tuple, List
|
||||
@@ -678,6 +679,13 @@ def get_tts_backend_for_engine(engine: str) -> TTSBackend:
|
||||
"""
|
||||
global _tts_backends
|
||||
|
||||
# Test mode: every engine resolves to the fake backend so the full
|
||||
# generation pipeline runs without model weights (see fake_backend.py).
|
||||
if os.environ.get("VOICEBOX_FAKE_TTS") == "1":
|
||||
from .fake_backend import get_fake_backend
|
||||
|
||||
return get_fake_backend()
|
||||
|
||||
# Fast path: check without lock
|
||||
if engine in _tts_backends:
|
||||
return _tts_backends[engine]
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Fake TTS backend for UI and E2E testing.
|
||||
|
||||
Activated by ``VOICEBOX_FAKE_TTS=1``. Every engine resolves to this backend,
|
||||
which synthesizes a quiet sine tone sized to the input text — so the full
|
||||
generation pipeline (task queue, SSE progress, database rows, audio serving)
|
||||
runs exactly as in production, minus model weights and GPU time.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import ClassVar, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SAMPLE_RATE = 24_000
|
||||
SECONDS_PER_CHAR = 0.02
|
||||
MIN_DURATION_S = 0.25
|
||||
TONE_HZ = 440.0
|
||||
AMPLITUDE = 0.1
|
||||
|
||||
|
||||
class FakeTTSBackend:
|
||||
"""Implements the TTSBackend protocol without any model."""
|
||||
|
||||
MODEL_CONFIGS: ClassVar[list] = []
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._loaded = False
|
||||
|
||||
async def load_model(self, model_size: str = "default") -> None:
|
||||
if self._loaded:
|
||||
return
|
||||
# Brief pause so the UI's loading_model state is observable.
|
||||
await asyncio.sleep(0.1)
|
||||
self._loaded = True
|
||||
logger.info("Fake TTS backend loaded (VOICEBOX_FAKE_TTS)")
|
||||
|
||||
async def load_model_async(self, model_size: str = "default") -> None:
|
||||
# Qwen engines are loaded through this variant (see load_engine_model).
|
||||
await self.load_model(model_size)
|
||||
|
||||
async def create_voice_prompt(
|
||||
self,
|
||||
audio_path: str,
|
||||
reference_text: str,
|
||||
use_cache: bool = True,
|
||||
) -> tuple[dict, bool]:
|
||||
return ({"fake": True, "audio_path": audio_path, "reference_text": reference_text}, False)
|
||||
|
||||
async def combine_voice_prompts(
|
||||
self,
|
||||
audio_paths: list[str],
|
||||
reference_texts: list[str],
|
||||
) -> tuple[np.ndarray, str]:
|
||||
combined_text = " ".join(reference_texts)
|
||||
return np.zeros(SAMPLE_RATE, dtype=np.float32), combined_text
|
||||
|
||||
async def generate(
|
||||
self,
|
||||
text: str,
|
||||
voice_prompt: dict,
|
||||
language: str = "en",
|
||||
seed: Optional[int] = None,
|
||||
instruct: Optional[str] = None,
|
||||
) -> tuple[np.ndarray, int]:
|
||||
duration_s = max(MIN_DURATION_S, len(text) * SECONDS_PER_CHAR)
|
||||
# Yield once so cancellation has a window, mirroring real inference.
|
||||
await asyncio.sleep(0.05)
|
||||
t = np.linspace(0.0, duration_s, int(SAMPLE_RATE * duration_s), endpoint=False)
|
||||
audio = (AMPLITUDE * np.sin(2.0 * np.pi * TONE_HZ * t)).astype(np.float32)
|
||||
return audio, SAMPLE_RATE
|
||||
|
||||
def unload_model(self) -> None:
|
||||
self._loaded = False
|
||||
|
||||
def is_loaded(self) -> bool:
|
||||
return self._loaded
|
||||
|
||||
def _get_model_path(self, model_size: str) -> str:
|
||||
return "fake"
|
||||
|
||||
|
||||
_fake_backend: Optional[FakeTTSBackend] = None
|
||||
|
||||
|
||||
def get_fake_backend() -> FakeTTSBackend:
|
||||
global _fake_backend
|
||||
if _fake_backend is None:
|
||||
_fake_backend = FakeTTSBackend()
|
||||
return _fake_backend
|
||||
@@ -0,0 +1,25 @@
|
||||
# Minimal dependency set to boot the backend on a CPU-only CI runner.
|
||||
# No TTS/STT model libraries — inference is covered by the fake TTS
|
||||
# backend (VOICEBOX_FAKE_TTS=1). Install CPU torch first on Linux:
|
||||
# pip install torch --index-url https://download.pytorch.org/whl/cpu
|
||||
# then: pip install -r backend/requirements-ci.txt
|
||||
|
||||
fastapi>=0.109.0
|
||||
uvicorn[standard]>=0.27.0
|
||||
pydantic>=2.5.0
|
||||
sqlalchemy>=2.0.0
|
||||
alembic>=1.13.0
|
||||
torch>=2.2.0
|
||||
huggingface_hub>=0.20.0
|
||||
numpy
|
||||
soundfile
|
||||
python-multipart
|
||||
sse-starlette
|
||||
psutil
|
||||
requests
|
||||
httpx
|
||||
fastmcp
|
||||
librosa
|
||||
pillow
|
||||
pydub
|
||||
pedalboard
|
||||
Reference in New Issue
Block a user