mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-19 14:50:38 -07:00
feat: chunked TTS generation for long text (engine-agnostic)
Text exceeding max_chunk_chars (default 800) is automatically split at sentence boundaries, generated per-chunk, and concatenated with a 50ms crossfade. Works with all engines (Qwen, LuxTTS, Chatterbox, Turbo). - Abbreviation-aware sentence splitter (Dr., Mr., e.g., decimals) - CJK sentence-ending punctuation support - Paralinguistic tag preservation ([laugh], [cough], etc.) - Per-chunk seed variation to avoid correlated RNG artefacts - Per-chunk Chatterbox trim (catches hallucination at each boundary) - max_chunk_chars exposed as per-request param on GenerationRequest - Text max_length raised to 50,000 characters Closes #99
This commit is contained in:
+2
-1
@@ -52,12 +52,13 @@ class ProfileSampleResponse(BaseModel):
|
||||
class GenerationRequest(BaseModel):
|
||||
"""Request model for voice generation."""
|
||||
profile_id: str
|
||||
text: str = Field(..., min_length=1, max_length=5000)
|
||||
text: str = Field(..., min_length=1, max_length=50000)
|
||||
language: str = Field(default="en", pattern="^(zh|en|ja|ko|de|fr|ru|pt|es|it|he)$")
|
||||
seed: Optional[int] = Field(None, ge=0)
|
||||
model_size: Optional[str] = Field(default="1.7B", pattern="^(1\\.7B|0\\.6B)$")
|
||||
instruct: Optional[str] = Field(None, max_length=500)
|
||||
engine: Optional[str] = Field(default="qwen", pattern="^(qwen|luxtts|chatterbox|chatterbox_turbo)$")
|
||||
max_chunk_chars: int = Field(default=800, ge=100, le=5000, description="Max characters per chunk for long text splitting")
|
||||
|
||||
|
||||
class GenerationResponse(BaseModel):
|
||||
|
||||
Reference in New Issue
Block a user