mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-10-02 16:45:15 -07:00
Update dependencies and enhance UI components for better functionality
- Added new Tauri plugins: @tauri-apps/plugin-process and @tauri-apps/plugin-updater to improve application capabilities. - Introduced UpdateStatus component to display update information in the UI. - Enhanced GenerationForm to include an optional instruct field for additional input. - Refactored various components for improved styling and responsiveness, including Sidebar, AudioPlayer, and ProfileCard. - Updated API models and schemas to accommodate new instruct parameter in generation requests and responses. - Improved documentation for autoupdater setup and usage.
This commit is contained in:
+2
-1
@@ -39,7 +39,7 @@ class ProfileSample(Base):
|
||||
class Generation(Base):
|
||||
"""Generation history database model."""
|
||||
__tablename__ = "generations"
|
||||
|
||||
|
||||
id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4()))
|
||||
profile_id = Column(String, ForeignKey("profiles.id"), nullable=False)
|
||||
text = Column(Text, nullable=False)
|
||||
@@ -47,6 +47,7 @@ class Generation(Base):
|
||||
audio_path = Column(String, nullable=False)
|
||||
duration = Column(Float, nullable=False)
|
||||
seed = Column(Integer)
|
||||
instruct = Column(Text)
|
||||
created_at = Column(DateTime, default=datetime.utcnow)
|
||||
|
||||
|
||||
|
||||
+8
-4
@@ -28,10 +28,11 @@ async def create_generation(
|
||||
duration: float,
|
||||
seed: Optional[int],
|
||||
db: Session,
|
||||
instruct: Optional[str] = None,
|
||||
) -> GenerationResponse:
|
||||
"""
|
||||
Create a new generation history entry.
|
||||
|
||||
|
||||
Args:
|
||||
profile_id: Profile ID used for generation
|
||||
text: Generated text
|
||||
@@ -40,7 +41,8 @@ async def create_generation(
|
||||
duration: Audio duration in seconds
|
||||
seed: Random seed used (if any)
|
||||
db: Database session
|
||||
|
||||
instruct: Natural language instruction used (if any)
|
||||
|
||||
Returns:
|
||||
Created generation entry
|
||||
"""
|
||||
@@ -52,13 +54,14 @@ async def create_generation(
|
||||
audio_path=audio_path,
|
||||
duration=duration,
|
||||
seed=seed,
|
||||
instruct=instruct,
|
||||
created_at=datetime.utcnow(),
|
||||
)
|
||||
|
||||
|
||||
db.add(db_generation)
|
||||
db.commit()
|
||||
db.refresh(db_generation)
|
||||
|
||||
|
||||
return GenerationResponse.model_validate(db_generation)
|
||||
|
||||
|
||||
@@ -139,6 +142,7 @@ async def list_generations(
|
||||
audio_path=generation.audio_path,
|
||||
duration=generation.duration,
|
||||
seed=generation.seed,
|
||||
instruct=generation.instruct,
|
||||
created_at=generation.created_at,
|
||||
))
|
||||
|
||||
|
||||
+7
-4
@@ -259,18 +259,19 @@ async def generate_speech(
|
||||
voice_prompt,
|
||||
data.language,
|
||||
data.seed,
|
||||
data.instruct,
|
||||
)
|
||||
|
||||
|
||||
# Calculate duration
|
||||
duration = len(audio) / sample_rate
|
||||
|
||||
|
||||
# Save audio
|
||||
generation_id = str(uuid.uuid4())
|
||||
audio_path = config.get_generations_dir() / f"{generation_id}.wav"
|
||||
|
||||
|
||||
from .utils.audio import save_audio
|
||||
save_audio(audio, str(audio_path), sample_rate)
|
||||
|
||||
|
||||
# Create history entry
|
||||
generation = await history.create_generation(
|
||||
profile_id=data.profile_id,
|
||||
@@ -280,6 +281,7 @@ async def generate_speech(
|
||||
duration=duration,
|
||||
seed=data.seed,
|
||||
db=db,
|
||||
instruct=data.instruct,
|
||||
)
|
||||
|
||||
return generation
|
||||
@@ -342,6 +344,7 @@ async def get_generation(
|
||||
audio_path=gen.audio_path,
|
||||
duration=gen.duration,
|
||||
seed=gen.seed,
|
||||
instruct=gen.instruct,
|
||||
created_at=gen.created_at,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
"""
|
||||
Database migration script to add instruct column to generations table.
|
||||
|
||||
Run this once to update existing databases:
|
||||
python -m backend.migrate_add_instruct
|
||||
"""
|
||||
|
||||
import sqlite3
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def migrate():
|
||||
"""Add instruct column to generations table if it doesn't exist."""
|
||||
# Get data directory
|
||||
data_dir = os.environ.get("VOICEBOX_DATA_DIR")
|
||||
if data_dir:
|
||||
db_path = Path(data_dir) / "voicebox.db"
|
||||
else:
|
||||
db_path = Path.cwd() / "data" / "voicebox.db"
|
||||
|
||||
if not db_path.exists():
|
||||
print(f"Database not found at {db_path}, skipping migration")
|
||||
return
|
||||
|
||||
conn = sqlite3.connect(db_path)
|
||||
cursor = conn.cursor()
|
||||
|
||||
# Check if instruct column already exists
|
||||
cursor.execute("PRAGMA table_info(generations)")
|
||||
columns = [row[1] for row in cursor.fetchall()]
|
||||
|
||||
if 'instruct' in columns:
|
||||
print("instruct column already exists, skipping migration")
|
||||
conn.close()
|
||||
return
|
||||
|
||||
# Add instruct column
|
||||
print("Adding instruct column to generations table...")
|
||||
cursor.execute("ALTER TABLE generations ADD COLUMN instruct TEXT")
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
print("Migration complete!")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
migrate()
|
||||
@@ -50,6 +50,7 @@ class GenerationRequest(BaseModel):
|
||||
language: str = Field(default="en", pattern="^(en|zh)$")
|
||||
seed: Optional[int] = Field(None, ge=0)
|
||||
model_size: Optional[str] = Field(default="1.7B", pattern="^(1\\.7B|0\\.6B)$")
|
||||
instruct: Optional[str] = Field(None, max_length=500)
|
||||
|
||||
|
||||
class GenerationResponse(BaseModel):
|
||||
@@ -61,6 +62,7 @@ class GenerationResponse(BaseModel):
|
||||
audio_path: str
|
||||
duration: float
|
||||
seed: Optional[int]
|
||||
instruct: Optional[str]
|
||||
created_at: datetime
|
||||
|
||||
class Config:
|
||||
@@ -85,6 +87,7 @@ class HistoryResponse(BaseModel):
|
||||
audio_path: str
|
||||
duration: float
|
||||
seed: Optional[int]
|
||||
instruct: Optional[str]
|
||||
created_at: datetime
|
||||
|
||||
class Config:
|
||||
|
||||
+9
-6
@@ -241,35 +241,38 @@ class TTSModel:
|
||||
voice_prompt: dict,
|
||||
language: str = "en",
|
||||
seed: Optional[int] = None,
|
||||
instruct: Optional[str] = None,
|
||||
) -> Tuple[np.ndarray, int]:
|
||||
"""
|
||||
Generate audio from text using voice prompt.
|
||||
|
||||
|
||||
Args:
|
||||
text: Text to synthesize
|
||||
voice_prompt: Voice prompt dictionary from create_voice_prompt
|
||||
language: Language code (en or zh)
|
||||
seed: Random seed for reproducibility
|
||||
|
||||
instruct: Natural language instruction for speech delivery control
|
||||
|
||||
Returns:
|
||||
Tuple of (audio_array, sample_rate)
|
||||
"""
|
||||
self.load_model()
|
||||
|
||||
|
||||
# Set seed if provided
|
||||
if seed is not None:
|
||||
torch.manual_seed(seed)
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.manual_seed(seed)
|
||||
|
||||
|
||||
# Generate audio
|
||||
wavs, sample_rate = self.model.generate_voice_clone(
|
||||
text=text,
|
||||
voice_clone_prompt=voice_prompt,
|
||||
instruct=instruct,
|
||||
)
|
||||
|
||||
|
||||
audio = wavs[0] # Get first result
|
||||
|
||||
|
||||
return audio, sample_rate
|
||||
|
||||
async def generate_from_reference(
|
||||
|
||||
Reference in New Issue
Block a user