Files
voicebox/backend/routes/speak.py
T
c339d2c324 fix(speak): honour the voice profile's language instead of forcing English
Both speak surfaces built their GenerationRequest with a hardcoded "en"
fallback and never consulted the resolved profile, so a profile created
with language="fr" was still synthesised as English unless the caller
passed language= explicitly.

This hurts the MCP path most: an agent calling voicebox.speak has no way
to know the bound profile's language, so it cannot pass the argument
either. Every agent-triggered generation on a non-English profile came
out with an English accent.

The fallback chain is now explicit argument -> resolved profile's
language -> "en", which matches how engine and personality already
consult the resolved binding. The "en" backstop is kept so profiles with
no language set behave exactly as before.

Adds backend/tests/test_speak_language.py covering both surfaces: the
fallback, explicit-argument precedence, and the unchanged "en" default.

Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
2026-10-04 00:00:22 +00:00

95 lines
2.8 KiB
Python

"""POST /speak — REST wrapper around voicebox.speak for non-MCP callers.
Shell scripts, ACP, A2A, or any agent that doesn't speak MCP can hit this
endpoint to play text through a cloned voice. Uses the same profile
resolution and generation pipeline as the MCP tool, so per-client
bindings (via X-Voicebox-Client-Id) work identically.
"""
from __future__ import annotations
import logging
from fastapi import APIRouter, Depends, HTTPException, Request
from sqlalchemy.orm import Session
from .. import models
from ..database import MCPClientBinding, get_db
from ..mcp_server import events as mcp_events
from ..mcp_server.resolve import resolve_profile
logger = logging.getLogger(__name__)
router = APIRouter()
@router.post("/speak", response_model=models.GenerationResponse)
async def speak(
data: models.SpeakRequest,
request: Request,
db: Session = Depends(get_db),
):
"""Speak text in a voice profile. Mirrors voicebox.speak (MCP).
Response shape matches POST /generate — a ``GenerationResponse`` with
``status="generating"`` and an ``id`` the caller polls at
``GET /generate/{id}/status``.
"""
client_id = request.headers.get("X-Voicebox-Client-Id")
profile = resolve_profile(data.profile, client_id, db)
if profile is None:
if data.profile:
raise HTTPException(
status_code=404,
detail=f"Voice profile '{data.profile}' not found.",
)
raise HTTPException(
status_code=400,
detail=(
"No voice profile resolved. Pass `profile` (name or id), "
"or configure a default in Voicebox → Settings → MCP."
),
)
binding = None
if client_id:
binding = (
db.query(MCPClientBinding)
.filter(MCPClientBinding.client_id == client_id)
.first()
)
# Resolve per-client personality default when the caller didn't pin it.
personality_flag = data.personality
if personality_flag is None and binding is not None:
personality_flag = bool(binding.default_personality)
engine = data.engine
if engine is None and binding is not None:
engine = binding.default_engine
from .generations import generate_speech
generation = await generate_speech(
models.GenerationRequest(
profile_id=profile.id,
text=data.text,
language=data.language or profile.language or "en",
engine=engine,
personality=bool(personality_flag),
),
db,
)
mcp_events.publish(
"speak-start",
{
"generation_id": getattr(generation, "id", None),
"profile_name": profile.name,
"source": "rest",
"client_id": client_id,
},
)
return generation