From d4face494f00d48337305c0a998221ba0c9f9bcb Mon Sep 17 00:00:00 2001 From: Nikhil Jangid <161968345+nikhiljangid120@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:29:23 -0500 Subject: [PATCH 01/35] fix: honor profile engine for generation requests Allow omitted engines to fall through to the profile default. --- backend/models.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/models.py b/backend/models.py index 7970ce41..5b3d8217 100644 --- a/backend/models.py +++ b/backend/models.py @@ -85,7 +85,7 @@ class GenerationRequest(BaseModel): seed: Optional[int] = Field(None, ge=0) model_size: Optional[str] = Field(default="1.7B", pattern="^(1\\.7B|0\\.6B|1B|3B)$") instruct: Optional[str] = Field(None, max_length=500) - engine: Optional[str] = Field(default="qwen", pattern="^(qwen|qwen_custom_voice|luxtts|chatterbox|chatterbox_turbo|tada|kokoro)$") + engine: Optional[str] = Field(default=None, pattern="^(qwen|qwen_custom_voice|luxtts|chatterbox|chatterbox_turbo|tada|kokoro)$") personality: bool = Field( default=False, description="When true and the profile has a personality prompt, the input text is rewritten in-character before TTS.", From cdeeb38be4e6af464685f55d316f1c4be78cf38d Mon Sep 17 00:00:00 2001 From: Nikhil Jangid <161968345+nikhiljangid120@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:31:21 -0500 Subject: [PATCH 02/35] test: cover generation engine defaults Verify omitted engines remain unset while explicit values are validated. --- .../test_generation_engine_resolution.py | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) create mode 100644 backend/tests/test_generation_engine_resolution.py diff --git a/backend/tests/test_generation_engine_resolution.py b/backend/tests/test_generation_engine_resolution.py new file mode 100644 index 00000000..7cd3dc34 --- /dev/null +++ b/backend/tests/test_generation_engine_resolution.py @@ -0,0 +1,27 @@ +"""Tests for generation request engine selection.""" + +import pytest +from pydantic import ValidationError + +from backend import models + + +def _request(**kwargs) -> models.GenerationRequest: + return models.GenerationRequest(profile_id="profile-1", text="hello", **kwargs) + + +def test_omitted_engine_does_not_override_profile_default(): + request = _request() + + assert request.engine is None + + +def test_explicit_engine_is_preserved(): + request = _request(engine="chatterbox") + + assert request.engine == "chatterbox" + + +def test_invalid_explicit_engine_is_rejected(): + with pytest.raises(ValidationError): + _request(engine="invalid") From f2271ae84538a57cc20e84fc44d0f5a0bea50085 Mon Sep 17 00:00:00 2001 From: Nikhil Jangid <161968345+nikhiljangid120@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:33:25 -0500 Subject: [PATCH 03/35] docs: note profile engine default fix Document the generation engine precedence correction. --- CHANGELOG.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0b03f5e4..d1d59953 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,8 +7,14 @@ ## [Unreleased] +### Fixed + +- Voice generation requests that omit `engine` now honor the selected profile's + configured engine instead of silently defaulting to Qwen. + ### Linux + - **ROCm setup works on Linux AMD systems.** Docker ROCm builds now keep PyTorch on the ROCm wheel index during dependency installation, so later installs do not replace it with CUDA wheels. The ROCm compose overlay no longer assumes From f5d63540a6a0cba3a6046d2f4188d2b808ebd3aa Mon Sep 17 00:00:00 2001 From: Nikhil Jangid <161968345+nikhiljangid120@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:35:21 -0500 Subject: [PATCH 04/35] docs: tidy changelog spacing Keep the unreleased changelog formatting consistent. --- CHANGELOG.md | 1 - 1 file changed, 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d1d59953..9f71fda2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,6 @@ ### Linux - - **ROCm setup works on Linux AMD systems.** Docker ROCm builds now keep PyTorch on the ROCm wheel index during dependency installation, so later installs do not replace it with CUDA wheels. The ROCm compose overlay no longer assumes From 848bb2e4c64ae9b298c90c815aed70edbe6730e6 Mon Sep 17 00:00:00 2001 From: Matheus Suardi Date: Mon, 28 Sep 2026 11:34:44 +0100 Subject: [PATCH 05/35] fix(server): stop failing requests after the app's stdout pipe closes When the server outlives the Tauri app that spawned it (keep-running mode, or a sidecar the next launch reuses), its stdout/stderr pipe has no reader. Every later print()/tqdm write raises BrokenPipeError, so POST /captures and /transcribe return "[Errno 32] Broken pipe". Wrap stdout/stderr so they fall back to devnull on the first failed write instead of raising. Co-Authored-By: Claude Opus 5.5 (1M context) --- backend/server.py | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/backend/server.py b/backend/server.py index 047f4fbd..f1c8ab43 100644 --- a/backend/server.py +++ b/backend/server.py @@ -27,6 +27,39 @@ if not _is_writable(sys.stdout): if not _is_writable(sys.stderr): sys.stderr = open(os.devnull, 'w') + +class _PipeSafeStream: + """Falls back to devnull once the pipe to the Tauri app is gone. + + The app reads our stdout/stderr through a pipe. When the server outlives + it (keep-running mode, or a sidecar the next launch reuses), every later + print()/tqdm write raises "[Errno 32] Broken pipe" and fails whatever + request triggered it, e.g. POST /captures. + """ + + def __init__(self, stream): + self._stream = stream + + def write(self, s): + try: + return self._stream.write(s) + except (OSError, ValueError): + self._stream = open(os.devnull, 'w') + return len(s) + + def flush(self): + try: + self._stream.flush() + except (OSError, ValueError): + self._stream = open(os.devnull, 'w') + + def __getattr__(self, name): + return getattr(self._stream, name) + + +sys.stdout = _PipeSafeStream(sys.stdout) +sys.stderr = _PipeSafeStream(sys.stderr) + # PyInstaller + multiprocessing: child processes re-execute the frozen binary # with internal arguments. freeze_support() handles this and exits early. import multiprocessing From c9f5f2c3e75091ff3c56cac3d9d73753aff7b8b2 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:32:36 +0000 Subject: [PATCH 06/35] fix(server): route writelines through the pipe-safe write writelines() was forwarded straight to the wrapped stream by __getattr__, so it could still raise BrokenPipeError after the app's pipe closed. --- backend/server.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/backend/server.py b/backend/server.py index f1c8ab43..3ee5c5bc 100644 --- a/backend/server.py +++ b/backend/server.py @@ -47,6 +47,10 @@ class _PipeSafeStream: self._stream = open(os.devnull, 'w') return len(s) + def writelines(self, lines): + for line in lines: + self.write(line) + def flush(self): try: self._stream.flush() From 4cb8ad3b1ff8529363cc8d5b754328ce2478176a Mon Sep 17 00:00:00 2001 From: Margalit Date: Fri, 18 Sep 2026 01:58:34 +0300 Subject: [PATCH 07/35] Add 10 missing paralinguistic delivery tags to ParalinguisticInput --- .../components/Generation/ParalinguisticInput.tsx | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/app/src/components/Generation/ParalinguisticInput.tsx b/app/src/components/Generation/ParalinguisticInput.tsx index d6c44210..cfa44bff 100644 --- a/app/src/components/Generation/ParalinguisticInput.tsx +++ b/app/src/components/Generation/ParalinguisticInput.tsx @@ -23,9 +23,19 @@ const PARALINGUISTIC_TAGS = [ { tag: '[sniff]', label: 'sniff', emoji: '\u{1F443}' }, { tag: '[shush]', label: 'shush', emoji: '\u{1F92B}' }, { tag: '[clear throat]', label: 'clear throat', emoji: '\u{1F64A}' }, + { tag: '[angry]', label: 'angry', emoji: '\u{1F621}' }, + { tag: '[crying]', label: 'crying', emoji: '\u{1F622}' }, + { tag: '[dramatic]', label: 'dramatic', emoji: '\u{1F3AD}' }, + { tag: '[fear]', label: 'fear', emoji: '\u{1F628}' }, + { tag: '[happy]', label: 'happy', emoji: '\u{1F60A}' }, + { tag: '[narration]', label: 'narration', emoji: '\u{1F4D6}' }, + { tag: '[sarcastic]', label: 'sarcastic', emoji: '\u{1F60F}' }, + { tag: '[surprised]', label: 'surprised', emoji: '\u{1F632}' }, + { tag: '[whispering]', label: 'whispering', emoji: '\u{1F92B}' }, + { tag: '[advertisement]', label: 'advertisement', emoji: '\u{1F4E2}' }, ] as const; -const TAG_REGEX = /\[(laugh|chuckle|gasp|cough|sigh|groan|sniff|shush|clear throat)\]/gi; +const TAG_REGEX = /\[(laugh|chuckle|gasp|cough|sigh|groan|sniff|shush|clear throat|angry|crying|dramatic|fear|happy|narration|sarcastic|surprised|whispering|advertisement)\]/gi; // Data attribute used to identify badge spans in the DOM const BADGE_ATTR = 'data-ptag'; From 50d8d6a34ff15711232dedbc443ea6eda9117903 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:24:06 +0000 Subject: [PATCH 08/35] style: wrap TAG_REGEX to satisfy biome formatter --- app/src/components/Generation/ParalinguisticInput.tsx | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/app/src/components/Generation/ParalinguisticInput.tsx b/app/src/components/Generation/ParalinguisticInput.tsx index cfa44bff..74ad99a3 100644 --- a/app/src/components/Generation/ParalinguisticInput.tsx +++ b/app/src/components/Generation/ParalinguisticInput.tsx @@ -35,7 +35,8 @@ const PARALINGUISTIC_TAGS = [ { tag: '[advertisement]', label: 'advertisement', emoji: '\u{1F4E2}' }, ] as const; -const TAG_REGEX = /\[(laugh|chuckle|gasp|cough|sigh|groan|sniff|shush|clear throat|angry|crying|dramatic|fear|happy|narration|sarcastic|surprised|whispering|advertisement)\]/gi; +const TAG_REGEX = + /\[(laugh|chuckle|gasp|cough|sigh|groan|sniff|shush|clear throat|angry|crying|dramatic|fear|happy|narration|sarcastic|surprised|whispering|advertisement)\]/gi; // Data attribute used to identify badge spans in the DOM const BADGE_ATTR = 'data-ptag'; From 214adc5ff25a2368a7cc42912a340479c30334ae Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:31:46 +0000 Subject: [PATCH 09/35] fix: keep the active tag row in view and document the full tag set The 19-row menu is taller than its 280px max-height, so arrow-key navigation past the fold lost its highlight. Scroll the active row into view on index change, list the delivery tags in the README, and give [sarcastic] and [whispering] emoji that are not already used by [chuckle] and [shush]. --- README.md | 4 +++- .../components/Generation/ParalinguisticInput.tsx | 14 ++++++++++++-- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index b1ee81fc..bda86132 100644 --- a/README.md +++ b/README.md @@ -130,7 +130,9 @@ literally as text. With **Chatterbox Turbo** selected, type `/` in the text input to open the tag inserter and add expressive tags inline with speech: -`[laugh]` `[chuckle]` `[gasp]` `[cough]` `[sigh]` `[groan]` `[sniff]` `[shush]` `[clear throat]` +Sounds: `[laugh]` `[chuckle]` `[gasp]` `[cough]` `[sigh]` `[groan]` `[sniff]` `[shush]` `[clear throat]` + +Delivery: `[angry]` `[crying]` `[dramatic]` `[fear]` `[happy]` `[narration]` `[sarcastic]` `[surprised]` `[whispering]` `[advertisement]` ### Post-Processing Effects diff --git a/app/src/components/Generation/ParalinguisticInput.tsx b/app/src/components/Generation/ParalinguisticInput.tsx index 74ad99a3..205e2905 100644 --- a/app/src/components/Generation/ParalinguisticInput.tsx +++ b/app/src/components/Generation/ParalinguisticInput.tsx @@ -29,9 +29,9 @@ const PARALINGUISTIC_TAGS = [ { tag: '[fear]', label: 'fear', emoji: '\u{1F628}' }, { tag: '[happy]', label: 'happy', emoji: '\u{1F60A}' }, { tag: '[narration]', label: 'narration', emoji: '\u{1F4D6}' }, - { tag: '[sarcastic]', label: 'sarcastic', emoji: '\u{1F60F}' }, + { tag: '[sarcastic]', label: 'sarcastic', emoji: '\u{1F644}' }, { tag: '[surprised]', label: 'surprised', emoji: '\u{1F632}' }, - { tag: '[whispering]', label: 'whispering', emoji: '\u{1F92B}' }, + { tag: '[whispering]', label: 'whispering', emoji: '\u{1F62F}' }, { tag: '[advertisement]', label: 'advertisement', emoji: '\u{1F4E2}' }, ] as const; @@ -147,6 +147,7 @@ export const ParalinguisticInput = forwardRef(null); + const menuListRef = useRef(null); const lastSerializedRef = useRef(''); const isComposingRef = useRef(false); @@ -155,6 +156,14 @@ export const ParalinguisticInput = forwardRef { + if (!showMenu) return; + const item = menuListRef.current?.children[menuIndex] as HTMLElement | undefined; + item?.scrollIntoView({ block: 'nearest' }); + }, [showMenu, menuIndex]); + // Filtered tag list for the autocomplete menu const filteredTags = PARALINGUISTIC_TAGS.filter((t) => t.label.toLowerCase().includes(menuFilter.toLowerCase()), @@ -392,6 +401,7 @@ export const ParalinguisticInput = forwardRef Date: Sun, 6 Sep 2026 01:31:24 +0200 Subject: [PATCH 10/35] fix(backend): batch generation-version queries to eliminate N+1 in history listing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit list_generations() fetched versions with one SELECT per generation on the page (50 rows -> 51 queries). Add _get_versions_for_generations() which loads all versions for the page in a single WHERE generation_id IN (...) query and groups them in memory; the single-generation helper now delegates to it so story item details behave identically. Generated with Codebuff 🤖 Co-Authored-By: Codebuff --- backend/services/history.py | 46 +++++++++++++++++++++++++++++-------- 1 file changed, 36 insertions(+), 10 deletions(-) diff --git a/backend/services/history.py b/backend/services/history.py index 3062f7d6..c2adfe10 100644 --- a/backend/services/history.py +++ b/backend/services/history.py @@ -15,21 +15,30 @@ from ..database import Generation as DBGeneration, GenerationVersion as DBGenera from .. import config -def _get_versions_for_generation(generation_id: str, db: Session) -> tuple: - """Get versions list and active version ID for a generation.""" +def _get_versions_for_generations(generation_ids: list[str], db: Session) -> dict: + """Fetch versions for many generations in a single query. + + Returns a mapping of ``generation_id -> (versions, active_version_id)`` + using the same shape as ``_get_versions_for_generation()``, so callers + can batch a whole page of generations without an N+1 query. + """ import json + + ids = list(dict.fromkeys(generation_ids)) + if not ids: + return {} + versions_rows = ( db.query(DBGenerationVersion) - .filter_by(generation_id=generation_id) + .filter(DBGenerationVersion.generation_id.in_(ids)) .order_by(DBGenerationVersion.created_at) .all() ) - if not versions_rows: - return None, None - versions = [] - active_version_id = None + versions_by_generation: dict[str, list] = {} + active_by_generation: dict[str, Optional[str]] = {} for v in versions_rows: + versions = versions_by_generation.setdefault(v.generation_id, []) effects_chain = None if v.effects_chain: try: @@ -47,9 +56,20 @@ def _get_versions_for_generation(generation_id: str, db: Session) -> tuple: created_at=v.created_at, )) if v.is_default: - active_version_id = v.id + active_by_generation[v.generation_id] = v.id - return versions, active_version_id + return { + generation_id: ( + versions_by_generation.get(generation_id), + active_by_generation.get(generation_id), + ) + for generation_id in ids + } + + +def _get_versions_for_generation(generation_id: str, db: Session) -> tuple: + """Get versions list and active version ID for a single generation.""" + return _get_versions_for_generations([generation_id], db)[generation_id] async def create_generation( @@ -205,10 +225,16 @@ async def list_generations( # Execute query results = q.all() + # Fetch versions for every generation on this page with a single + # query instead of one SELECT per generation (N+1). + versions_by_generation = _get_versions_for_generations( + [generation.id for generation, _ in results], db + ) + # Convert to HistoryResponse with profile_name items = [] for generation, profile_name in results: - versions, active_version_id = _get_versions_for_generation(generation.id, db) + versions, active_version_id = versions_by_generation[generation.id] items.append(HistoryResponse( id=generation.id, profile_id=generation.profile_id, From 8b8c1429dbe889ca6bf7a49c3060b965fab23f3d Mon Sep 17 00:00:00 2001 From: Ousama Ben Younes Date: Sun, 23 Aug 2026 02:09:00 +0000 Subject: [PATCH 11/35] fix(backend): honour VOICEBOX_FORCE_CPU in device selection The override is documented in gpu-acceleration.mdx and listed as step 1 of the get_torch_device() precedence in tts-generation.mdx, but grepping the tree for VOICEBOX_FORCE_CPU matched only those two doc files - nothing read it. Users whose GPU has no compiled kernels in the bundled PyTorch had no way to fall back to CPU short of renaming the installed CUDA backend directory. Resolve it before torch is imported, so it still works when the installed build is itself the reason CPU is wanted. --- backend/backends/base.py | 17 ++++++ backend/tests/test_force_cpu_env.py | 80 +++++++++++++++++++++++++++++ 2 files changed, 97 insertions(+) create mode 100644 backend/tests/test_force_cpu_env.py diff --git a/backend/backends/base.py b/backend/backends/base.py index 70ec11ef..fa1c90ac 100644 --- a/backend/backends/base.py +++ b/backend/backends/base.py @@ -6,6 +6,7 @@ voice prompt combination, and model loading progress tracking. """ import logging +import os import platform from contextlib import contextmanager from pathlib import Path @@ -77,6 +78,13 @@ def is_model_cached( return False +# Documented escape hatch (docs/content/docs/overview/gpu-acceleration.mdx): +# users whose GPU has no compiled kernels in the bundled PyTorch set this to run +# on CPU instead of crashing at generation time. +FORCE_CPU_ENV_VAR = "VOICEBOX_FORCE_CPU" +FORCE_CPU_ENABLED_VALUE = "1" + + def get_torch_device( *, allow_xpu: bool = False, @@ -92,7 +100,16 @@ def get_torch_device( allow_directml: Check for DirectML (Windows) support. allow_mps: Allow MPS (Apple Silicon). If False, MPS falls back to CPU. force_cpu_on_mac: Force CPU on macOS regardless of GPU availability. + + The VOICEBOX_FORCE_CPU override wins over every other candidate, and is + resolved before torch is imported so it still works when the installed + build is the reason CPU is wanted. """ + # Stripped: on Windows, where this override matters most, it is usually set + # through the GUI environment editor. + if os.environ.get(FORCE_CPU_ENV_VAR, "").strip() == FORCE_CPU_ENABLED_VALUE: + return "cpu" + if force_cpu_on_mac and platform.system() == "Darwin": return "cpu" diff --git a/backend/tests/test_force_cpu_env.py b/backend/tests/test_force_cpu_env.py new file mode 100644 index 00000000..d8cd35a0 --- /dev/null +++ b/backend/tests/test_force_cpu_env.py @@ -0,0 +1,80 @@ +""" +Regression tests for the VOICEBOX_FORCE_CPU environment override. + +The docs promise (docs/content/docs/developer/tts-generation.mdx) that +get_torch_device() layers "VOICEBOX_FORCE_CPU environment override" ahead of +CUDA/XPU/MPS detection, and gpu-acceleration.mdx tells users to set it to fall +back to CPU when the bundled PyTorch has no kernels for their GPU. + +torch is stubbed through sys.modules so these run without a torch install and +without any GPU. + +Usage: + python -m pytest backend/tests/test_force_cpu_env.py -v +""" + +import sys + +import pytest + +from backend.backends.base import get_torch_device + +# The documented public name and value of the override. Pinned here independently +# of the production constants so a rename of either fails these tests. +FORCE_CPU_ENV_VAR = "VOICEBOX_FORCE_CPU" + +# Sentinel for "the variable is not set at all". +UNSET = None + + +class _FakeCuda: + @staticmethod + def is_available() -> bool: + return True + + +class _FakeTorch: + """Minimal stand-in for a CUDA-enabled torch install.""" + + cuda = _FakeCuda + + +@pytest.fixture +def cuda_available(monkeypatch): + """Make torch report a usable CUDA device without installing torch.""" + monkeypatch.setitem(sys.modules, "torch", _FakeTorch) + + +def _set_override(monkeypatch, value): + if value is UNSET: + monkeypatch.delenv(FORCE_CPU_ENV_VAR, raising=False) + else: + monkeypatch.setenv(FORCE_CPU_ENV_VAR, value) + + +@pytest.mark.parametrize("value", ["1", " 1 "]) +def test_force_cpu_wins_over_available_cuda(monkeypatch, cuda_available, value): + """The documented value must beat an otherwise usable CUDA device. + + Surrounding whitespace is tolerated: on Windows, where this override + matters most, it is typically set through the GUI environment editor.""" + _set_override(monkeypatch, value) + + assert get_torch_device(allow_xpu=True, allow_directml=True, allow_mps=True) == "cpu" + + +@pytest.mark.parametrize("value", [UNSET, "", "0"]) +def test_without_override_cuda_is_still_selected(monkeypatch, cuda_available, value): + """Unset or disabled must not disturb normal device detection.""" + _set_override(monkeypatch, value) + + assert get_torch_device() == "cuda" + + +def test_force_cpu_does_not_need_torch(monkeypatch): + """The override is honoured before torch is imported, so it works on a + broken/incompatible torch install — which is the case it exists for.""" + _set_override(monkeypatch, "1") + monkeypatch.setitem(sys.modules, "torch", None) # makes `import torch` raise + + assert get_torch_device() == "cpu" From 4a03cd7e778fd50218c5c7355c07517cb04eb712 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:32:12 +0000 Subject: [PATCH 12/35] fix(backend): log when VOICEBOX_FORCE_CPU forces the CPU device Without a trace the Settings GPU label and /health keep reporting CUDA while generation runs on CPU, so the user has no way to confirm the override engaged. --- backend/backends/base.py | 1 + 1 file changed, 1 insertion(+) diff --git a/backend/backends/base.py b/backend/backends/base.py index fa1c90ac..83acf2c3 100644 --- a/backend/backends/base.py +++ b/backend/backends/base.py @@ -108,6 +108,7 @@ def get_torch_device( # Stripped: on Windows, where this override matters most, it is usually set # through the GUI environment editor. if os.environ.get(FORCE_CPU_ENV_VAR, "").strip() == FORCE_CPU_ENABLED_VALUE: + logger.info("%s=%s set, forcing CPU device", FORCE_CPU_ENV_VAR, FORCE_CPU_ENABLED_VALUE) return "cpu" if force_cpu_on_mac and platform.system() == "Darwin": From 0e1e9922a887154b225d3c203557fa02190cbd9a Mon Sep 17 00:00:00 2001 From: devangkantharia Date: Sat, 8 Aug 2026 04:18:19 +0530 Subject: [PATCH 13/35] fix(dictation): skip LLM readiness gate when auto_refine is off (#753) When the user disables auto_refine (LLM polish) in Settings, the Qwen refinement model is no longer required for dictation to arm. Previously, canRecord checked llmReady unconditionally, so useChordSync called disable_hotkey whenever Qwen was not downloaded -- even if the user never intended to use refinement. Changes: - useDictationReadiness: gate llmReady behind autoRefine in canRecord, missing, and the polling predicates so hotkeys arm with just Whisper STT when refinement is off. - DictationReadinessChecklist: hide the LLM row when autoRefine is false -- the checklist now shows only Whisper STT. Closes #753 --- .../DictationReadinessChecklist.tsx | 2 +- app/src/lib/hooks/useDictationReadiness.ts | 24 ++++++++++++------- 2 files changed, 17 insertions(+), 9 deletions(-) diff --git a/app/src/components/CapturesTab/DictationReadinessChecklist.tsx b/app/src/components/CapturesTab/DictationReadinessChecklist.tsx index 136f1f58..ff72e9e3 100644 --- a/app/src/components/CapturesTab/DictationReadinessChecklist.tsx +++ b/app/src/components/CapturesTab/DictationReadinessChecklist.tsx @@ -222,7 +222,7 @@ export function DictationReadinessChecklist({ /> )} - {readiness.llm && ( + {readiness.autoRefine && readiness.llm && ( } title={t('captures.readiness.llm.label', { name: readiness.llm.display_name })} diff --git a/app/src/lib/hooks/useDictationReadiness.ts b/app/src/lib/hooks/useDictationReadiness.ts index fd755a2f..2b985e0b 100644 --- a/app/src/lib/hooks/useDictationReadiness.ts +++ b/app/src/lib/hooks/useDictationReadiness.ts @@ -3,6 +3,7 @@ import { useAccessibilityPermission } from '@/components/AccessibilityGate/Acces import { useInputMonitoringPermission } from '@/components/InputMonitoringGate/InputMonitoringGate'; import { apiClient } from '@/lib/api/client'; import type { ModelReadiness } from '@/lib/api/types'; +import { useCaptureSettings } from '@/lib/hooks/useSettings'; import { usePlatform } from '@/platform/PlatformContext'; const READINESS_POLL_INTERVAL_MS = 5_000; @@ -17,6 +18,8 @@ export interface DictationReadiness { missing: ReadinessGate[]; stt: ModelReadiness | undefined; llm: ModelReadiness | undefined; + /** Whether the user has auto-refine (LLM polish) enabled. */ + autoRefine: boolean; inputMonitoring: boolean; accessibility: boolean; refetch: () => void; @@ -46,6 +49,8 @@ export interface DictationReadiness { export function useDictationReadiness(): DictationReadiness { const platform = usePlatform(); const isTauri = platform.metadata.isTauri; + const { settings } = useCaptureSettings(); + const autoRefine = settings?.auto_refine ?? true; const { needsPermission: inputMonNeeds, @@ -61,17 +66,19 @@ export function useDictationReadiness(): DictationReadiness { const { data, isLoading, refetch } = useQuery({ queryKey: ['capture-readiness'], queryFn: () => apiClient.getCaptureReadiness(), - // Poll only while a model is still missing/downloading. Once both are - // green the endpoint's answer can't change until the user swaps models - // in settings, and that path invalidates the query explicitly from - // useSettings. refetchOnWindowFocus stays gated to the same condition. + // Poll only while a required model is still missing/downloading. Once + // all required models are green the endpoint's answer can't change + // until the user swaps models in settings, and that path invalidates + // the query explicitly from useSettings. refetchOnWindowFocus stays + // gated to the same condition. refetchInterval: (query) => { const d = query.state.data; - return d && d.stt.ready && d.llm.ready ? false : READINESS_POLL_INTERVAL_MS; + const allGreen = d && d.stt.ready && (!autoRefine || d.llm.ready); + return allGreen ? false : READINESS_POLL_INTERVAL_MS; }, refetchOnWindowFocus: (query) => { const d = query.state.data; - return !(d && d.stt.ready && d.llm.ready); + return !(d && d.stt.ready && (!autoRefine || d.llm.ready)); }, }); @@ -84,10 +91,10 @@ export function useDictationReadiness(): DictationReadiness { const missing: ReadinessGate[] = []; if (!sttReady) missing.push('stt'); - if (!llmReady) missing.push('llm'); + if (autoRefine && !llmReady) missing.push('llm'); if (!inputMonitoring) missing.push('input_monitoring'); if (!accessibility) missing.push('accessibility'); - const canRecord = sttReady && llmReady && inputMonitoring; + const canRecord = sttReady && (!autoRefine || llmReady) && inputMonitoring; return { isLoading, @@ -96,6 +103,7 @@ export function useDictationReadiness(): DictationReadiness { missing, stt: data?.stt, llm: data?.llm, + autoRefine, inputMonitoring, accessibility, refetch: () => { From ca56137ca0f17efde9e81ef988bec612e7b77e4c Mon Sep 17 00:00:00 2001 From: devangkantharia Date: Sat, 8 Aug 2026 22:10:13 +0530 Subject: [PATCH 14/35] fix(settings): invalidate capture-readiness on auto_refine toggle (#753) --- app/src/lib/hooks/useSettings.ts | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/app/src/lib/hooks/useSettings.ts b/app/src/lib/hooks/useSettings.ts index 40750be7..f8f6aad6 100644 --- a/app/src/lib/hooks/useSettings.ts +++ b/app/src/lib/hooks/useSettings.ts @@ -48,7 +48,11 @@ export function useCaptureSettings() { // call, but its cached response keeps serving the previous // model's state until the next 5 s poll. Invalidate on model // swaps so the readiness checklist re-checks immediately. - if (patch.stt_model !== undefined || patch.llm_model !== undefined) { + if ( + patch.stt_model !== undefined || + patch.llm_model !== undefined || + patch.auto_refine !== undefined + ) { queryClient.invalidateQueries({ queryKey: ['capture-readiness'] }); } }, From 9ba2b330699b9372bdc272b0c688dd33540b1967 Mon Sep 17 00:00:00 2001 From: Tomas Rimkus Date: Fri, 21 Aug 2026 15:21:24 +0300 Subject: [PATCH 15/35] fix(ui): make long error toasts readable and copyable A failed generation put the server's error straight into a toast. The transformers "Unrecognized model" error is ~4.8KB, of which the first sentence carries the meaning and the remaining 4.7KB is an alphabetical list of every architecture it knows. In a 420px toast with overflow-hidden and no scroll, that clipped the text at both ends and pushed the close button off-screen: unreadable and undismissable. - ToastDescription is capped at 40vh and scrolls, wraps on whitespace and breaks long unspaced tokens so a path cannot widen the toast. - The toast aligns to the start rather than centring, so the title stays visible next to a tall description. - condenseError() keeps the head of an oversized error, cutting at the first newline or the last sentence end inside a 400-char budget, and reports how many characters it dropped. Short errors pass through untouched. - When it does truncate, the toast offers a Copy action for the full text and points at Settings -> Logs. Verified against the real 4795-char error: 4795 -> 400 chars keeping both meaningful sentences. The hook moves to .tsx to render ToastAction, matching useAutoUpdater.tsx which is a .tsx hook for the same reason. createElement was tried first but this repo's ToastActionElement type is the older shadcn definition (ReactElement) which only accepts JSX-constructed elements. Co-Authored-By: Claude Opus 5 (1M context) --- app/src/components/ui/toast.tsx | 14 +++- ...nProgress.ts => useGenerationProgress.tsx} | 21 +++++- app/src/lib/utils/errorText.ts | 67 +++++++++++++++++++ 3 files changed, 99 insertions(+), 3 deletions(-) rename app/src/lib/hooks/{useGenerationProgress.ts => useGenerationProgress.tsx} (86%) create mode 100644 app/src/lib/utils/errorText.ts diff --git a/app/src/components/ui/toast.tsx b/app/src/components/ui/toast.tsx index 35150afd..8243b6d4 100644 --- a/app/src/components/ui/toast.tsx +++ b/app/src/components/ui/toast.tsx @@ -22,7 +22,9 @@ const ToastViewport = React.forwardRef< ToastViewport.displayName = ToastPrimitives.Viewport.displayName; const toastVariants = cva( - 'group pointer-events-auto relative flex w-full items-center justify-between space-x-4 overflow-hidden rounded-md border p-6 pr-8 shadow-lg transition-all data-[swipe=cancel]:translate-x-0 data-[swipe=end]:translate-x-[var(--radix-toast-swipe-end-x)] data-[swipe=move]:translate-x-[var(--radix-toast-swipe-move-x)] data-[swipe=move]:transition-none data-[state=open]:animate-in data-[state=closed]:animate-out data-[swipe=end]:animate-out data-[state=closed]:fade-out-80 data-[state=closed]:slide-out-to-right-full data-[state=open]:slide-in-from-top-full data-[state=open]:sm:slide-in-from-bottom-full', + // items-start rather than items-center: a long description scrolls inside a + // capped box, and centring it would push the title out of view. + 'group pointer-events-auto relative flex w-full items-start justify-between space-x-4 overflow-hidden rounded-md border p-6 pr-8 shadow-lg transition-all data-[swipe=cancel]:translate-x-0 data-[swipe=end]:translate-x-[var(--radix-toast-swipe-end-x)] data-[swipe=move]:translate-x-[var(--radix-toast-swipe-move-x)] data-[swipe=move]:transition-none data-[state=open]:animate-in data-[state=closed]:animate-out data-[swipe=end]:animate-out data-[state=closed]:fade-out-80 data-[state=closed]:slide-out-to-right-full data-[state=open]:slide-in-from-top-full data-[state=open]:sm:slide-in-from-bottom-full', { variants: { variant: { @@ -98,7 +100,15 @@ const ToastDescription = React.forwardRef< >(({ className, ...props }, ref) => ( )); diff --git a/app/src/lib/hooks/useGenerationProgress.ts b/app/src/lib/hooks/useGenerationProgress.tsx similarity index 86% rename from app/src/lib/hooks/useGenerationProgress.ts rename to app/src/lib/hooks/useGenerationProgress.tsx index 23eb9623..d81e9547 100644 --- a/app/src/lib/hooks/useGenerationProgress.ts +++ b/app/src/lib/hooks/useGenerationProgress.tsx @@ -1,7 +1,9 @@ import { useQueryClient } from '@tanstack/react-query'; import { useEffect, useRef } from 'react'; +import { ToastAction } from '@/components/ui/toast'; import { useToast } from '@/components/ui/use-toast'; import { apiClient } from '@/lib/api/client'; +import { condenseError } from '@/lib/utils/errorText'; import { useGenerationSettings } from '@/lib/hooks/useSettings'; import { useGenerationStore } from '@/stores/generationStore'; import { usePlayerStore } from '@/stores/playerStore'; @@ -131,10 +133,27 @@ export function useGenerationProgress() { queryClient.refetchQueries({ queryKey: ['history'] }); + const condensed = condenseError( + data.error || 'An error occurred during generation', + ); toast({ title: data.status === 'not_found' ? 'Generation not found' : 'Generation failed', - description: data.error || 'An error occurred during generation', + description: condensed.truncated + ? `${condensed.display}\n\n(${condensed.omitted} more characters — copy for the full error, or see Settings → Logs)` + : condensed.display, variant: 'destructive', + // Only offered when there is more to read than what is shown, so + // the common short error keeps a plain toast. + action: condensed.truncated ? ( + { + void navigator.clipboard.writeText(condensed.full); + }} + > + Copy + + ) : undefined, }); } } catch { diff --git a/app/src/lib/utils/errorText.ts b/app/src/lib/utils/errorText.ts new file mode 100644 index 00000000..9a41373e --- /dev/null +++ b/app/src/lib/utils/errorText.ts @@ -0,0 +1,67 @@ +/** + * Condense a server error for display in a toast. + * + * Some backend errors are enormous and mostly noise. The transformers + * "Unrecognized model" error is ~4.8KB, of which the first sentence carries all + * the meaning and the remaining 4.7KB is an alphabetical list of every model + * architecture it knows about. Rendering that in a 420px toast clipped the text + * at both ends and pushed the close button off-screen. + * + * The rule is deliberately generic rather than pattern-matching any one + * library: keep the head, cut at the most natural boundary available inside the + * budget, and report how much was dropped so nobody assumes they read it all. + */ + +/** Characters of an error worth putting in a toast. Roughly the first + * paragraph — enough for a sentence or two of real message. */ +const TOAST_ERROR_BUDGET = 400; + +/** Below this there is nothing to gain by condensing. */ +const MIN_TO_CONDENSE = TOAST_ERROR_BUDGET + 120; + +export interface CondensedError { + /** What to show in the toast. */ + display: string; + /** The untouched original, for copying. */ + full: string; + /** Whether `display` is shorter than `full`. */ + truncated: boolean; + /** How many characters `display` leaves out. */ + omitted: number; +} + +export function condenseError(raw: string | null | undefined): CondensedError { + const full = (raw ?? '').trim(); + + if (full.length <= MIN_TO_CONDENSE) { + return { display: full, full, truncated: false, omitted: 0 }; + } + + // A traceback's first line is nearly always the message; prefer it whenever + // it fits, since a newline is a stronger boundary than any punctuation. + const firstLine = full.split('\n', 1)[0].trim(); + let head = + firstLine.length > 0 && firstLine.length <= TOAST_ERROR_BUDGET + ? firstLine + : full.slice(0, TOAST_ERROR_BUDGET); + + if (head.length < full.length && head === full.slice(0, head.length)) { + // Back off to the last sentence end inside the budget so the text does not + // stop mid-word. Only accept it if it keeps most of the budget — otherwise + // a stray early period would throw away usable context. + const lastStop = Math.max(head.lastIndexOf('. '), head.lastIndexOf('? ')); + if (lastStop > TOAST_ERROR_BUDGET * 0.4) { + head = head.slice(0, lastStop + 1); + } + } + + head = head.trimEnd(); + const omitted = full.length - head.length; + + // Guard against the boundary search having produced nothing shorter. + if (omitted <= 0) { + return { display: full, full, truncated: false, omitted: 0 }; + } + + return { display: `${head} …`, full, truncated: true, omitted }; +} From 96c6f9cad92a059bfb664acc792471945d484898 Mon Sep 17 00:00:00 2001 From: Tomas Rimkus Date: Fri, 21 Aug 2026 18:06:38 +0300 Subject: [PATCH 16/35] fix(ui): preserve the original error text and handle clipboard failures Two CodeRabbit findings on #1058. condenseError trimmed the input before storing it in `full`, which is documented as the untouched original and is what the Copy action hands over. The trim now applies only to the working copy used for measuring and cutting, so `full` is byte-for-byte what the server sent while `display` and `omitted` still ignore surrounding blank space. The Copy handler called navigator.clipboard.writeText with no guard. Outside a secure context the property access itself throws, and writeText rejects when permission is denied; neither was handled, so a click could become an unhandled rejection with no sign that nothing was copied. Both paths are now caught and reported, pointing at Settings -> Logs as the fallback. Not taken: aligning MIN_TO_CONDENSE with the 400-char budget. The gap is deliberate -- cutting a 450-char error to 400 saves 50 characters in a description that already scrolls, and no Copy action is needed there because `display` holds the whole message. Documented in place. Co-Authored-By: Claude Opus 5 (1M context) --- app/src/lib/hooks/useGenerationProgress.tsx | 18 +++++++++- app/src/lib/utils/errorText.ts | 37 ++++++++++++++------- 2 files changed, 42 insertions(+), 13 deletions(-) diff --git a/app/src/lib/hooks/useGenerationProgress.tsx b/app/src/lib/hooks/useGenerationProgress.tsx index d81e9547..9a345a70 100644 --- a/app/src/lib/hooks/useGenerationProgress.tsx +++ b/app/src/lib/hooks/useGenerationProgress.tsx @@ -148,7 +148,23 @@ export function useGenerationProgress() { { - void navigator.clipboard.writeText(condensed.full); + // Two ways this fails: the Clipboard API is absent outside + // a secure context (property access throws), or writeText + // rejects because permission was denied. The try/catch + // covers both, so the click never becomes an unhandled + // rejection and the user is told nothing was copied. + void (async () => { + try { + await navigator.clipboard.writeText(condensed.full); + } catch { + toast({ + title: 'Could not copy', + description: + 'Clipboard unavailable. The full error is in Settings → Logs.', + variant: 'destructive', + }); + } + })(); }} > Copy diff --git a/app/src/lib/utils/errorText.ts b/app/src/lib/utils/errorText.ts index 9a41373e..c5ebce73 100644 --- a/app/src/lib/utils/errorText.ts +++ b/app/src/lib/utils/errorText.ts @@ -16,36 +16,49 @@ * paragraph — enough for a sentence or two of real message. */ const TOAST_ERROR_BUDGET = 400; -/** Below this there is nothing to gain by condensing. */ +/** Below this, condensing is not worth it and the whole message is shown. + * + * Deliberately above the budget rather than equal to it. An error of 450 + * characters would otherwise be cut to 400 to save 50 — a worse result than + * showing all of it, since the description scrolls anyway. The gap also gives + * the threshold hysteresis instead of flipping between full and truncated + * around a single character. No Copy action is offered in this range because + * nothing is being withheld: `display` already holds the entire message. + */ const MIN_TO_CONDENSE = TOAST_ERROR_BUDGET + 120; export interface CondensedError { - /** What to show in the toast. */ + /** What to show in the toast. Trimmed, and shortened when oversized. */ display: string; - /** The untouched original, for copying. */ + /** The original string exactly as received, for copying. Never modified. */ full: string; - /** Whether `display` is shorter than `full`. */ + /** Whether `display` omits part of the message. */ truncated: boolean; /** How many characters `display` leaves out. */ omitted: number; } export function condenseError(raw: string | null | undefined): CondensedError { - const full = (raw ?? '').trim(); + // `full` is what the Copy action hands over, so it stays byte-for-byte what + // the server sent. All the measuring and cutting below works on the trimmed + // copy instead — surrounding blank space should not count toward the budget + // or the omitted count. + const full = raw ?? ''; + const text = full.trim(); - if (full.length <= MIN_TO_CONDENSE) { - return { display: full, full, truncated: false, omitted: 0 }; + if (text.length <= MIN_TO_CONDENSE) { + return { display: text, full, truncated: false, omitted: 0 }; } // A traceback's first line is nearly always the message; prefer it whenever // it fits, since a newline is a stronger boundary than any punctuation. - const firstLine = full.split('\n', 1)[0].trim(); + const firstLine = text.split('\n', 1)[0].trim(); let head = firstLine.length > 0 && firstLine.length <= TOAST_ERROR_BUDGET ? firstLine - : full.slice(0, TOAST_ERROR_BUDGET); + : text.slice(0, TOAST_ERROR_BUDGET); - if (head.length < full.length && head === full.slice(0, head.length)) { + if (head.length < text.length && head === text.slice(0, head.length)) { // Back off to the last sentence end inside the budget so the text does not // stop mid-word. Only accept it if it keeps most of the budget — otherwise // a stray early period would throw away usable context. @@ -56,11 +69,11 @@ export function condenseError(raw: string | null | undefined): CondensedError { } head = head.trimEnd(); - const omitted = full.length - head.length; + const omitted = text.length - head.length; // Guard against the boundary search having produced nothing shorter. if (omitted <= 0) { - return { display: full, full, truncated: false, omitted: 0 }; + return { display: text, full, truncated: false, omitted: 0 }; } return { display: `${head} …`, full, truncated: true, omitted }; From e3b9c989776d6c0153693199ce76d627d827d31a Mon Sep 17 00:00:00 2001 From: Tomas Rimkus Date: Fri, 21 Aug 2026 18:21:30 +0300 Subject: [PATCH 17/35] fix(ui): return untruncated errors exactly as received MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to discussion_r3831361300 on #1058. Both non-truncated return paths handed back the trimmed working copy, so an error like " request timed out\n" came back altered even though nothing had been omitted. That contradicted the stated intent that short errors pass through untouched, and left `display` differing from `full` for no reason. `display` is now byte-identical to `full` whenever `truncated` is false — the trimmed copy is only used for measuring against the budget and for building the shortened head. Documented on the field. Verified across padded short errors, clean short errors, empty and whitespace-only input, and either side of the threshold: display === full on every untruncated case, and `full` matches the input exactly in all of them. One visible consequence: with whitespace-pre-wrap on the description, an error carrying leading or trailing newlines now renders with that blank space. Trivial for the messages this sees in practice, and the alternative was silently editing the text. Co-Authored-By: Claude Opus 5 (1M context) --- app/src/lib/utils/errorText.ts | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/app/src/lib/utils/errorText.ts b/app/src/lib/utils/errorText.ts index c5ebce73..5b26b38d 100644 --- a/app/src/lib/utils/errorText.ts +++ b/app/src/lib/utils/errorText.ts @@ -28,7 +28,12 @@ const TOAST_ERROR_BUDGET = 400; const MIN_TO_CONDENSE = TOAST_ERROR_BUDGET + 120; export interface CondensedError { - /** What to show in the toast. Trimmed, and shortened when oversized. */ + /** What to show in the toast. + * + * Byte-identical to `full` whenever nothing is omitted, so a message that + * fits is never altered. Only an oversized one is rewritten, into a trimmed + * head followed by an ellipsis. + */ display: string; /** The original string exactly as received, for copying. Never modified. */ full: string; @@ -40,14 +45,15 @@ export interface CondensedError { export function condenseError(raw: string | null | undefined): CondensedError { // `full` is what the Copy action hands over, so it stays byte-for-byte what - // the server sent. All the measuring and cutting below works on the trimmed - // copy instead — surrounding blank space should not count toward the budget - // or the omitted count. + // the server sent. The measuring and cutting below works on a trimmed copy + // instead — surrounding blank space should not count toward the budget or + // the omitted count — but `text` is never what gets returned as `display` + // unless the message is actually being shortened. const full = raw ?? ''; const text = full.trim(); if (text.length <= MIN_TO_CONDENSE) { - return { display: text, full, truncated: false, omitted: 0 }; + return { display: full, full, truncated: false, omitted: 0 }; } // A traceback's first line is nearly always the message; prefer it whenever @@ -71,9 +77,10 @@ export function condenseError(raw: string | null | undefined): CondensedError { head = head.trimEnd(); const omitted = text.length - head.length; - // Guard against the boundary search having produced nothing shorter. + // Guard against the boundary search having produced nothing shorter. Nothing + // is omitted here either, so the message goes back exactly as it arrived. if (omitted <= 0) { - return { display: text, full, truncated: false, omitted: 0 }; + return { display: full, full, truncated: false, omitted: 0 }; } return { display: `${head} …`, full, truncated: true, omitted }; From 8086c818b6eab69a8a7d73043afeeed579659728 Mon Sep 17 00:00:00 2001 From: Tomas Rimkus Date: Fri, 21 Aug 2026 18:31:35 +0300 Subject: [PATCH 18/35] refactor(ui): name the head budget and the truncation suffix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the budget comment on #1058 by the second route the review offered -- defining the constant as the head budget rather than reserving the suffix inside it. TOAST_ERROR_BUDGET read as though it bounded `display`, but `display` is the head plus " …", so it could be 402. Renamed to HEAD_BUDGET and documented as bounding the message rather than the rendered string, with the suffix now a named constant instead of a literal in the template. Reserving the two characters was the alternative, but nothing downstream has a hard limit -- the description box scrolls -- so it would have shortened the message to satisfy a round number. Verified: head <= 400 and display <= 402 on every truncating input, including no-space text, a short first line, sentence-boundary backoff and the real 4795-char error, with the untouched-when-not-truncated and exact-`full` invariants still holding. Co-Authored-By: Claude Opus 5 (1M context) --- app/src/lib/utils/errorText.ts | 26 ++++++++++++++++++-------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/app/src/lib/utils/errorText.ts b/app/src/lib/utils/errorText.ts index 5b26b38d..1e42a69a 100644 --- a/app/src/lib/utils/errorText.ts +++ b/app/src/lib/utils/errorText.ts @@ -12,9 +12,19 @@ * budget, and report how much was dropped so nobody assumes they read it all. */ -/** Characters of an error worth putting in a toast. Roughly the first - * paragraph — enough for a sentence or two of real message. */ -const TOAST_ERROR_BUDGET = 400; +/** Longest head of an oversized error kept for the toast, before the ellipsis. + * + * Roughly the first paragraph — enough for a sentence or two of real message. + * This bounds the *message* rather than the rendered string: `display` is this + * plus `TRUNCATION_SUFFIX` when something was cut. Spending two of these + * characters on the ellipsis instead would shorten the message to no purpose, + * since nothing downstream has a hard character limit — the description box + * scrolls. + */ +const HEAD_BUDGET = 400; + +/** Marks a `display` value as incomplete. Appended after the head. */ +const TRUNCATION_SUFFIX = ' …'; /** Below this, condensing is not worth it and the whole message is shown. * @@ -25,7 +35,7 @@ const TOAST_ERROR_BUDGET = 400; * around a single character. No Copy action is offered in this range because * nothing is being withheld: `display` already holds the entire message. */ -const MIN_TO_CONDENSE = TOAST_ERROR_BUDGET + 120; +const MIN_TO_CONDENSE = HEAD_BUDGET + 120; export interface CondensedError { /** What to show in the toast. @@ -60,16 +70,16 @@ export function condenseError(raw: string | null | undefined): CondensedError { // it fits, since a newline is a stronger boundary than any punctuation. const firstLine = text.split('\n', 1)[0].trim(); let head = - firstLine.length > 0 && firstLine.length <= TOAST_ERROR_BUDGET + firstLine.length > 0 && firstLine.length <= HEAD_BUDGET ? firstLine - : text.slice(0, TOAST_ERROR_BUDGET); + : text.slice(0, HEAD_BUDGET); if (head.length < text.length && head === text.slice(0, head.length)) { // Back off to the last sentence end inside the budget so the text does not // stop mid-word. Only accept it if it keeps most of the budget — otherwise // a stray early period would throw away usable context. const lastStop = Math.max(head.lastIndexOf('. '), head.lastIndexOf('? ')); - if (lastStop > TOAST_ERROR_BUDGET * 0.4) { + if (lastStop > HEAD_BUDGET * 0.4) { head = head.slice(0, lastStop + 1); } } @@ -83,5 +93,5 @@ export function condenseError(raw: string | null | undefined): CondensedError { return { display: full, full, truncated: false, omitted: 0 }; } - return { display: `${head} …`, full, truncated: true, omitted }; + return { display: `${head}${TRUNCATION_SUFFIX}`, full, truncated: true, omitted }; } From 872121c0f71792e73645441ba300ebbfd66190a8 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:20:49 +0000 Subject: [PATCH 19/35] style: apply biome import order and formatting to useGenerationProgress --- app/src/lib/hooks/useGenerationProgress.tsx | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/app/src/lib/hooks/useGenerationProgress.tsx b/app/src/lib/hooks/useGenerationProgress.tsx index 9a345a70..9c15e37a 100644 --- a/app/src/lib/hooks/useGenerationProgress.tsx +++ b/app/src/lib/hooks/useGenerationProgress.tsx @@ -3,8 +3,8 @@ import { useEffect, useRef } from 'react'; import { ToastAction } from '@/components/ui/toast'; import { useToast } from '@/components/ui/use-toast'; import { apiClient } from '@/lib/api/client'; -import { condenseError } from '@/lib/utils/errorText'; import { useGenerationSettings } from '@/lib/hooks/useSettings'; +import { condenseError } from '@/lib/utils/errorText'; import { useGenerationStore } from '@/stores/generationStore'; import { usePlayerStore } from '@/stores/playerStore'; @@ -133,9 +133,7 @@ export function useGenerationProgress() { queryClient.refetchQueries({ queryKey: ['history'] }); - const condensed = condenseError( - data.error || 'An error occurred during generation', - ); + const condensed = condenseError(data.error || 'An error occurred during generation'); toast({ title: data.status === 'not_found' ? 'Generation not found' : 'Generation failed', description: condensed.truncated From e440b44ae7003ea916e12772958781f28d0516d5 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:27:34 +0000 Subject: [PATCH 20/35] fix(ui): keep a first line that fits whole instead of re-cutting it at a sentence end --- app/src/lib/utils/errorText.ts | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/app/src/lib/utils/errorText.ts b/app/src/lib/utils/errorText.ts index 1e42a69a..f1fc7eea 100644 --- a/app/src/lib/utils/errorText.ts +++ b/app/src/lib/utils/errorText.ts @@ -69,15 +69,15 @@ export function condenseError(raw: string | null | undefined): CondensedError { // A traceback's first line is nearly always the message; prefer it whenever // it fits, since a newline is a stronger boundary than any punctuation. const firstLine = text.split('\n', 1)[0].trim(); - let head = - firstLine.length > 0 && firstLine.length <= HEAD_BUDGET - ? firstLine - : text.slice(0, HEAD_BUDGET); + const useFirstLine = firstLine.length > 0 && firstLine.length <= HEAD_BUDGET; + let head = useFirstLine ? firstLine : text.slice(0, HEAD_BUDGET); - if (head.length < text.length && head === text.slice(0, head.length)) { - // Back off to the last sentence end inside the budget so the text does not - // stop mid-word. Only accept it if it keeps most of the budget — otherwise - // a stray early period would throw away usable context. + if (!useFirstLine) { + // A raw slice can stop mid-word, so back off to the last sentence end + // inside the budget. A first line that fit already ends at a newline, the + // stronger boundary, and is kept whole. Only accept the backoff if it + // keeps most of the budget — otherwise a stray early period would throw + // away usable context. const lastStop = Math.max(head.lastIndexOf('. '), head.lastIndexOf('? ')); if (lastStop > HEAD_BUDGET * 0.4) { head = head.slice(0, lastStop + 1); From c339d2c324b1da94a521efa6e27f3a2913822ac6 Mon Sep 17 00:00:00 2001 From: Serge Simono Date: Mon, 17 Aug 2026 05:31:05 +0200 Subject: [PATCH 21/35] fix(speak): honour the voice profile's language instead of forcing English Both speak surfaces built their GenerationRequest with a hardcoded "en" fallback and never consulted the resolved profile, so a profile created with language="fr" was still synthesised as English unless the caller passed language= explicitly. This hurts the MCP path most: an agent calling voicebox.speak has no way to know the bound profile's language, so it cannot pass the argument either. Every agent-triggered generation on a non-English profile came out with an English accent. The fallback chain is now explicit argument -> resolved profile's language -> "en", which matches how engine and personality already consult the resolved binding. The "en" backstop is kept so profiles with no language set behave exactly as before. Adds backend/tests/test_speak_language.py covering both surfaces: the fallback, explicit-argument precedence, and the unchanged "en" default. Co-Authored-By: Claude Opus 5 (1M context) --- backend/mcp_server/tools.py | 2 +- backend/routes/speak.py | 2 +- backend/tests/test_speak_language.py | 163 +++++++++++++++++++++++++++ 3 files changed, 165 insertions(+), 2 deletions(-) create mode 100644 backend/tests/test_speak_language.py diff --git a/backend/mcp_server/tools.py b/backend/mcp_server/tools.py index 093e57ae..61003f11 100644 --- a/backend/mcp_server/tools.py +++ b/backend/mcp_server/tools.py @@ -104,7 +104,7 @@ def register_tools(mcp: FastMCP) -> None: profile_name=vp.name, text=text, engine=resolved_engine, - language=language, + language=language or vp.language, personality=use_persona, model_size=model_size, db=db, diff --git a/backend/routes/speak.py b/backend/routes/speak.py index 0c81846c..92efc75d 100644 --- a/backend/routes/speak.py +++ b/backend/routes/speak.py @@ -75,7 +75,7 @@ async def speak( models.GenerationRequest( profile_id=profile.id, text=data.text, - language=data.language or "en", + language=data.language or profile.language or "en", engine=engine, personality=bool(personality_flag), ), diff --git a/backend/tests/test_speak_language.py b/backend/tests/test_speak_language.py new file mode 100644 index 00000000..7a8f1e5e --- /dev/null +++ b/backend/tests/test_speak_language.py @@ -0,0 +1,163 @@ +"""Tests for voice-profile language fallback on the two speak surfaces. + +Both speak paths built their ``GenerationRequest`` with a hardcoded ``"en"`` +fallback and never consulted the resolved profile, so a profile created with +``language="fr"`` was still synthesised as English unless every caller passed +``language=`` explicitly. Agents going through MCP had no way to know the +profile's language, so they couldn't pass it either. + +These tests pin the fix: the fallback chain is now explicit argument → +resolved profile's language → ``"en"``, matching how ``engine`` and +``personality`` already consult the resolved binding. +""" + +import pytest + +import backend.routes.generations as generations +import backend.routes.speak as speak_route +from backend import models +from backend.mcp_server import tools + + +class _FakeGeneration: + """Minimal stand-in for GenerationResponse consumed by the speak paths.""" + + id = "gen-test" + status = "generating" + + def model_dump(self, mode="json"): + return {"id": self.id, "status": self.status} + + +class _FakeProfile: + def __init__(self, language): + self.id = "p1" + self.name = "Siwis" + self.language = language + self.personality = None + + +class _FakeQuery: + def filter(self, *args, **kwargs): + return self + + def first(self): + # No per-client binding — engine/personality fall through to their + # own defaults, leaving language as the only variable under test. + return None + + +class _FakeDB: + def query(self, *args, **kwargs): + return _FakeQuery() + + def close(self): + pass + + +class _FakeRequest: + """Stands in for starlette's Request — only headers are read.""" + + def __init__(self, client_id=None): + self.headers = {"X-Voicebox-Client-Id": client_id} if client_id else {} + + +@pytest.fixture +def captured_request(monkeypatch): + """Capture the GenerationRequest instead of running a real generation. + + Both speak paths import ``generate_speech`` lazily from + ``routes.generations``, so patching the attribute on that module + intercepts the call on either surface. + """ + captured = {} + + async def fake_generate_speech(req, db): + captured["req"] = req + return _FakeGeneration() + + monkeypatch.setattr(generations, "generate_speech", fake_generate_speech) + monkeypatch.setattr(speak_route.mcp_events, "publish", lambda *a, **k: None) + monkeypatch.setattr(tools.mcp_events, "publish", lambda *a, **k: None) + return captured + + +# ─── REST: POST /speak ──────────────────────────────────────────────────── + + +async def _call_rest(monkeypatch, profile_language, requested_language=None): + monkeypatch.setattr( + speak_route, + "resolve_profile", + lambda profile, client_id, db: _FakeProfile(profile_language), + ) + await speak_route.speak( + models.SpeakRequest(text="Bonjour", language=requested_language), + _FakeRequest(client_id="claude-code"), + _FakeDB(), + ) + + +async def test_rest_speak_falls_back_to_profile_language( + captured_request, monkeypatch +): + await _call_rest(monkeypatch, profile_language="fr") + assert captured_request["req"].language == "fr" + + +async def test_rest_speak_explicit_language_wins(captured_request, monkeypatch): + # An explicit argument still overrides the profile — a French profile can + # be asked to read an English string. + await _call_rest(monkeypatch, profile_language="fr", requested_language="en") + assert captured_request["req"].language == "en" + + +async def test_rest_speak_defaults_to_en_without_profile_language( + captured_request, monkeypatch +): + # Profiles predating the language column resolve to None; the "en" + # backstop keeps their behaviour unchanged. + await _call_rest(monkeypatch, profile_language=None) + assert captured_request["req"].language == "en" + + +# ─── MCP: voicebox.speak ────────────────────────────────────────────────── + + +async def _call_mcp(monkeypatch, profile_language, requested_language=None): + from mcp.server.fastmcp import FastMCP + + monkeypatch.setattr( + tools, + "resolve_profile", + lambda profile, client_id, db: _FakeProfile(profile_language), + ) + monkeypatch.setattr(tools, "get_db", lambda: iter([_FakeDB()])) + + mcp = FastMCP("test") + tools.register_tools(mcp) + args = {"text": "Bonjour"} + if requested_language is not None: + args["language"] = requested_language + await mcp.call_tool("voicebox.speak", args) + + +async def test_mcp_speak_falls_back_to_profile_language( + captured_request, monkeypatch +): + # The agent-facing path matters most: an MCP client can't know the + # profile's language, so omitting it must not silently mean English. + await _call_mcp(monkeypatch, profile_language="fr") + assert captured_request["req"].language == "fr" + + +async def test_mcp_speak_explicit_language_wins(captured_request, monkeypatch): + await _call_mcp(monkeypatch, profile_language="fr", requested_language="en") + assert captured_request["req"].language == "en" + + +async def test_mcp_speak_defaults_to_en_without_profile_language( + captured_request, monkeypatch +): + await _call_mcp(monkeypatch, profile_language=None) + assert captured_request["req"].language == "en" From 1a3942f3b3040aed326bdd0db0ad96e97425e880 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:23:46 +0000 Subject: [PATCH 22/35] style: ruff-format the new speak language tests --- backend/tests/test_speak_language.py | 16 ++++------------ 1 file changed, 4 insertions(+), 12 deletions(-) diff --git a/backend/tests/test_speak_language.py b/backend/tests/test_speak_language.py index 7a8f1e5e..97f9f57f 100644 --- a/backend/tests/test_speak_language.py +++ b/backend/tests/test_speak_language.py @@ -98,9 +98,7 @@ async def _call_rest(monkeypatch, profile_language, requested_language=None): ) -async def test_rest_speak_falls_back_to_profile_language( - captured_request, monkeypatch -): +async def test_rest_speak_falls_back_to_profile_language(captured_request, monkeypatch): await _call_rest(monkeypatch, profile_language="fr") assert captured_request["req"].language == "fr" @@ -112,9 +110,7 @@ async def test_rest_speak_explicit_language_wins(captured_request, monkeypatch): assert captured_request["req"].language == "en" -async def test_rest_speak_defaults_to_en_without_profile_language( - captured_request, monkeypatch -): +async def test_rest_speak_defaults_to_en_without_profile_language(captured_request, monkeypatch): # Profiles predating the language column resolve to None; the "en" # backstop keeps their behaviour unchanged. await _call_rest(monkeypatch, profile_language=None) @@ -142,9 +138,7 @@ async def _call_mcp(monkeypatch, profile_language, requested_language=None): await mcp.call_tool("voicebox.speak", args) -async def test_mcp_speak_falls_back_to_profile_language( - captured_request, monkeypatch -): +async def test_mcp_speak_falls_back_to_profile_language(captured_request, monkeypatch): # The agent-facing path matters most: an MCP client can't know the # profile's language, so omitting it must not silently mean English. await _call_mcp(monkeypatch, profile_language="fr") @@ -156,8 +150,6 @@ async def test_mcp_speak_explicit_language_wins(captured_request, monkeypatch): assert captured_request["req"].language == "en" -async def test_mcp_speak_defaults_to_en_without_profile_language( - captured_request, monkeypatch -): +async def test_mcp_speak_defaults_to_en_without_profile_language(captured_request, monkeypatch): await _call_mcp(monkeypatch, profile_language=None) assert captured_request["req"].language == "en" From 6ad47dda8928b412f27ac8937a80488b3cdf0795 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:30:39 +0000 Subject: [PATCH 23/35] test(speak): build the MCP test server from fastmcp, the package production imports --- backend/tests/test_speak_language.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/backend/tests/test_speak_language.py b/backend/tests/test_speak_language.py index 97f9f57f..0997a110 100644 --- a/backend/tests/test_speak_language.py +++ b/backend/tests/test_speak_language.py @@ -121,7 +121,10 @@ async def test_rest_speak_defaults_to_en_without_profile_language(captured_reque async def _call_mcp(monkeypatch, profile_language, requested_language=None): - from mcp.server.fastmcp import FastMCP + # Build the server from the same ``fastmcp`` package production imports so + # the registered ``voicebox.speak`` wrapper — where the profile fallback + # lives — is the code under test. + from fastmcp import FastMCP monkeypatch.setattr( tools, From f62aff0809a2f74da1703076a54c46c3e5dcc178 Mon Sep 17 00:00:00 2001 From: Alejandro Gaston Alvarez Date: Mon, 27 Jul 2026 11:58:06 +0200 Subject: [PATCH 24/35] fix(cache): don't treat orphaned .incomplete blobs as an in-progress download is_model_cached() marked a model as not-cached whenever any .incomplete blob existed in its cache dir, even when a completed blob with the same hash already sat next to it. A retried/concurrent download can leave this orphan behind after the real transfer already finished, which made the model appear perpetually "downloading" and re-trigger a full re-download on every load. Only .incomplete files with no matching completed blob now count as a genuinely in-progress download. --- backend/backends/base.py | 15 ++++-- backend/tests/test_is_model_cached.py | 74 +++++++++++++++++++++++++++ 2 files changed, 85 insertions(+), 4 deletions(-) create mode 100644 backend/tests/test_is_model_cached.py diff --git a/backend/backends/base.py b/backend/backends/base.py index 83acf2c3..9360be79 100644 --- a/backend/backends/base.py +++ b/backend/backends/base.py @@ -48,11 +48,18 @@ def is_model_cached( if not repo_cache.exists(): return False - # Incomplete blobs mean a download is still in progress + # An .incomplete blob means a download is still in progress -- unless + # a completed blob with the same hash already sits next to it, which + # happens when a retried/concurrent download leaves a stale .incomplete + # behind after the real transfer already finished. Only orphaned + # .incomplete files (no matching completed blob) count as "in progress". blobs_dir = repo_cache / "blobs" - if blobs_dir.exists() and any(blobs_dir.glob("*.incomplete")): - logger.debug(f"Found .incomplete files for {hf_repo}") - return False + if blobs_dir.exists(): + for incomplete in blobs_dir.glob("*.incomplete"): + completed = incomplete.with_name(incomplete.name.removesuffix(".incomplete")) + if not completed.exists(): + logger.debug(f"Found in-progress .incomplete file for {hf_repo}") + return False snapshots_dir = repo_cache / "snapshots" if not snapshots_dir.exists(): diff --git a/backend/tests/test_is_model_cached.py b/backend/tests/test_is_model_cached.py new file mode 100644 index 00000000..6b9d883f --- /dev/null +++ b/backend/tests/test_is_model_cached.py @@ -0,0 +1,74 @@ +""" +Unit tests for ``is_model_cached``'s handling of stale ``.incomplete`` blobs. + +A retried or concurrent download can leave an orphaned ``.incomplete`` file +next to its now-completed counterpart (same blob hash, no suffix). Only a +genuinely in-progress download -- an ``.incomplete`` with no completed blob +alongside it -- should mark the model as not cached. + +``is_model_cached`` is extracted and exec'd standalone (instead of importing +``backend.backends.base``) so this test doesn't pull in the module's sibling +imports (audio/progress/hf_progress/tasks), which in turn require the full +ML stack (torch/transformers/librosa/fastapi/...) this pure filesystem check +never touches. +""" + +import ast +import logging +from pathlib import Path +from typing import Optional + +_SOURCE = (Path(__file__).parent.parent / "backends" / "base.py").read_text() +_MODULE = ast.parse(_SOURCE) +_FUNC_SRC = next( + ast.get_source_segment(_SOURCE, node) + for node in _MODULE.body + if isinstance(node, ast.FunctionDef) and node.name == "is_model_cached" +) + +_namespace = { + "Path": Path, + "Optional": Optional, + "logger": logging.getLogger("test_is_model_cached"), +} +exec(_FUNC_SRC, _namespace) # noqa: S102 +is_model_cached = _namespace["is_model_cached"] + + +def _make_repo_cache(tmp_path, repo="org/model"): + repo_dir = tmp_path / ("models--" + repo.replace("/", "--")) + blobs_dir = repo_dir / "blobs" + snapshots_dir = repo_dir / "snapshots" / "abc123" + blobs_dir.mkdir(parents=True) + snapshots_dir.mkdir(parents=True) + return repo_dir, blobs_dir, snapshots_dir + + +def test_orphaned_incomplete_blob_does_not_block_cache_hit(tmp_path, monkeypatch): + import huggingface_hub.constants as hf_constants + + monkeypatch.setattr(hf_constants, "HF_HUB_CACHE", str(tmp_path)) + + repo = "org/model" + _, blobs_dir, snapshots_dir = _make_repo_cache(tmp_path, repo) + + completed_blob = blobs_dir / "deadbeef" + completed_blob.write_bytes(b"weights") + (blobs_dir / "deadbeef.incomplete").write_bytes(b"stale partial") + (snapshots_dir / "model.safetensors").symlink_to(completed_blob) + + assert is_model_cached(repo) is True + + +def test_genuinely_in_progress_download_is_not_cached(tmp_path, monkeypatch): + import huggingface_hub.constants as hf_constants + + monkeypatch.setattr(hf_constants, "HF_HUB_CACHE", str(tmp_path)) + + repo = "org/model" + _, blobs_dir, snapshots_dir = _make_repo_cache(tmp_path, repo) + + (blobs_dir / "feedface.incomplete").write_bytes(b"partial") + (snapshots_dir / "config.json").write_text("{}") + + assert is_model_cached(repo) is False From e86a2dcaa5b14fe7c52fbe033c63090c1d8ea978 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:29:44 +0000 Subject: [PATCH 25/35] fix(cache): share the orphaned-.incomplete check with /models/status --- backend/backends/base.py | 33 +++++++++++++++++---------- backend/routes/models.py | 6 ++--- backend/tests/test_is_model_cached.py | 4 ++-- 3 files changed, 26 insertions(+), 17 deletions(-) diff --git a/backend/backends/base.py b/backend/backends/base.py index 9360be79..a0e2cc78 100644 --- a/backend/backends/base.py +++ b/backend/backends/base.py @@ -22,6 +22,24 @@ from ..utils.tasks import get_task_manager logger = logging.getLogger(__name__) +def has_in_progress_download(blobs_dir: Path) -> bool: + """ + Whether a HuggingFace repo's ``blobs`` dir holds a genuinely in-progress download. + + An ``.incomplete`` blob means a download is still in progress -- unless a + completed blob with the same hash already sits next to it, which happens + when a retried/concurrent download leaves a stale ``.incomplete`` behind + after the real transfer already finished. Only orphaned ``.incomplete`` + files (no matching completed blob) count as "in progress". + """ + if not blobs_dir.exists(): + return False + return any( + not incomplete.with_name(incomplete.name.removesuffix(".incomplete")).exists() + for incomplete in blobs_dir.glob("*.incomplete") + ) + + def is_model_cached( hf_repo: str, *, @@ -48,18 +66,9 @@ def is_model_cached( if not repo_cache.exists(): return False - # An .incomplete blob means a download is still in progress -- unless - # a completed blob with the same hash already sits next to it, which - # happens when a retried/concurrent download leaves a stale .incomplete - # behind after the real transfer already finished. Only orphaned - # .incomplete files (no matching completed blob) count as "in progress". - blobs_dir = repo_cache / "blobs" - if blobs_dir.exists(): - for incomplete in blobs_dir.glob("*.incomplete"): - completed = incomplete.with_name(incomplete.name.removesuffix(".incomplete")) - if not completed.exists(): - logger.debug(f"Found in-progress .incomplete file for {hf_repo}") - return False + if has_in_progress_download(repo_cache / "blobs"): + logger.debug(f"Found in-progress .incomplete file for {hf_repo}") + return False snapshots_dir = repo_cache / "snapshots" if not snapshots_dir.exists(): diff --git a/backend/routes/models.py b/backend/routes/models.py index f6d56566..fcad0d35 100644 --- a/backend/routes/models.py +++ b/backend/routes/models.py @@ -244,6 +244,7 @@ async def get_model_status(): use_scan_cache = False from ..backends import get_all_model_configs, check_model_loaded + from ..backends.base import has_in_progress_download registry_configs = get_all_model_configs() model_configs = [ @@ -293,8 +294,7 @@ async def get_model_status(): try: cache_dir = hf_constants.HF_HUB_CACHE blobs_dir = Path(cache_dir) / ("models--" + repo_id.replace("/", "--")) / "blobs" - if blobs_dir.exists(): - has_incomplete = any(blobs_dir.glob("*.incomplete")) + has_incomplete = has_in_progress_download(blobs_dir) except Exception: pass @@ -314,7 +314,7 @@ async def get_model_status(): if repo_cache.exists(): blobs_dir = repo_cache / "blobs" - has_incomplete = blobs_dir.exists() and any(blobs_dir.glob("*.incomplete")) + has_incomplete = has_in_progress_download(blobs_dir) if not has_incomplete: snapshots_dir = repo_cache / "snapshots" diff --git a/backend/tests/test_is_model_cached.py b/backend/tests/test_is_model_cached.py index 6b9d883f..61c071d0 100644 --- a/backend/tests/test_is_model_cached.py +++ b/backend/tests/test_is_model_cached.py @@ -20,10 +20,10 @@ from typing import Optional _SOURCE = (Path(__file__).parent.parent / "backends" / "base.py").read_text() _MODULE = ast.parse(_SOURCE) -_FUNC_SRC = next( +_FUNC_SRC = "\n\n".join( ast.get_source_segment(_SOURCE, node) for node in _MODULE.body - if isinstance(node, ast.FunctionDef) and node.name == "is_model_cached" + if isinstance(node, ast.FunctionDef) and node.name in ("has_in_progress_download", "is_model_cached") ) _namespace = { From 36bb6c1ad7c1858a0ab8afdc5df07bf260b63360 Mon Sep 17 00:00:00 2001 From: Will Anderson Date: Thu, 14 May 2026 12:10:34 -0500 Subject: [PATCH 26/35] Enable SQLite WAL journal mode and 5 s busy timeout Switch from the default DELETE/ROLLBACK journal to WAL so concurrent readers (SSE status polls, history queries) are not blocked while the generation worker holds a write transaction. Set a 5-second busy timeout to eliminate "database is locked" errors under brief write contention. Both PRAGMAs are applied via a custom creator function so every connection in the pool gets the settings at open time, not just the first one. --- backend/database/session.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/backend/database/session.py b/backend/database/session.py index de4cccd9..f67864f1 100644 --- a/backend/database/session.py +++ b/backend/database/session.py @@ -1,6 +1,7 @@ """Engine creation, initialization, and session management.""" import logging +import sqlite3 as _sqlite3 import uuid from sqlalchemy import create_engine @@ -21,6 +22,22 @@ from .seed import backfill_generation_versions, seed_builtin_presets logger = logging.getLogger(__name__) + +def _make_connection(db_path: str) -> _sqlite3.Connection: + """Open a SQLite connection with WAL journal mode and a 5-second busy timeout. + + WAL allows concurrent readers while a write is in progress (the default + DELETE journal blocks all readers). This matters for voicebox because SSE + status polls and history queries run concurrently with the generation worker + writing to the same database. The busy timeout prevents "database is + locked" errors when two writers briefly contend on the same write slot. + """ + conn = _sqlite3.connect(db_path, check_same_thread=False) + conn.execute("PRAGMA journal_mode=WAL") + conn.execute("PRAGMA busy_timeout=5000") + return conn + + # Initialized by init_db() engine = None SessionLocal = None @@ -37,6 +54,14 @@ def init_db() -> None: engine = create_engine( f"sqlite:///{_db_path}", connect_args={"check_same_thread": False}, + # Each connection enables WAL journal mode and sets a 5-second busy + # timeout. WAL allows concurrent readers during a write (the default + # DELETE/ROLLBACK journal blocks all readers), which matters for + # voicebox because SSE status polls and history queries run + # concurrently with the generation worker writing to the same db. + # busy_timeout prevents "database is locked" errors when two + # connections briefly contend on the same write slot. + creator=lambda: _make_connection(str(_db_path)), ) SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine) From ccc092bad0344bdaf1520e15367fec265aede1f8 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:31:15 +0000 Subject: [PATCH 27/35] perf(db): set the SQLite pragmas from a connect hook and clear WAL sidecars on db reset --- backend/database/session.py | 36 +++++++------------ .../content/docs/overview/troubleshooting.mdx | 6 ++-- justfile | 4 +-- 3 files changed, 19 insertions(+), 27 deletions(-) diff --git a/backend/database/session.py b/backend/database/session.py index f67864f1..cb051ec2 100644 --- a/backend/database/session.py +++ b/backend/database/session.py @@ -1,10 +1,9 @@ """Engine creation, initialization, and session management.""" import logging -import sqlite3 as _sqlite3 import uuid -from sqlalchemy import create_engine +from sqlalchemy import create_engine, event from sqlalchemy.orm import sessionmaker from .. import config @@ -23,21 +22,6 @@ from .seed import backfill_generation_versions, seed_builtin_presets logger = logging.getLogger(__name__) -def _make_connection(db_path: str) -> _sqlite3.Connection: - """Open a SQLite connection with WAL journal mode and a 5-second busy timeout. - - WAL allows concurrent readers while a write is in progress (the default - DELETE journal blocks all readers). This matters for voicebox because SSE - status polls and history queries run concurrently with the generation worker - writing to the same database. The busy timeout prevents "database is - locked" errors when two writers briefly contend on the same write slot. - """ - conn = _sqlite3.connect(db_path, check_same_thread=False) - conn.execute("PRAGMA journal_mode=WAL") - conn.execute("PRAGMA busy_timeout=5000") - return conn - - # Initialized by init_db() engine = None SessionLocal = None @@ -54,15 +38,21 @@ def init_db() -> None: engine = create_engine( f"sqlite:///{_db_path}", connect_args={"check_same_thread": False}, - # Each connection enables WAL journal mode and sets a 5-second busy - # timeout. WAL allows concurrent readers during a write (the default - # DELETE/ROLLBACK journal blocks all readers), which matters for - # voicebox because SSE status polls and history queries run + ) + + @event.listens_for(engine, "connect") + def _set_sqlite_pragmas(dbapi_connection, _record) -> None: + # Each pooled connection enables WAL journal mode and sets a 5-second + # busy timeout. WAL allows concurrent readers during a write (the + # default DELETE/ROLLBACK journal blocks all readers), which matters + # for voicebox because SSE status polls and history queries run # concurrently with the generation worker writing to the same db. # busy_timeout prevents "database is locked" errors when two # connections briefly contend on the same write slot. - creator=lambda: _make_connection(str(_db_path)), - ) + cursor = dbapi_connection.cursor() + cursor.execute("PRAGMA journal_mode=WAL") + cursor.execute("PRAGMA busy_timeout=5000") + cursor.close() SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine) diff --git a/docs/content/docs/overview/troubleshooting.mdx b/docs/content/docs/overview/troubleshooting.mdx index f615b3c6..7668b873 100644 --- a/docs/content/docs/overview/troubleshooting.mdx +++ b/docs/content/docs/overview/troubleshooting.mdx @@ -419,12 +419,14 @@ bun run tauri build ```bash # macOS -rm ~/Library/Application\ Support/sh.voicebox.app/data/voicebox.db +rm ~/Library/Application\ Support/sh.voicebox.app/data/voicebox.db* # Windows -del %APPDATA%\sh.voicebox.app\data\voicebox.db +del %APPDATA%\sh.voicebox.app\data\voicebox.db* ``` +The wildcard also removes the `voicebox.db-wal` and `voicebox.db-shm` sidecar files that SQLite keeps next to the database in WAL mode, so the reset starts completely clean. + Restart the app to create a fresh database. ## Model Issues diff --git a/justfile b/justfile index 877f1519..9b9a480f 100644 --- a/justfile +++ b/justfile @@ -348,12 +348,12 @@ db-init: _ensure-venv # Reset database (delete + reinit) [unix] db-reset: - rm -f {{ backend_dir }}/data/voicebox.db + rm -f {{ backend_dir }}/data/voicebox.db {{ backend_dir }}/data/voicebox.db-wal {{ backend_dir }}/data/voicebox.db-shm just db-init [windows] db-reset: - if (Test-Path "{{ backend_dir }}/data/voicebox.db") { Remove-Item -Force "{{ backend_dir }}/data/voicebox.db" } + Remove-Item -Force -ErrorAction SilentlyContinue "{{ backend_dir }}/data/voicebox.db", "{{ backend_dir }}/data/voicebox.db-wal", "{{ backend_dir }}/data/voicebox.db-shm" just db-init # ─── Utilities ──────────────────────────────────────────────────────── From b97ffbfba970dd97c85bc93e941f6a861decf54a Mon Sep 17 00:00:00 2001 From: Will Anderson Date: Thu, 14 May 2026 12:08:07 -0500 Subject: [PATCH 28/35] Add indexes on high-traffic foreign keys and sort columns MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Queries throughout the codebase filter on generations.profile_id, generations.status, generations.created_at, story_items.story_id, story_items.generation_id, generation_versions.generation_id, and profile_samples.profile_id with every request. Without indexes SQLite falls back to a full table scan; as history grows (hundreds or thousands of generations) these scans become the dominant latency. Changes: - Add index=True on the most-queried FK and sort columns in models.py so new installs get them from Base.metadata.create_all - Add _migrate_add_indexes() called from run_migrations() so existing installs get the same indexes on next startup (uses CREATE INDEX IF NOT EXISTS — idempotent, <10 ms on any realistic dataset) --- backend/database/migrations.py | 41 ++++++++++++++++++++++++++++++++++ backend/database/models.py | 20 ++++++++--------- 2 files changed, 51 insertions(+), 10 deletions(-) diff --git a/backend/database/migrations.py b/backend/database/migrations.py index d353b58c..12f9cdd4 100644 --- a/backend/database/migrations.py +++ b/backend/database/migrations.py @@ -44,6 +44,7 @@ def run_migrations(engine) -> None: _migrate_capture_settings(engine, inspector, tables) _migrate_mcp_bindings(engine, inspector, tables) _normalize_storage_paths(engine, tables) + _migrate_add_indexes(engine, tables) # -- helpers --------------------------------------------------------------- @@ -292,6 +293,46 @@ def _supports_drop_column(engine) -> bool: return tuple(int(p) for p in sqlite3.sqlite_version.split(".")[:3]) >= (3, 35, 0) +def _migrate_add_indexes(engine, tables: set[str]) -> None: + """Create missing indexes on high-traffic foreign keys and sort columns. + + SQLite silently ignores ``CREATE INDEX IF NOT EXISTS``, so this is + safe to run on every startup regardless of whether the index already + exists. New installs get the indexes from ``Base.metadata.create_all`` + (via the ``index=True`` column flags); this migration brings existing + databases into parity without dropping or recreating any data. + """ + indexes = [ + # generations — filtered by profile, ordered/filtered by date, filtered by status + ("ix_generations_profile_id", "generations", "profile_id"), + ("ix_generations_created_at", "generations", "created_at"), + ("ix_generations_status", "generations", "status"), + # story_items — every story lookup filters by story_id; join on generation_id + ("ix_story_items_story_id", "story_items", "story_id"), + ("ix_story_items_generation_id", "story_items", "generation_id"), + # generation_versions — always filtered/joined on generation_id + ("ix_generation_versions_generation_id", "generation_versions", "generation_id"), + # profile_samples — loaded per-profile on every voice prompt build + ("ix_profile_samples_profile_id", "profile_samples", "profile_id"), + # captures — ordered by date in list view + ("ix_captures_created_at", "captures", "created_at"), + # channel_device_mappings — looked up per channel + ("ix_channel_device_mappings_channel_id", "channel_device_mappings", "channel_id"), + ] + + with engine.connect() as conn: + for index_name, table, column in indexes: + if table not in tables: + continue + conn.execute( + text( + f"CREATE INDEX IF NOT EXISTS {index_name}" + f" ON {table} ({column})" + ) + ) + conn.commit() + + def _normalize_storage_paths(engine, tables: set[str]) -> None: """Normalize stored file paths to be relative to the configured data dir.""" from pathlib import Path diff --git a/backend/database/models.py b/backend/database/models.py index b85a55b1..3ba75fc8 100644 --- a/backend/database/models.py +++ b/backend/database/models.py @@ -3,7 +3,7 @@ from datetime import datetime import uuid -from sqlalchemy import Column, String, Integer, Float, DateTime, Text, ForeignKey, Boolean, JSON +from sqlalchemy import Column, Index, String, Integer, Float, DateTime, Text, ForeignKey, Boolean, JSON from sqlalchemy.ext.declarative import declarative_base from ..utils.capture_chords import ( @@ -54,7 +54,7 @@ class ProfileSample(Base): __tablename__ = "profile_samples" id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4())) - profile_id = Column(String, ForeignKey("profiles.id"), nullable=False) + profile_id = Column(String, ForeignKey("profiles.id"), nullable=False, index=True) audio_path = Column(String, nullable=False) reference_text = Column(Text, nullable=False) @@ -65,7 +65,7 @@ class Generation(Base): __tablename__ = "generations" id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4())) - profile_id = Column(String, ForeignKey("profiles.id"), nullable=False) + profile_id = Column(String, ForeignKey("profiles.id"), nullable=False, index=True) text = Column(Text, nullable=False) language = Column(String, default="en") audio_path = Column(String, nullable=True) @@ -74,7 +74,7 @@ class Generation(Base): instruct = Column(Text) engine = Column(String, default="qwen") model_size = Column(String, nullable=True) - status = Column(String, default="completed") + status = Column(String, default="completed", index=True) error = Column(Text, nullable=True) is_favorited = Column(Boolean, default=False) # Origin of this generation — "manual" for plain /generate calls, @@ -82,7 +82,7 @@ class Generation(Base): # profile's personality LLM before TTS. Future sources (bulk import, # agent replies, etc.) can extend this. source = Column(String, nullable=False, default="manual") - created_at = Column(DateTime, default=datetime.utcnow) + created_at = Column(DateTime, default=datetime.utcnow, index=True) class Story(Base): @@ -103,8 +103,8 @@ class StoryItem(Base): __tablename__ = "story_items" id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4())) - story_id = Column(String, ForeignKey("stories.id"), nullable=False) - generation_id = Column(String, ForeignKey("generations.id"), nullable=False) + story_id = Column(String, ForeignKey("stories.id"), nullable=False, index=True) + generation_id = Column(String, ForeignKey("generations.id"), nullable=False, index=True) version_id = Column(String, ForeignKey("generation_versions.id"), nullable=True) start_time_ms = Column(Integer, nullable=False, default=0) track = Column(Integer, nullable=False, default=0) @@ -132,7 +132,7 @@ class GenerationVersion(Base): __tablename__ = "generation_versions" id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4())) - generation_id = Column(String, ForeignKey("generations.id"), nullable=False) + generation_id = Column(String, ForeignKey("generations.id"), nullable=False, index=True) label = Column(String, nullable=False) audio_path = Column(String, nullable=False) effects_chain = Column(Text, nullable=True) @@ -172,7 +172,7 @@ class ChannelDeviceMapping(Base): __tablename__ = "channel_device_mappings" id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4())) - channel_id = Column(String, ForeignKey("audio_channels.id"), nullable=False) + channel_id = Column(String, ForeignKey("audio_channels.id"), nullable=False, index=True) device_id = Column(String, nullable=False) @@ -300,4 +300,4 @@ class Capture(Base): stt_model = Column(String, nullable=True) llm_model = Column(String, nullable=True) refinement_flags = Column(Text, nullable=True) # JSON blob - created_at = Column(DateTime, default=datetime.utcnow) + created_at = Column(DateTime, default=datetime.utcnow, index=True) From 22a3f6b3dd4763f9ef84029ca43aeae615e0a937 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:24:09 +0000 Subject: [PATCH 29/35] fix(db): drop unused Index import --- backend/database/models.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/database/models.py b/backend/database/models.py index 3ba75fc8..a97074d2 100644 --- a/backend/database/models.py +++ b/backend/database/models.py @@ -3,7 +3,7 @@ from datetime import datetime import uuid -from sqlalchemy import Column, Index, String, Integer, Float, DateTime, Text, ForeignKey, Boolean, JSON +from sqlalchemy import Column, String, Integer, Float, DateTime, Text, ForeignKey, Boolean, JSON from sqlalchemy.ext.declarative import declarative_base from ..utils.capture_chords import ( From dd1ada8239c9e00b916491a4ae25e190b8f740c5 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:31:57 +0000 Subject: [PATCH 30/35] fix(db): log which indexes the migration actually created --- backend/database/migrations.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/backend/database/migrations.py b/backend/database/migrations.py index 12f9cdd4..d9011a56 100644 --- a/backend/database/migrations.py +++ b/backend/database/migrations.py @@ -321,8 +321,13 @@ def _migrate_add_indexes(engine, tables: set[str]) -> None: ] with engine.connect() as conn: + existing = { + row[0] + for row in conn.execute(text("SELECT name FROM sqlite_master WHERE type = 'index'")) + } + created = [] for index_name, table, column in indexes: - if table not in tables: + if table not in tables or index_name in existing: continue conn.execute( text( @@ -330,8 +335,12 @@ def _migrate_add_indexes(engine, tables: set[str]) -> None: f" ON {table} ({column})" ) ) + created.append(index_name) conn.commit() + if created: + logger.info("Created %d missing index(es): %s", len(created), ", ".join(created)) + def _normalize_storage_paths(engine, tables: set[str]) -> None: """Normalize stored file paths to be relative to the configured data dir.""" From 84900d4e86b7a7e7c7e772c5932b51590b96b3f1 Mon Sep 17 00:00:00 2001 From: anupamme Date: Tue, 4 Aug 2026 05:17:28 +0000 Subject: [PATCH 31/35] fix: CVE-2026-59940 security vulnerability Automated dependency upgrade by OrbisAI Security --- bun.lock | 5 ++++- package.json | 3 ++- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/bun.lock b/bun.lock index 74d751ec..f3d95c7b 100644 --- a/bun.lock +++ b/bun.lock @@ -7,6 +7,7 @@ "dependencies": { "loaders.css": "^0.1.2", "react-loaders": "^3.0.1", + "seroval": "1.5.3", }, "devDependencies": { "@biomejs/biome": "2.3.12", @@ -1051,7 +1052,7 @@ "semver": ["semver@6.3.1", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-BR7VvDCVHO+q2xBEWskxS6DJE1qRnb7DxzUrogb71CWoSficBxYsiAGd+Kl0mmq/MprG9yArRkyrQxTO6XjMzA=="], - "seroval": ["seroval@1.5.0", "", {}, "sha512-OE4cvmJ1uSPrKorFIH9/w/Qwuvi/IMcGbv5RKgcJ/zjA/IohDLU6SVaxFN9FwajbP7nsX0dQqMDes1whk3y+yw=="], + "seroval": ["seroval@1.5.3", "", {}, "sha512-BXe0x4buEeYiIKaRUnth1WqCILQ3k4O67KP/B4pC3pVz0Mv2c96ngA9QDREUYxWY1sb2RZVRqwI9RcpVMyHCVw=="], "seroval-plugins": ["seroval-plugins@1.5.0", "", { "peerDependencies": { "seroval": "^1.0" } }, "sha512-EAHqADIQondwRZIdeW2I636zgsODzoBDwb3PT/+7TLDWyw1Dy/Xv7iGUIEXXav7usHDE9HVhOU61irI3EnyyHA=="], @@ -1193,6 +1194,8 @@ "@tailwindcss/oxide-wasm32-wasi/tslib": ["tslib@2.8.1", "", { "bundled": true }, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="], + "@tanstack/router-core/seroval": ["seroval@1.5.0", "", {}, "sha512-OE4cvmJ1uSPrKorFIH9/w/Qwuvi/IMcGbv5RKgcJ/zjA/IohDLU6SVaxFN9FwajbP7nsX0dQqMDes1whk3y+yw=="], + "@typescript-eslint/typescript-estree/minimatch": ["minimatch@9.0.5", "", { "dependencies": { "brace-expansion": "^2.0.1" } }, "sha512-G6T0ZX48xgozx7587koeX9Ys2NYy6Gmv//P89sEte9V9whIapMNF4idKxnW2QtCcLiTWlb/wfCabAtAFWhhBow=="], "@typescript-eslint/typescript-estree/semver": ["semver@7.7.3", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-SdsKMrI9TdgjdweUSR9MweHA4EJ8YxHn8DFaDisvhVlUOe4BF1tLD7GAj0lIqWVl+dPb/rExr0Btby5loQm20Q=="], diff --git a/package.json b/package.json index ba9274f2..f79e340d 100644 --- a/package.json +++ b/package.json @@ -44,6 +44,7 @@ "packageManager": "bun@1.3.8", "dependencies": { "loaders.css": "^0.1.2", - "react-loaders": "^3.0.1" + "react-loaders": "^3.0.1", + "seroval": "1.5.3" } } From fc0bec8faf601fd0ac4f8f3c3c3dd0ce921f4071 Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:21:10 +0000 Subject: [PATCH 32/35] fix: force seroval 1.5.3 via bun overrides so @tanstack/router-core picks it up Adding seroval as a direct root dependency left the lockfile's @tanstack/router-core/seroval entry pinned at the vulnerable 1.5.0, so the only real consumer was still bundling the CVE-2026-59940 version. A bun 'overrides' entry forces every resolution to 1.5.3 and drops the unused direct dependency. --- bun.lock | 8 +++----- package.json | 4 +++- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/bun.lock b/bun.lock index f3d95c7b..f43f622e 100644 --- a/bun.lock +++ b/bun.lock @@ -7,7 +7,6 @@ "dependencies": { "loaders.css": "^0.1.2", "react-loaders": "^3.0.1", - "seroval": "1.5.3", }, "devDependencies": { "@biomejs/biome": "2.3.12", @@ -153,6 +152,9 @@ }, }, }, + "overrides": { + "seroval": "1.5.3", + }, "packages": { "@alloc/quick-lru": ["@alloc/quick-lru@5.2.0", "", {}, "sha512-UrcABB+4bUrFABwbluTIBErXwvbsU/V7TZWfmbgJfbkwiBuziS9gxdODUyuiecfdGQ85jglMW6juS3+z5TsKLw=="], @@ -1194,8 +1196,6 @@ "@tailwindcss/oxide-wasm32-wasi/tslib": ["tslib@2.8.1", "", { "bundled": true }, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="], - "@tanstack/router-core/seroval": ["seroval@1.5.0", "", {}, "sha512-OE4cvmJ1uSPrKorFIH9/w/Qwuvi/IMcGbv5RKgcJ/zjA/IohDLU6SVaxFN9FwajbP7nsX0dQqMDes1whk3y+yw=="], - "@typescript-eslint/typescript-estree/minimatch": ["minimatch@9.0.5", "", { "dependencies": { "brace-expansion": "^2.0.1" } }, "sha512-G6T0ZX48xgozx7587koeX9Ys2NYy6Gmv//P89sEte9V9whIapMNF4idKxnW2QtCcLiTWlb/wfCabAtAFWhhBow=="], "@typescript-eslint/typescript-estree/semver": ["semver@7.7.3", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-SdsKMrI9TdgjdweUSR9MweHA4EJ8YxHn8DFaDisvhVlUOe4BF1tLD7GAj0lIqWVl+dPb/rExr0Btby5loQm20Q=="], @@ -1206,8 +1206,6 @@ "@voicebox/landing/tailwindcss": ["tailwindcss@3.4.19", "", { "dependencies": { "@alloc/quick-lru": "^5.2.0", "arg": "^5.0.2", "chokidar": "^3.6.0", "didyoumean": "^1.2.2", "dlv": "^1.1.3", "fast-glob": "^3.3.2", "glob-parent": "^6.0.2", "is-glob": "^4.0.3", "jiti": "^1.21.7", "lilconfig": "^3.1.3", "micromatch": "^4.0.8", "normalize-path": "^3.0.0", "object-hash": "^3.0.0", "picocolors": "^1.1.1", "postcss": "^8.4.47", "postcss-import": "^15.1.0", "postcss-js": "^4.0.1", "postcss-load-config": "^4.0.2 || ^5.0 || ^6.0", "postcss-nested": "^6.2.0", "postcss-selector-parser": "^6.1.2", "resolve": "^1.22.8", "sucrase": "^3.35.0" }, "bin": { "tailwind": "lib/cli.js", "tailwindcss": "lib/cli.js" } }, "sha512-3ofp+LL8E+pK/JuPLPggVAIaEuhvIz4qNcf3nA1Xn2o/7fb7s/TYpHhwGDv1ZU3PkBluUVaF8PyCHcm48cKLWQ=="], - "@voicebox/web/wavesurfer.js": ["wavesurfer.js@7.12.1", "", {}, "sha512-NswPjVHxk0Q1F/VMRemCPUzSojjuHHisQrBqQiRXg7MVbe3f5vQ6r0rTTXA/a/neC/4hnOEC4YpXca4LpH0SUg=="], - "chokidar/glob-parent": ["glob-parent@5.1.2", "", { "dependencies": { "is-glob": "^4.0.1" } }, "sha512-AOIgSQCepiJYwP3ARnGx+5VnTu2HBYdzbGP45eLw1vr3zB3vZLeyed1sC9hnbcOc9/SrMyM5RPQrkGz4aS9Zow=="], "eslint/js-yaml": ["js-yaml@4.1.1", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": { "js-yaml": "bin/js-yaml.js" } }, "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA=="], diff --git a/package.json b/package.json index f79e340d..1dd38f1b 100644 --- a/package.json +++ b/package.json @@ -44,7 +44,9 @@ "packageManager": "bun@1.3.8", "dependencies": { "loaders.css": "^0.1.2", - "react-loaders": "^3.0.1", + "react-loaders": "^3.0.1" + }, + "overrides": { "seroval": "1.5.3" } } From 89901040814ceecf40a7a34b8fc272079a37d185 Mon Sep 17 00:00:00 2001 From: Jashwanth Date: Sun, 28 Jun 2026 10:34:50 +0530 Subject: [PATCH 33/35] fix: use underscore-separated MCP tool names so Claude Desktop accepts them Claude Desktop validates tool names against ^[a-zA-Z0-9_-]{1,64}$ and rejects the whole tool list when any name contains a dot, so the Voicebox MCP server was unusable there. Rename voicebox.speak, voicebox.transcribe, voicebox.list_captures and voicebox.list_profiles to voicebox_speak, voicebox_transcribe, voicebox_list_captures and voicebox_list_profiles. Fixes #790 --- backend/mcp_server/tools.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/backend/mcp_server/tools.py b/backend/mcp_server/tools.py index 61003f11..fae79e48 100644 --- a/backend/mcp_server/tools.py +++ b/backend/mcp_server/tools.py @@ -36,7 +36,7 @@ def register_tools(mcp: FastMCP) -> None: """Attach all Voicebox tools to the given FastMCP instance.""" @mcp.tool( - name="voicebox.speak", + name="voicebox_speak", description=( "Speak text in a Voicebox voice profile. Returns a generation id " "the caller can poll at /generate/{id}/status. Audio plays on the " @@ -113,7 +113,7 @@ def register_tools(mcp: FastMCP) -> None: db.close() @mcp.tool( - name="voicebox.transcribe", + name="voicebox_transcribe", description=( "Transcribe an audio clip to text using Voicebox's local Whisper. " "Pass exactly one of `audio_base64` (bytes as base64) or " @@ -171,7 +171,7 @@ def register_tools(mcp: FastMCP) -> None: tmp_path.unlink(missing_ok=True) @mcp.tool( - name="voicebox.list_captures", + name="voicebox_list_captures", description=( "List recent voice captures (dictations, recordings, uploads) " "with their transcripts. Most-recent first." @@ -199,7 +199,7 @@ def register_tools(mcp: FastMCP) -> None: db.close() @mcp.tool( - name="voicebox.list_profiles", + name="voicebox_list_profiles", description=( "List available voice profiles (both cloned voices and presets). " "Use the returned `name` with voicebox.speak(profile=...)." From 37af18088720fc97fbdf4bc281eee3b6f8a59ffc Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:23:10 +0000 Subject: [PATCH 34/35] docs: refer to the MCP tools by their new underscore names Update the README, MCP docs, in-app MCP page, i18n strings, landing copy, docstrings and comments to match the renamed tools. CHANGELOG entries and docs/plans are historical and left as they were. --- README.md | 10 +++---- app/src/components/ServerTab/MCPPage.tsx | 8 ++--- app/src/i18n/locales/en/translation.json | 2 +- app/src/i18n/locales/es/translation.json | 2 +- app/src/i18n/locales/fr/translation.json | 2 +- app/src/i18n/locales/it/translation.json | 2 +- app/src/i18n/locales/ja/translation.json | 2 +- app/src/i18n/locales/ko/translation.json | 2 +- app/src/i18n/locales/pt-BR/translation.json | 2 +- app/src/i18n/locales/zh-CN/translation.json | 2 +- app/src/i18n/locales/zh-TW/translation.json | 2 +- backend/database/migrations.py | 2 +- backend/database/models.py | 2 +- backend/mcp_server/README.md | 12 ++++---- backend/mcp_server/context.py | 6 ++-- backend/mcp_server/events.py | 2 +- backend/mcp_server/server.py | 4 +-- backend/mcp_server/tools.py | 9 +++--- backend/models.py | 4 +-- backend/routes/speak.py | 4 +-- backend/tests/test_mcp_speak.py | 2 +- docs/PROJECT_STATUS.md | 2 +- docs/content/docs/overview/mcp-server.mdx | 30 +++++++++---------- .../docs/overview/voice-personalities.mdx | 4 +-- landing/src/components/AgentIntegration.tsx | 10 +++---- landing/src/components/CapturesMockup.tsx | 2 +- 26 files changed, 66 insertions(+), 65 deletions(-) diff --git a/README.md b/README.md index bda86132..1a7f3ae3 100644 --- a/README.md +++ b/README.md @@ -80,7 +80,7 @@ The two cloud incumbents sit on opposite halves of the voice I/O loop — Eleven - **Unlimited length** — auto-chunking with crossfade for scripts, articles, and chapters - **Stories editor** — multi-track timeline for conversations, podcasts, and narratives - **Voice input** — global dictation hotkey with push-to-talk and toggle modes, accessibility-verified auto-paste on macOS, in-app mic on every text field, Whisper-based STT -- **Agent voice output** — one tool call (`voicebox.speak`) and any MCP-aware agent (Claude Code, Cursor, Cline) speaks to you in a voice you've cloned +- **Agent voice output** — one tool call (`voicebox_speak`) and any MCP-aware agent (Claude Code, Cursor, Cline) speaks to you in a voice you've cloned - **Voice personalities** — attach a free-form persona to any voice profile, then Compose, Rewrite, or Respond via a bundled local LLM — agents can invoke the same modes over MCP - **API-first** — REST API plus a built-in MCP server for integrating voice I/O into your own apps and agents - **Native performance** — built with Tauri (Rust), not Electron @@ -234,7 +234,7 @@ Every agent gets a voice. One tool call and any MCP-aware agent can speak to you ```ts // In any MCP-aware agent: -await voicebox.speak({ +await voicebox_speak({ text: "Deploy complete.", profile: "Morgan", }); @@ -254,7 +254,7 @@ Attach a free-form personality to any voice profile — who this voice is, how t - **Compose** — a shuffle button that drops a fresh in-character line into the textarea; edit and speak, or click again for a different take - **Speak in character** — a toggle that routes your input text through the personality LLM to be rewritten in their voice before TTS -Agents can reach the same rewrite path over MCP by passing `personality: true` to `voicebox.speak`, turning the tool into a text-in → personality-LLM → TTS pipeline. The same LLM backs dictation's refinement step — one LLM in the app, one model cache, one GPU-memory footprint. +Agents can reach the same rewrite path over MCP by passing `personality: true` to `voicebox_speak`, turning the tool into a text-in → personality-LLM → TTS pipeline. The same LLM backs dictation's refinement step — one LLM in the app, one model cache, one GPU-memory footprint. **Local LLM options:** Qwen3 0.6B / 1.7B / 4B, sharing the TTS runtime (MLX on Apple Silicon, PyTorch elsewhere). @@ -347,11 +347,11 @@ claude mcp add voicebox \ } ``` -Four tools ship: `voicebox.speak`, `voicebox.transcribe`, `voicebox.list_captures`, `voicebox.list_profiles`. Per-client voice bindings are managed in **Voicebox → Settings → MCP**. See the [full MCP guide](docs/content/docs/overview/mcp-server.mdx) for tool signatures, resolution precedence, the speaking-pill contract, and security notes. +Four tools ship: `voicebox_speak`, `voicebox_transcribe`, `voicebox_list_captures`, `voicebox_list_profiles`. Per-client voice bindings are managed in **Voicebox → Settings → MCP**. See the [full MCP guide](docs/content/docs/overview/mcp-server.mdx) for tool signatures, resolution precedence, the speaking-pill contract, and security notes. ```ts // In any MCP-aware agent: -await voicebox.speak({ +await voicebox_speak({ text: "Tests passing. Ready to merge.", profile: "Morgan", // optional — falls back to the per-client binding personality: true, // optional — rewrites text through the profile's personality LLM first diff --git a/app/src/components/ServerTab/MCPPage.tsx b/app/src/components/ServerTab/MCPPage.tsx index ddb4de1f..3dc00dbf 100644 --- a/app/src/components/ServerTab/MCPPage.tsx +++ b/app/src/components/ServerTab/MCPPage.tsx @@ -276,19 +276,19 @@ export function MCPPage() {

{t('settings.mcp.sidebar.toolsTitle')}

  • - voicebox.speak + voicebox_speak
    {t('settings.mcp.sidebar.tools.speak')}
  • - voicebox.transcribe + voicebox_transcribe
    {t('settings.mcp.sidebar.tools.transcribe')}
  • - voicebox.list_captures + voicebox_list_captures
    {t('settings.mcp.sidebar.tools.listCaptures')}
  • - voicebox.list_profiles + voicebox_list_profiles
    {t('settings.mcp.sidebar.tools.listProfiles')}
diff --git a/app/src/i18n/locales/en/translation.json b/app/src/i18n/locales/en/translation.json index 7f96d9d0..6f60d180 100644 --- a/app/src/i18n/locales/en/translation.json +++ b/app/src/i18n/locales/en/translation.json @@ -1055,7 +1055,7 @@ }, "defaultVoice": { "title": "Default voice", - "description": "Used when an agent calls voicebox.speak without a specific profile and has no per-client binding.", + "description": "Used when an agent calls voicebox_speak without a specific profile and has no per-client binding.", "label": "Default playback voice", "labelHint": "Shared with the Captures-tab 'Play as voice' dropdown — one default voice for passive playback.", "none": "(none)" diff --git a/app/src/i18n/locales/es/translation.json b/app/src/i18n/locales/es/translation.json index f1cbfd19..eb77e209 100644 --- a/app/src/i18n/locales/es/translation.json +++ b/app/src/i18n/locales/es/translation.json @@ -1055,7 +1055,7 @@ }, "defaultVoice": { "title": "Voz predeterminada", - "description": "Se usa cuando un agente llama a voicebox.speak sin un perfil específico y no tiene una vinculación por cliente.", + "description": "Se usa cuando un agente llama a voicebox_speak sin un perfil específico y no tiene una vinculación por cliente.", "label": "Voz de reproducción predeterminada", "labelHint": "Compartida con el desplegable 'Reproducir como voz' de la pestaña Capturas: una voz predeterminada para la reproducción pasiva.", "none": "(ninguna)" diff --git a/app/src/i18n/locales/fr/translation.json b/app/src/i18n/locales/fr/translation.json index 8ce2aae6..9ada81b5 100644 --- a/app/src/i18n/locales/fr/translation.json +++ b/app/src/i18n/locales/fr/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "Voix par défaut", - "description": "Utilisée quand un agent appelle voicebox.speak sans profil spécifique et sans liaison par client.", + "description": "Utilisée quand un agent appelle voicebox_speak sans profil spécifique et sans liaison par client.", "label": "Voix de lecture par défaut", "labelHint": "Partagé avec la liste déroulante « Lire avec » de l'onglet Captures — une voix par défaut pour la lecture passive.", "none": "(aucune)" diff --git a/app/src/i18n/locales/it/translation.json b/app/src/i18n/locales/it/translation.json index 5c21cb9d..c34418cc 100644 --- a/app/src/i18n/locales/it/translation.json +++ b/app/src/i18n/locales/it/translation.json @@ -1055,7 +1055,7 @@ }, "defaultVoice": { "title": "Voce predefinita", - "description": "Utilizzata quando un agente chiama voicebox.speak senza specificare un profilo e non ha un'associazione per singolo client.", + "description": "Utilizzata quando un agente chiama voicebox_speak senza specificare un profilo e non ha un'associazione per singolo client.", "label": "Voce di riproduzione predefinita", "labelHint": "Condivisa con il menu a discesa 'Riproduci come voce' della scheda Acquisizioni — una sola voce predefinita per la riproduzione passiva.", "none": "(nessuna)" diff --git a/app/src/i18n/locales/ja/translation.json b/app/src/i18n/locales/ja/translation.json index a7f0fd24..7e21775d 100644 --- a/app/src/i18n/locales/ja/translation.json +++ b/app/src/i18n/locales/ja/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "デフォルトボイス", - "description": "エージェントが特定のプロファイルを指定せず、クライアントごとのバインディングもない状態で voicebox.speak を呼び出したときに使われます。", + "description": "エージェントが特定のプロファイルを指定せず、クライアントごとのバインディングもない状態で voicebox_speak を呼び出したときに使われます。", "label": "デフォルトの再生ボイス", "labelHint": "キャプチャタブの「ボイスで再生」ドロップダウンと共有 — パッシブ再生用に 1 つのデフォルトボイスを設定します。", "none": "(なし)" diff --git a/app/src/i18n/locales/ko/translation.json b/app/src/i18n/locales/ko/translation.json index 94aa26fb..96698c2c 100644 --- a/app/src/i18n/locales/ko/translation.json +++ b/app/src/i18n/locales/ko/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "기본 음성", - "description": "에이전트가 특정 프로필 없이 voicebox.speak를 호출하고 클라이언트별 바인딩도 없을 때 사용됩니다.", + "description": "에이전트가 특정 프로필 없이 voicebox_speak를 호출하고 클라이언트별 바인딩도 없을 때 사용됩니다.", "label": "기본 재생 음성", "labelHint": "캡처 탭의 'Play as 음성' 드롭다운과 공유 — 수동 재생용 기본 음성입니다.", "none": "(없음)" diff --git a/app/src/i18n/locales/pt-BR/translation.json b/app/src/i18n/locales/pt-BR/translation.json index d2885d8e..401e8eea 100644 --- a/app/src/i18n/locales/pt-BR/translation.json +++ b/app/src/i18n/locales/pt-BR/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "Voz padrão", - "description": "Usada quando um agente chama voicebox.speak sem um perfil específico e não tem vínculo por cliente.", + "description": "Usada quando um agente chama voicebox_speak sem um perfil específico e não tem vínculo por cliente.", "label": "Voz de reprodução padrão", "labelHint": "Compartilhada com o menu 'Reproduzir como voz' da aba Capturas — uma voz padrão para reprodução passiva.", "none": "(nenhuma)" diff --git a/app/src/i18n/locales/zh-CN/translation.json b/app/src/i18n/locales/zh-CN/translation.json index b0c78241..3eb41e87 100644 --- a/app/src/i18n/locales/zh-CN/translation.json +++ b/app/src/i18n/locales/zh-CN/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "默认声音", - "description": "当代理调用 voicebox.speak 但未指定具体档案、且没有按客户端绑定时使用。", + "description": "当代理调用 voicebox_speak 但未指定具体档案、且没有按客户端绑定时使用。", "label": "默认播放声音", "labelHint": "与「捕获」标签页的「播放为」下拉菜单共享——被动播放的统一默认声音。", "none": "(无)" diff --git a/app/src/i18n/locales/zh-TW/translation.json b/app/src/i18n/locales/zh-TW/translation.json index e946fc95..18a1efab 100644 --- a/app/src/i18n/locales/zh-TW/translation.json +++ b/app/src/i18n/locales/zh-TW/translation.json @@ -1050,7 +1050,7 @@ }, "defaultVoice": { "title": "預設聲音", - "description": "當代理呼叫 voicebox.speak 卻未指定聲音檔案,且沒有對應客戶端綁定時使用。", + "description": "當代理呼叫 voicebox_speak 卻未指定聲音檔案,且沒有對應客戶端綁定時使用。", "label": "預設播放聲音", "labelHint": "與「擷取」分頁的「以聲音播放」下拉選單共用——一個用於被動播放的預設聲音。", "none": "(無)" diff --git a/backend/database/migrations.py b/backend/database/migrations.py index d9011a56..a4de4379 100644 --- a/backend/database/migrations.py +++ b/backend/database/migrations.py @@ -250,7 +250,7 @@ def _migrate_mcp_bindings(engine, inspector, tables: set[str]) -> None: """Drop the legacy ``default_intent`` column and add ``default_personality``. The intent tri-state (respond / rewrite / compose) has been collapsed - to a boolean: when true, ``voicebox.speak`` rewrites input through the + to a boolean: when true, ``voicebox_speak`` rewrites input through the profile's personality LLM before TTS. """ if "mcp_client_bindings" not in tables: diff --git a/backend/database/models.py b/backend/database/models.py index a97074d2..cbb1cdaf 100644 --- a/backend/database/models.py +++ b/backend/database/models.py @@ -272,7 +272,7 @@ class MCPClientBinding(Base): label = Column(String, nullable=True) # display name profile_id = Column(String, ForeignKey("profiles.id"), nullable=True) default_engine = Column(String, nullable=True) - # When true, voicebox.speak routes through the profile's personality LLM + # When true, voicebox_speak routes through the profile's personality LLM # (rewrite) before TTS by default. Callers can still override per call. default_personality = Column(Boolean, nullable=False, default=False) last_seen_at = Column(DateTime, nullable=True) diff --git a/backend/mcp_server/README.md b/backend/mcp_server/README.md index 4c9b426f..527ec143 100644 --- a/backend/mcp_server/README.md +++ b/backend/mcp_server/README.md @@ -49,10 +49,10 @@ claude mcp add voicebox \ | Name | Purpose | |---|---| -| `voicebox.speak` | Speak text in a voice profile. Returns a generation id you can poll. | -| `voicebox.transcribe` | Whisper transcription of a base64 blob or an absolute local path. | -| `voicebox.list_captures` | Recent captures (dictation / recording / file) with transcripts. | -| `voicebox.list_profiles` | Available voice profiles (cloned + preset). | +| `voicebox_speak` | Speak text in a voice profile. Returns a generation id you can poll. | +| `voicebox_transcribe` | Whisper transcription of a base64 blob or an absolute local path. | +| `voicebox_list_captures` | Recent captures (dictation / recording / file) with transcripts. | +| `voicebox_list_profiles` | Available voice profiles (cloned + preset). | All tools resolve voice profiles in this precedence: @@ -69,8 +69,8 @@ Settings → MCP. npx @modelcontextprotocol/inspector http://127.0.0.1:17493/mcp ``` -Point it at the URL, hit "List tools," call `voicebox.list_profiles` -first to confirm wiring, then `voicebox.speak` for end-to-end. +Point it at the URL, hit "List tools," call `voicebox_list_profiles` +first to confirm wiring, then `voicebox_speak` for end-to-end. ## Non-MCP REST surface diff --git a/backend/mcp_server/context.py b/backend/mcp_server/context.py index ecf1801a..0577d5e9 100644 --- a/backend/mcp_server/context.py +++ b/backend/mcp_server/context.py @@ -33,7 +33,7 @@ current_client_id: ContextVar[str | None] = ContextVar( ) # Remote address of the in-flight request. Used by tools that gate -# host-filesystem access to loopback callers (see voicebox.transcribe). +# host-filesystem access to loopback callers (see voicebox_transcribe). current_remote_addr: ContextVar[str | None] = ContextVar( "current_remote_addr", default=None ) @@ -61,12 +61,12 @@ def request_is_loopback() -> bool: # ignored so the Settings UI's "last heard from" column only reflects # calls that actually acted on the client's bindings. # -# - /mcp — FastMCP tool calls (voicebox.speak, voicebox.transcribe, …) +# - /mcp — FastMCP tool calls (voicebox_speak, voicebox_transcribe, …) # and the /mcp/bindings admin surface. The admin surface is never # called with the header in practice (the frontend manages bindings # over plain REST), so the `startswith("/mcp")` match doesn't cause # false stamps. -# - /speak — REST mirror of voicebox.speak for non-MCP agents (shell +# - /speak — REST mirror of voicebox_speak for non-MCP agents (shell # scripts, ACP, A2A). Uses the same per-client binding lookup, so its # callers belong in the last-seen list too. _STAMPED_PATH_PREFIXES: tuple[str, ...] = ("/mcp", "/speak") diff --git a/backend/mcp_server/events.py b/backend/mcp_server/events.py index 7d7afc71..10817c47 100644 --- a/backend/mcp_server/events.py +++ b/backend/mcp_server/events.py @@ -1,6 +1,6 @@ """In-memory pub/sub for speaking-pill SSE broadcasts. -MCP ``voicebox.speak`` calls and the REST ``POST /speak`` route publish +MCP ``voicebox_speak`` calls and the REST ``POST /speak`` route publish start/end events that DictateWindow subscribes to via /events/speak, so the floating pill surfaces whenever an agent is speaking. """ diff --git a/backend/mcp_server/server.py b/backend/mcp_server/server.py index 3a408df9..e8c89016 100644 --- a/backend/mcp_server/server.py +++ b/backend/mcp_server/server.py @@ -27,8 +27,8 @@ def build_mcp_server() -> FastMCP: mcp = FastMCP( name="voicebox", instructions=( - "Voicebox is a local voice I/O layer. Use `voicebox.speak` to " - "play text in a voice profile, `voicebox.transcribe` for " + "Voicebox is a local voice I/O layer. Use `voicebox_speak` to " + "play text in a voice profile, `voicebox_transcribe` for " "audio→text, and the `list_*` tools to discover profiles and " "captures." ), diff --git a/backend/mcp_server/tools.py b/backend/mcp_server/tools.py index fae79e48..264c8ca8 100644 --- a/backend/mcp_server/tools.py +++ b/backend/mcp_server/tools.py @@ -1,8 +1,9 @@ """Voicebox MCP tool implementations. -Thin wrappers over existing services/routes. Tools are registered with dotted -names (``voicebox.speak`` etc.) so they look natural in agent logs — -the Python function name stays snake_case. +Thin wrappers over existing services/routes. Tools are registered with +underscore-separated names (``voicebox_speak`` etc.): MCP clients such as +Claude Desktop validate tool names against ``^[a-zA-Z0-9_-]{1,64}$`` and +reject the whole tool list if any name contains a dot (#790). """ from __future__ import annotations @@ -202,7 +203,7 @@ def register_tools(mcp: FastMCP) -> None: name="voicebox_list_profiles", description=( "List available voice profiles (both cloned voices and presets). " - "Use the returned `name` with voicebox.speak(profile=...)." + "Use the returned `name` with voicebox_speak(profile=...)." ), ) async def voicebox_list_profiles() -> dict[str, Any]: diff --git a/backend/models.py b/backend/models.py index 5b3d8217..a42f3b7d 100644 --- a/backend/models.py +++ b/backend/models.py @@ -309,7 +309,7 @@ class GenerationSettingsUpdate(BaseModel): class MCPClientBindingResponse(BaseModel): """Per-MCP-client voice binding — what voice / engine the server should - use when a given client_id calls voicebox.speak without args, plus an + use when a given client_id calls voicebox_speak without args, plus an opt-in personality-rewrite default.""" client_id: str @@ -346,7 +346,7 @@ class MCPClientBindingListResponse(BaseModel): class SpeakRequest(BaseModel): - """Body for POST /speak — non-MCP REST surface that mirrors voicebox.speak.""" + """Body for POST /speak — non-MCP REST surface that mirrors voicebox_speak.""" text: str = Field(..., min_length=1, max_length=10000) profile: Optional[str] = Field( diff --git a/backend/routes/speak.py b/backend/routes/speak.py index 92efc75d..ae239ea1 100644 --- a/backend/routes/speak.py +++ b/backend/routes/speak.py @@ -1,4 +1,4 @@ -"""POST /speak — REST wrapper around voicebox.speak for non-MCP callers. +"""POST /speak — REST wrapper around voicebox_speak for non-MCP callers. Shell scripts, ACP, A2A, or any agent that doesn't speak MCP can hit this endpoint to play text through a cloned voice. Uses the same profile @@ -30,7 +30,7 @@ async def speak( request: Request, db: Session = Depends(get_db), ): - """Speak text in a voice profile. Mirrors voicebox.speak (MCP). + """Speak text in a voice profile. Mirrors voicebox_speak (MCP). Response shape matches POST /generate — a ``GenerationResponse`` with ``status="generating"`` and an ``id`` the caller polls at diff --git a/backend/tests/test_mcp_speak.py b/backend/tests/test_mcp_speak.py index ae79506c..53d893fa 100644 --- a/backend/tests/test_mcp_speak.py +++ b/backend/tests/test_mcp_speak.py @@ -1,4 +1,4 @@ -"""Tests for the voicebox.speak MCP tool's ``model_size`` plumbing (issue #884). +"""Tests for the voicebox_speak MCP tool's ``model_size`` plumbing (issue #884). The MCP speak path used to build its ``GenerationRequest`` without a ``model_size``, so every agent-triggered generation silently fell back to the diff --git a/docs/PROJECT_STATUS.md b/docs/PROJECT_STATUS.md index eef11d8a..60caae3b 100644 --- a/docs/PROJECT_STATUS.md +++ b/docs/PROJECT_STATUS.md @@ -121,7 +121,7 @@ POST /generate Shipped 2026-04-25 (PR #544). Voicebox went from a voice-cloning studio to a full voice studio — dictation in, agent speech out, a local LLM in the middle. - **Dictation** — global hotkey capture (push-to-talk + toggle chords), on-screen pill with live state, auto-paste into the focused field with clipboard save/restore, chord-picker UI. Scoped Accessibility permission (transcripts still land if paste is denied). -- **MCP server** at `http://127.0.0.1:17493/mcp` — `voicebox.speak` / `.transcribe` / `.list_captures` / `.list_profiles`. Streamable HTTP primary transport, stdio sidecar shim, per-client voice binding via `X-Voicebox-Client-Id`. Speaking pill always shows agent-initiated output. +- **MCP server** at `http://127.0.0.1:17493/mcp` — `voicebox_speak` / `voicebox_transcribe` / `voicebox_list_captures` / `voicebox_list_profiles`. Streamable HTTP primary transport, stdio sidecar shim, per-client voice binding via `X-Voicebox-Client-Id`. Speaking pill always shows agent-initiated output. - **Personality** — voice profiles carry an optional ≤2000-char persona. Compose (shuffle an in-character line) and Speak-in-character (rewrite input before TTS), both on a local Qwen3 LLM that doubles as the refinement model. - **Refinement** — on-device Qwen3 strips fillers, fixes punctuation, optional self-correction rewrites; Whisper hallucination-loop stripping at a 6-token threshold; per-capture flag snapshots; model picker (0.6B / 1.7B / 4B). - **`POST /speak` REST wrapper** and **i18next foundation** (English + zh-CN) also landed. diff --git a/docs/content/docs/overview/mcp-server.mdx b/docs/content/docs/overview/mcp-server.mdx index 4163cdd6..e947a7a4 100644 --- a/docs/content/docs/overview/mcp-server.mdx +++ b/docs/content/docs/overview/mcp-server.mdx @@ -105,15 +105,15 @@ for the shim to connect. | Tool | Use | |---|---| -| `voicebox.speak` | Speak text in a voice profile. Returns a `generation_id` to poll. | -| `voicebox.transcribe` | Whisper transcription of base64 audio or an absolute local path. | -| `voicebox.list_captures` | Recent captures with transcripts, paginated. | -| `voicebox.list_profiles` | Available voice profiles (cloned + preset). | +| `voicebox_speak` | Speak text in a voice profile. Returns a `generation_id` to poll. | +| `voicebox_transcribe` | Whisper transcription of base64 audio or an absolute local path. | +| `voicebox_list_captures` | Recent captures with transcripts, paginated. | +| `voicebox_list_profiles` | Available voice profiles (cloned + preset). | -### `voicebox.speak` +### `voicebox_speak` ```ts -voicebox.speak({ +voicebox_speak({ text: "Deploy complete.", profile?: "Morgan", // name or id; falls back to per-client binding, then default engine?: "qwen", // qwen | qwen_custom_voice | luxtts | chatterbox | chatterbox_turbo | tada | kokoro @@ -138,10 +138,10 @@ Returns: - **Persona mode** — `personality: true` and the profile must have a personality prompt set. The LLM rewrites the text in character before TTS. See [Voice Personalities](/overview/voice-personalities). -### `voicebox.transcribe` +### `voicebox_transcribe` ```ts -voicebox.transcribe({ +voicebox_transcribe({ audio_base64?: "", // exactly one of these two audio_path?: "/absolute/path/to/file.wav", language?: "en", @@ -151,18 +151,18 @@ voicebox.transcribe({ Returns `{ text, duration, language, model }`. 200 MB ceiling on either path. -### `voicebox.list_captures` +### `voicebox_list_captures` `{ limit?: 20, offset?: 0 }` → `{ captures: [...], total }`. `limit` is clamped to `1..=200`. -### `voicebox.list_profiles` +### `voicebox_list_profiles` No args → `{ profiles: [{ id, name, voice_type, language, has_personality }] }`. ## Voice resolution -Every call to `voicebox.speak` (and `POST /speak`) resolves the voice profile +Every call to `voicebox_speak` (and `POST /speak`) resolves the voice profile in this order: @@ -195,7 +195,7 @@ carries: | `label` | Display name in the Settings UI (e.g. "Claude Code"). | | `profile_id` | The voice this client uses when `profile` isn't passed. | | `default_engine` | Override the TTS engine for this client. | -| `default_personality` | When true, `voicebox.speak` routes through the profile's personality LLM (rewrite) by default. | +| `default_personality` | When true, `voicebox_speak` routes through the profile's personality LLM (rewrite) by default. | | `last_seen_at` | Last time the server saw a request from this client. | `last_seen_at` is stamped automatically by middleware on every `/mcp/*` @@ -239,8 +239,8 @@ agent: npx @modelcontextprotocol/inspector http://127.0.0.1:17493/mcp ``` -Start with `voicebox.list_profiles` to confirm wiring, then -`voicebox.speak` for end-to-end — you should hear audio and see the +Start with `voicebox_list_profiles` to confirm wiring, then +`voicebox_speak` for end-to-end — you should hear audio and see the generation land in the Captures tab. @@ -262,7 +262,7 @@ generation land in the Captures tab. boundary. If you're scripting against a shared host, prefer `audio_base64` so you don't have to think about path sandboxing. - **Voice cloning consent applies.** See [Voice Cloning](/overview/voice-cloning#limitations) - — an agent being able to call `voicebox.speak` in someone's voice + — an agent being able to call `voicebox_speak` in someone's voice doesn't change the ethics of whose voices you clone. ## Implementation notes diff --git a/docs/content/docs/overview/voice-personalities.mdx b/docs/content/docs/overview/voice-personalities.mdx index 65c797d3..c7dd95a8 100644 --- a/docs/content/docs/overview/voice-personalities.mdx +++ b/docs/content/docs/overview/voice-personalities.mdx @@ -128,7 +128,7 @@ until you flip it back off. - **Agents that speak in a voice you own.** Combine the persona toggle with the built-in [MCP Server](/overview/mcp-server) so Claude Code, Cursor, Cline, or any MCP-aware agent can talk back through a profile with a - personality. The agent calls `voicebox.speak({ text, profile, personality: + personality. The agent calls `voicebox_speak({ text, profile, personality: true })` and Voicebox rewrites the text in character before speaking. - **Interactive characters.** Games, narrative tools, accessibility experiences. A character with a personality description plus a cloned @@ -151,7 +151,7 @@ Personalities are accessible via REST: | `POST` | `/generate` | Include `personality: true` to run input text through the personality LLM before TTS. Same for `POST /speak`. | `POST /generate` with `personality: true` is the same primitive MCP's -`voicebox.speak` tool uses when you pass `personality: true`. Scripts and +`voicebox_speak` tool uses when you pass `personality: true`. Scripts and agents can use it directly. ## Limits and gotchas diff --git a/landing/src/components/AgentIntegration.tsx b/landing/src/components/AgentIntegration.tsx index 305cea3f..fc4034f5 100644 --- a/landing/src/components/AgentIntegration.tsx +++ b/landing/src/components/AgentIntegration.tsx @@ -23,7 +23,7 @@ const SCENARIOS: Scenario[] = [ { prefix: '$', text: 'claude run', tone: 'accent' }, { prefix: '✓', text: 'Tests passing (42 files)', tone: 'success' }, { prefix: '✓', text: 'Build succeeded in 12.4s', tone: 'success' }, - { prefix: '→', text: 'voicebox.speak({ profile: "Morgan" })', tone: 'dim' }, + { prefix: '→', text: 'voicebox_speak({ profile: "Morgan" })', tone: 'dim' }, ], utterance: 'Tests passing. Ready to merge.', }, @@ -35,7 +35,7 @@ const SCENARIOS: Scenario[] = [ { prefix: '$', text: 'cursor agent:deploy', tone: 'accent' }, { prefix: '✓', text: 'Migration applied (4 tables)', tone: 'success' }, { prefix: '✓', text: 'Deploy complete', tone: 'success' }, - { prefix: '→', text: 'voicebox.speak({ profile: "Scarlett" })', tone: 'dim' }, + { prefix: '→', text: 'voicebox_speak({ profile: "Scarlett" })', tone: 'dim' }, ], utterance: 'Deploy shipped. Prod is green.', }, @@ -46,7 +46,7 @@ const SCENARIOS: Scenario[] = [ log: [ { prefix: '$', text: 'cline task:review', tone: 'accent' }, { prefix: '!', text: '3 files need attention', tone: 'dim' }, - { prefix: '→', text: 'voicebox.speak({ profile: "Jarvis" })', tone: 'dim' }, + { prefix: '→', text: 'voicebox_speak({ profile: "Jarvis" })', tone: 'dim' }, ], utterance: 'Review ready. Three files to look at.', }, @@ -202,7 +202,7 @@ const MCP_CONFIG = `{ }`; const SPEAK_EXAMPLE = `// In any MCP-aware agent: -await voicebox.speak({ +await voicebox_speak({ text: "Deploy complete.", profile: "Morgan", })`; @@ -300,7 +300,7 @@ export function AgentIntegration() {

One tool call —{' '} - voicebox.speak — + voicebox_speak — and any MCP-aware agent can talk to you in a voice you’ve cloned. Claude Code, Cursor, Cline, or anything that speaks MCP.

diff --git a/landing/src/components/CapturesMockup.tsx b/landing/src/components/CapturesMockup.tsx index 6e4994c1..ed551001 100644 --- a/landing/src/components/CapturesMockup.tsx +++ b/landing/src/components/CapturesMockup.tsx @@ -152,7 +152,7 @@ const CAPTURES: Capture[] = [ transcriptRaw: "draft an update for the blog about the agent voice feature the key point is one MCP tool call and any agent on your machine gets a voice claude code finishes a long task calls voicebox dot speak and you hear it in a voice you've cloned morgan scarlett whatever you set up same pill that shows when you're dictating also shows when an agent is speaking so you always know what's coming out of your machine closes the whole voice IO loop for agents", transcriptRefined: - "Draft an update for the blog about the agent voice feature. The key point: one MCP tool call, and any agent on your machine gets a voice. Claude Code finishes a long task, calls voicebox.speak, and you hear it in a voice you've cloned — Morgan, Scarlett, whatever you've set up. The same pill that shows when you're dictating also shows when an agent is speaking, so you always know what's coming out of your machine. It closes the full voice I/O loop for agents.", + "Draft an update for the blog about the agent voice feature. The key point: one MCP tool call, and any agent on your machine gets a voice. Claude Code finishes a long task, calls voicebox_speak, and you hear it in a voice you've cloned — Morgan, Scarlett, whatever you've set up. The same pill that shows when you're dictating also shows when an agent is speaking, so you always know what's coming out of your machine. It closes the full voice I/O loop for agents.", durationMs: 41000, ago: '22 min ago', createdAtLabel: 'Apr 22, 3:29 PM', From d09c5c39e31a73e9c15da4cd674d2f80d312fa2e Mon Sep 17 00:00:00 2001 From: jamiepine <32987599+jamiepine@users.noreply.github.com> Date: Sat, 3 Oct 2026 09:28:42 +0000 Subject: [PATCH 35/35] docs: update the last two comment references to the dotted MCP tool names --- backend/requirements.txt | 2 +- tauri/src-tauri/src/main.rs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/requirements.txt b/backend/requirements.txt index 3c6160a5..ddc9bb6c 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -64,7 +64,7 @@ pedalboard>=0.9.0 httpx>=0.27.0 # MCP server (Model Context Protocol) — lets local AI agents call -# voicebox.speak / .transcribe / .list_captures / .list_profiles +# voicebox_speak / voicebox_transcribe / voicebox_list_captures / voicebox_list_profiles fastmcp>=3.0,<4.0 sse-starlette>=2.0 diff --git a/tauri/src-tauri/src/main.rs b/tauri/src-tauri/src/main.rs index 0f44ac90..a1edd132 100644 --- a/tauri/src-tauri/src/main.rs +++ b/tauri/src-tauri/src/main.rs @@ -1434,7 +1434,7 @@ pub fn run() { } }); - // Agent-initiated speech (voicebox.speak over MCP or POST /speak) + // Agent-initiated speech (voicebox_speak over MCP or POST /speak) // pops the pill up so the user can see what's coming out of their // machine. The `dictate:show` listener is kept for any frontend // caller that wants to force-surface the pill directly, but the