mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-18 14:20:42 -07:00
feat(capture): dictation, personalities, 0.5.0
Ships the Capture release end to end. Global-hotkey dictation with synthetic paste into the focused app on macOS and Windows, an on-screen pill across recording / transcribing / refining, customizable push-to- talk and toggle chords, and an accessibility-permission prompt scoped to Settings → Captures with inline re-check feedback. Voice profiles gain optional personalities that power compose / rewrite / respond actions via a local Qwen3 LLM — shared with refinement, so there is one local LLM in the app, not two. Refinement hardened with deterministic Whisper-loop collapse before the LLM sees the transcript, per-capture flag snapshots for re-runs, and a ten-transcript evaluation harness across every bundled refinement size. Version bump 0.4.5 → 0.5.0. Co-Authored-By: Claude Opus 4.7 (1M context) <[email protected]>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
ed2eec591a
commit
87c582ad54
@@ -8,9 +8,12 @@ without changing any importers.
|
||||
from .models import (
|
||||
Base,
|
||||
AudioChannel,
|
||||
Capture,
|
||||
CaptureSettings,
|
||||
ChannelDeviceMapping,
|
||||
EffectPreset,
|
||||
Generation,
|
||||
GenerationSettings,
|
||||
GenerationVersion,
|
||||
ProfileChannelMapping,
|
||||
ProfileSample,
|
||||
@@ -25,9 +28,12 @@ __all__ = [
|
||||
# Models
|
||||
"Base",
|
||||
"AudioChannel",
|
||||
"Capture",
|
||||
"CaptureSettings",
|
||||
"ChannelDeviceMapping",
|
||||
"EffectPreset",
|
||||
"Generation",
|
||||
"GenerationSettings",
|
||||
"GenerationVersion",
|
||||
"ProfileChannelMapping",
|
||||
"ProfileSample",
|
||||
|
||||
@@ -34,6 +34,7 @@ def run_migrations(engine) -> None:
|
||||
_migrate_generations(engine, inspector, tables)
|
||||
_migrate_effect_presets(engine, inspector, tables)
|
||||
_migrate_generation_versions(engine, inspector, tables)
|
||||
_migrate_capture_settings(engine, inspector, tables)
|
||||
_normalize_storage_paths(engine, tables)
|
||||
|
||||
|
||||
@@ -146,6 +147,8 @@ def _migrate_profiles(engine, inspector, tables: set[str]) -> None:
|
||||
_add_column(engine, "profiles", "design_prompt TEXT", "design_prompt")
|
||||
if "default_engine" not in columns:
|
||||
_add_column(engine, "profiles", "default_engine VARCHAR", "default_engine")
|
||||
if "personality" not in columns:
|
||||
_add_column(engine, "profiles", "personality TEXT", "personality")
|
||||
|
||||
|
||||
def _migrate_generations(engine, inspector, tables: set[str]) -> None:
|
||||
@@ -164,6 +167,13 @@ def _migrate_generations(engine, inspector, tables: set[str]) -> None:
|
||||
_add_column(engine, "generations", "model_size VARCHAR", "model_size")
|
||||
if "is_favorited" not in columns:
|
||||
_add_column(engine, "generations", "is_favorited BOOLEAN DEFAULT 0", "is_favorited")
|
||||
if "source" not in columns:
|
||||
_add_column(
|
||||
engine,
|
||||
"generations",
|
||||
"source VARCHAR NOT NULL DEFAULT 'manual'",
|
||||
"source",
|
||||
)
|
||||
|
||||
|
||||
def _migrate_effect_presets(engine, inspector, tables: set[str]) -> None:
|
||||
@@ -182,6 +192,40 @@ def _migrate_generation_versions(engine, inspector, tables: set[str]) -> None:
|
||||
_add_column(engine, "generation_versions", "source_version_id VARCHAR", "source_version_id")
|
||||
|
||||
|
||||
def _migrate_capture_settings(engine, inspector, tables: set[str]) -> None:
|
||||
if "capture_settings" not in tables:
|
||||
return
|
||||
columns = _get_columns(inspector, "capture_settings")
|
||||
if "allow_auto_paste" not in columns:
|
||||
_add_column(
|
||||
engine,
|
||||
"capture_settings",
|
||||
"allow_auto_paste BOOLEAN NOT NULL DEFAULT 1",
|
||||
"allow_auto_paste",
|
||||
)
|
||||
if "default_playback_voice_id" not in columns:
|
||||
_add_column(
|
||||
engine,
|
||||
"capture_settings",
|
||||
"default_playback_voice_id VARCHAR",
|
||||
"default_playback_voice_id",
|
||||
)
|
||||
if "chord_push_to_talk_keys" not in columns:
|
||||
_add_column(
|
||||
engine,
|
||||
"capture_settings",
|
||||
"chord_push_to_talk_keys TEXT NOT NULL DEFAULT '[\"MetaRight\",\"AltGr\"]'",
|
||||
"chord_push_to_talk_keys",
|
||||
)
|
||||
if "chord_toggle_to_talk_keys" not in columns:
|
||||
_add_column(
|
||||
engine,
|
||||
"capture_settings",
|
||||
"chord_toggle_to_talk_keys TEXT NOT NULL DEFAULT '[\"MetaRight\",\"AltGr\",\"Space\"]'",
|
||||
"chord_toggle_to_talk_keys",
|
||||
)
|
||||
|
||||
|
||||
def _normalize_storage_paths(engine, tables: set[str]) -> None:
|
||||
"""Normalize stored file paths to be relative to the configured data dir."""
|
||||
from pathlib import Path
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
from datetime import datetime
|
||||
import uuid
|
||||
|
||||
from sqlalchemy import Column, String, Integer, Float, DateTime, Text, ForeignKey, Boolean
|
||||
from sqlalchemy import Column, String, Integer, Float, DateTime, Text, ForeignKey, Boolean, JSON
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
|
||||
Base = declarative_base()
|
||||
@@ -33,6 +33,10 @@ class VoiceProfile(Base):
|
||||
preset_voice_id = Column(String, nullable=True) # e.g. "am_adam" — only for preset
|
||||
design_prompt = Column(Text, nullable=True) # text description — only for designed
|
||||
default_engine = Column(String, nullable=True) # auto-selected engine, locked for preset
|
||||
# Free-form character prompt used by the compose / rewrite / respond / speak
|
||||
# endpoints. Describes *what* this voice says and how, orthogonal to how
|
||||
# it sounds (which is handled by the preset / cloning metadata above).
|
||||
personality = Column(Text, nullable=True)
|
||||
|
||||
created_at = Column(DateTime, default=datetime.utcnow)
|
||||
updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
|
||||
@@ -67,6 +71,10 @@ class Generation(Base):
|
||||
status = Column(String, default="completed")
|
||||
error = Column(Text, nullable=True)
|
||||
is_favorited = Column(Boolean, default=False)
|
||||
# Origin of this generation — "manual" for regular /generate calls,
|
||||
# "personality_speak" for rows created by POST /profiles/{id}/speak.
|
||||
# Future sources (bulk import, agent replies, etc.) can extend this.
|
||||
source = Column(String, nullable=False, default="manual")
|
||||
created_at = Column(DateTime, default=datetime.utcnow)
|
||||
|
||||
|
||||
@@ -167,3 +175,70 @@ class ProfileChannelMapping(Base):
|
||||
|
||||
profile_id = Column(String, ForeignKey("profiles.id"), primary_key=True)
|
||||
channel_id = Column(String, ForeignKey("audio_channels.id"), primary_key=True)
|
||||
|
||||
|
||||
class CaptureSettings(Base):
|
||||
"""Singleton row holding user defaults for the capture/refine flow.
|
||||
|
||||
Kept server-side so every window, CLI client, and API consumer reads the
|
||||
same preferences. The ``id`` column is always 1.
|
||||
"""
|
||||
|
||||
__tablename__ = "capture_settings"
|
||||
|
||||
id = Column(Integer, primary_key=True, default=1)
|
||||
stt_model = Column(String, nullable=False, default="turbo")
|
||||
language = Column(String, nullable=False, default="auto")
|
||||
auto_refine = Column(Boolean, nullable=False, default=True)
|
||||
llm_model = Column(String, nullable=False, default="0.6B")
|
||||
smart_cleanup = Column(Boolean, nullable=False, default=True)
|
||||
self_correction = Column(Boolean, nullable=False, default=True)
|
||||
preserve_technical = Column(Boolean, nullable=False, default=True)
|
||||
allow_auto_paste = Column(Boolean, nullable=False, default=True)
|
||||
default_playback_voice_id = Column(String, nullable=True)
|
||||
# Lists of rdev::Key variant names (e.g. "MetaRight", "AltGr"). Right-hand
|
||||
# modifiers by default so they don't collide with left-hand system
|
||||
# shortcuts (Cmd+Opt+I devtools, Cmd+Opt+Esc force-quit).
|
||||
chord_push_to_talk_keys = Column(
|
||||
JSON, nullable=False, default=lambda: ["MetaRight", "AltGr"]
|
||||
)
|
||||
chord_toggle_to_talk_keys = Column(
|
||||
JSON, nullable=False, default=lambda: ["MetaRight", "AltGr", "Space"]
|
||||
)
|
||||
updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
|
||||
|
||||
|
||||
class GenerationSettings(Base):
|
||||
"""Singleton row for long-form TTS generation preferences."""
|
||||
|
||||
__tablename__ = "generation_settings"
|
||||
|
||||
id = Column(Integer, primary_key=True, default=1)
|
||||
max_chunk_chars = Column(Integer, nullable=False, default=800)
|
||||
crossfade_ms = Column(Integer, nullable=False, default=50)
|
||||
normalize_audio = Column(Boolean, nullable=False, default=True)
|
||||
autoplay_on_generate = Column(Boolean, nullable=False, default=True)
|
||||
updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
|
||||
|
||||
|
||||
class Capture(Base):
|
||||
"""A single voice input capture (dictation, recording, or uploaded file).
|
||||
|
||||
Stores the original audio alongside the raw transcript and, optionally, a
|
||||
refined version produced by the LLM. Refinement flags are serialized as
|
||||
JSON so we can reproduce the prompt that generated the refined text.
|
||||
"""
|
||||
|
||||
__tablename__ = "captures"
|
||||
|
||||
id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4()))
|
||||
audio_path = Column(String, nullable=False)
|
||||
source = Column(String, nullable=False, default="file") # dictation | recording | file
|
||||
language = Column(String, nullable=True)
|
||||
duration_ms = Column(Integer, nullable=True)
|
||||
transcript_raw = Column(Text, nullable=False, default="")
|
||||
transcript_refined = Column(Text, nullable=True)
|
||||
stt_model = Column(String, nullable=True)
|
||||
llm_model = Column(String, nullable=True)
|
||||
refinement_flags = Column(Text, nullable=True) # JSON blob
|
||||
created_at = Column(DateTime, default=datetime.utcnow)
|
||||
|
||||
Reference in New Issue
Block a user