mirror of
https://github.com/jamiepine/voicebox.git
synced 2026-09-28 06:35:18 -07:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
60110eeb0a | ||
|
|
6548c7e65a | ||
|
|
2076114972 | ||
|
|
9bc5b261fe | ||
|
|
f5ca08086f |
+23
-12
@@ -55,10 +55,23 @@ def build_server(cuda=False):
|
||||
# numpy 2.x / torch ABI mismatch fix: install memmove fallback for
|
||||
# torch.from_numpy() before the app starts. Runtime hooks run after
|
||||
# FrozenImporter is registered so frozen torch/numpy are importable.
|
||||
# Paths are passed relative to backend_dir because os.chdir(backend_dir)
|
||||
# runs before PyInstaller. Absolute paths would get baked into the
|
||||
# generated .spec, breaking reproducible builds on other machines / CI.
|
||||
args.extend(
|
||||
[
|
||||
"--runtime-hook",
|
||||
str(backend_dir / "pyi_rth_numpy_compat.py"),
|
||||
"pyi_rth_numpy_compat.py",
|
||||
# Stub torch.compiler.disable before transformers imports
|
||||
# flex_attention, which otherwise triggers torch._dynamo →
|
||||
# torch._numpy._ufuncs and crashes at module load under
|
||||
# PyInstaller. See pyi_rth_torch_compiler_disable.py.
|
||||
"--runtime-hook",
|
||||
"pyi_rth_torch_compiler_disable.py",
|
||||
# Per-module collection overrides (e.g. forcing scipy.stats._distn_infrastructure
|
||||
# to bundle .py source alongside .pyc so the runtime hook can source-patch it).
|
||||
"--additional-hooks-dir",
|
||||
"pyi_hooks",
|
||||
]
|
||||
)
|
||||
|
||||
@@ -125,6 +138,11 @@ def build_server(cuda=False):
|
||||
"backend.backends.chatterbox_backend",
|
||||
"--hidden-import",
|
||||
"backend.backends.chatterbox_turbo_backend",
|
||||
# chatterbox multilingual uses spacy_pkuseg for Chinese word
|
||||
# segmentation, which ships pickled dict files (dicts/default.pkl)
|
||||
# and native .so extensions that --hidden-import alone won't bundle.
|
||||
"--collect-all",
|
||||
"spacy_pkuseg",
|
||||
"--hidden-import",
|
||||
"backend.backends.luxtts_backend",
|
||||
"--hidden-import",
|
||||
@@ -241,20 +259,13 @@ def build_server(cuda=False):
|
||||
"--collect-submodules",
|
||||
"tada",
|
||||
# Kokoro 82M — lightweight TTS engine using misaki G2P
|
||||
# collect-all is required because transformers introspects .py source
|
||||
# files at runtime (e.g. _can_set_attn_implementation opens the class
|
||||
# file); hidden-import alone only bundles bytecode.
|
||||
"--hidden-import",
|
||||
"backend.backends.kokoro_backend",
|
||||
"--hidden-import",
|
||||
"--collect-all",
|
||||
"kokoro",
|
||||
"--hidden-import",
|
||||
"kokoro.pipeline",
|
||||
"--hidden-import",
|
||||
"kokoro.model",
|
||||
"--hidden-import",
|
||||
"kokoro.istftnet",
|
||||
"--hidden-import",
|
||||
"kokoro.modules",
|
||||
"--hidden-import",
|
||||
"kokoro.custom_stft",
|
||||
# misaki ships G2P data files (dictionaries, phoneme tables)
|
||||
# that must be bundled for espeak/en/ja/zh G2P to work
|
||||
"--collect-all",
|
||||
|
||||
@@ -89,14 +89,6 @@ def resolve_storage_path(path: str | Path | None) -> Path | None:
|
||||
|
||||
return stored_path
|
||||
|
||||
# 0.3.0 records sometimes stored relative paths with the data-dir name
|
||||
# baked in (e.g. "data/profiles/..."). Joining those directly with
|
||||
# _data_dir produces a spurious "<data_dir>/data/profiles/..." nest.
|
||||
if stored_path.parts and stored_path.parts[0] == "data":
|
||||
stored_path = (
|
||||
Path(*stored_path.parts[1:]) if len(stored_path.parts) > 1 else Path()
|
||||
)
|
||||
|
||||
return (_data_dir / stored_path).resolve()
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
"""
|
||||
Force scipy.stats._distn_infrastructure to be bundled with its .py source file
|
||||
alongside the .pyc bytecode.
|
||||
|
||||
The runtime hook in backend/pyi_rth_torch_compiler_disable.py patches this
|
||||
module's source at load time (the module has a `del obj` at line 369 that
|
||||
raises NameError under PyInstaller's frozen importer). That patch reads the
|
||||
source via loader.get_source(), which only works if the .py file was
|
||||
actually collected into the bundle.
|
||||
"""
|
||||
|
||||
module_collection_mode = "pyz+py"
|
||||
@@ -0,0 +1,11 @@
|
||||
"""
|
||||
Force transformers.masking_utils to be bundled with its .py source alongside
|
||||
the .pyc bytecode so the runtime hook in
|
||||
backend/pyi_rth_torch_compiler_disable.py can source-patch it.
|
||||
|
||||
The patch forces the torch<2.6 code path, bypassing `with TransformGetItemToIndex()`
|
||||
which our torch._dynamo no-op stub can't implement for real — the real context
|
||||
manager uses dynamo graph transforms to avoid `.item()` calls inside vmap.
|
||||
"""
|
||||
|
||||
module_collection_mode = "pyz+py"
|
||||
@@ -0,0 +1,540 @@
|
||||
"""
|
||||
PyInstaller runtime hook: stub torch._dynamo to a no-op module.
|
||||
|
||||
Problem
|
||||
-------
|
||||
transformers triggers torch._dynamo import at module-load time (not just
|
||||
when torch.compile is called) via class-body decorators:
|
||||
|
||||
transformers/modeling_utils.py:1984
|
||||
@torch._dynamo.allow_in_graph
|
||||
class PreTrainedModel(...)
|
||||
|
||||
transformers/integrations/flex_attention.py:61
|
||||
@torch.compiler.disable(recursive=False)
|
||||
class WrappedFlexAttention...
|
||||
|
||||
The attribute access triggers torch.__getattr__ -> importlib.import_module
|
||||
-> torch._dynamo -> torch._dynamo.utils imports torch._numpy ->
|
||||
torch._numpy._ndarray imports torch._numpy._ufuncs, which crashes under
|
||||
PyInstaller with:
|
||||
|
||||
File "torch/_numpy/_ufuncs.py", line 235, in <module>
|
||||
vars()[name] = deco_binary_ufunc(ufunc)
|
||||
NameError: name 'name' is not defined
|
||||
|
||||
(The module-level `for name in _binary: vars()[name] = ...` pattern works
|
||||
in a regular venv but fails in the PyInstaller bundle. Root cause is in
|
||||
PyInstaller's importer / bytecode pipeline and not easily fixed upstream.)
|
||||
|
||||
Surfaces as Kokoro failing to load when `from transformers import AlbertModel`
|
||||
trips the decorator chain.
|
||||
|
||||
Fix
|
||||
---
|
||||
voicebox never uses torch.compile / torch._dynamo for inference, so we
|
||||
replace torch._dynamo with a no-op stub module before transformers is
|
||||
imported. Any attribute access on the stub returns a pass-through callable,
|
||||
so `@torch._dynamo.allow_in_graph`, `torch._dynamo.is_compiling()`,
|
||||
`torch._dynamo.mark_static_address(...)`, etc. all work.
|
||||
|
||||
This hook is pure sys.modules manipulation — we deliberately do NOT import
|
||||
torch here. Runtime hooks run before the app starts and before
|
||||
pyi_rth_numpy_compat has had a chance to patch torch.from_numpy (it runs
|
||||
in a background thread, waiting for torch to appear in sys.modules).
|
||||
Eager-importing torch at hook time would trip the numpy ABI issue and
|
||||
kill the server process at startup.
|
||||
|
||||
torch.compiler.disable does not need a separate stub: its implementation
|
||||
is effectively `import torch._dynamo; return torch._dynamo.disable(...)`,
|
||||
and since our stub is in sys.modules, that call resolves to our no-op
|
||||
_NoopDecorator pass-through.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
|
||||
|
||||
# Diagnostics — log hook activity to a file alongside the bundle so we can
|
||||
# see what's happening when the server is run as a sidecar (no stdout for
|
||||
# runtime hook prints). Safe no-op if the file can't be written.
|
||||
_DIAG_PATH = os.path.join(tempfile.gettempdir(), "voicebox_rt_hook.log")
|
||||
|
||||
|
||||
def _diag(msg: str) -> None:
|
||||
try:
|
||||
with open(_DIAG_PATH, "a", encoding="utf-8") as f:
|
||||
f.write(msg + "\n")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
_HOOK_VERSION = "v6-masking-utils-finder"
|
||||
_diag(f"=== runtime hook load @ pid={os.getpid()} version={_HOOK_VERSION} ===")
|
||||
|
||||
|
||||
class _NoopDecorator:
|
||||
"""Multi-role no-op: decorator, falsey predicate, and context manager.
|
||||
|
||||
Returned from calls like `torch._dynamo.disable()` (decorator),
|
||||
`torch._dynamo.is_compiling()` (predicate used in `if not ...`), and
|
||||
`with torch._dynamo._trace_wrapped_higher_order_op.TransformGetItemToIndex():`
|
||||
(context manager used to scope an fx graph transformation).
|
||||
|
||||
By implementing __call__, __bool__, __enter__, __exit__, and __iter__ we
|
||||
cover every use pattern we've seen transformers/torch use on a stubbed
|
||||
object. Anything we haven't covered will raise a clearer error than a
|
||||
silent wrong-result.
|
||||
"""
|
||||
|
||||
__slots__ = ()
|
||||
|
||||
def __call__(self, fn=None, *args, **kwargs):
|
||||
return fn
|
||||
|
||||
def __bool__(self) -> bool:
|
||||
return False
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False # don't suppress exceptions
|
||||
|
||||
def __iter__(self):
|
||||
return iter(())
|
||||
|
||||
|
||||
_noop_decorator_singleton = _NoopDecorator()
|
||||
|
||||
|
||||
def _noop_callable(*args, **kwargs):
|
||||
# Direct-decorator use: @torch._dynamo.foo (no parens) — fn is positional
|
||||
if len(args) == 1 and callable(args[0]) and not kwargs:
|
||||
return args[0]
|
||||
# Side-effect call with non-callable arg(s), e.g. mark_static_address(tensor)
|
||||
return _noop_decorator_singleton
|
||||
|
||||
|
||||
class _NoopDynamoModule(types.ModuleType):
|
||||
"""Permissive stub: every attribute is a pass-through callable.
|
||||
|
||||
Covers attributes transformers hits at import time (allow_in_graph) and
|
||||
runtime (is_compiling, mark_static_address, reset, disable, ...).
|
||||
|
||||
Dunder attributes (__file__, __spec__, __loader__, ...) raise
|
||||
AttributeError so probes like inspect.getmodule() — which does
|
||||
`hasattr(m, '__file__')` then `os.path.normpath(m.__file__)` — see the
|
||||
module as having no source file and fall through to its normal
|
||||
handling, instead of receiving a function and blowing up.
|
||||
"""
|
||||
|
||||
def __getattr__(self, name: str):
|
||||
if name.startswith("__") and name.endswith("__"):
|
||||
raise AttributeError(name)
|
||||
return _noop_callable
|
||||
|
||||
|
||||
class _DynamoLoader:
|
||||
"""Loader used by _DynamoMetaPathFinder to materialise stub submodules."""
|
||||
|
||||
def create_module(self, spec):
|
||||
return _NoopDynamoModule(spec.name)
|
||||
|
||||
def exec_module(self, module):
|
||||
# Mark every stub submodule as a package so deeper submodule imports
|
||||
# (`from torch._dynamo.X.Y import Z`) keep working.
|
||||
module.__path__ = []
|
||||
|
||||
|
||||
class _DynamoMetaPathFinder:
|
||||
"""Resolve any `torch._dynamo.X[.Y...]` import to a no-op stub module.
|
||||
|
||||
Without this, `from torch._dynamo._trace_wrapped_higher_order_op import X`
|
||||
fails even with torch._dynamo pre-populated in sys.modules — Python's
|
||||
import machinery checks the parent's __path__ and then looks up the
|
||||
child, and we need to provide both.
|
||||
"""
|
||||
|
||||
def find_spec(self, fullname, path=None, target=None):
|
||||
if fullname == "torch._dynamo":
|
||||
return None # handled by the pre-populated sys.modules entry
|
||||
if not fullname.startswith("torch._dynamo."):
|
||||
return None
|
||||
from importlib.machinery import ModuleSpec
|
||||
|
||||
return ModuleSpec(fullname, _DynamoLoader(), is_package=True)
|
||||
|
||||
|
||||
class _TransformersStubFinder:
|
||||
"""Replace specific transformers submodules with no-op stubs.
|
||||
|
||||
Two modules are targeted:
|
||||
|
||||
1. transformers.utils.auto_docstring
|
||||
The real @auto_docstring decorator loads
|
||||
transformers.models.auto.modeling_auto just to build example docstrings,
|
||||
which drags in GenerationMixin -> candidate_generator -> sklearn.metrics
|
||||
-> scipy.stats._distn_infrastructure and trips (2) below. Docstrings
|
||||
aren't functional for inference, so a pass-through decorator is safe.
|
||||
|
||||
2. transformers.generation.candidate_generator
|
||||
Imported at module scope by transformers.generation.utils. It does
|
||||
`from sklearn.metrics import roc_curve` at module load, which triggers:
|
||||
|
||||
File "scipy/stats/_distn_infrastructure.py", line 369, in <module>
|
||||
NameError: name 'obj' is not defined
|
||||
|
||||
This is a PyInstaller-specific module-load bug (same class as the
|
||||
torch._numpy._ufuncs crash) where a module-level `for obj in [s for s
|
||||
in dir() if ...]` loop evaluates to empty in the bundle, leaving `obj`
|
||||
unbound before `del obj`.
|
||||
|
||||
The exports (AssistedCandidateGenerator, EarlyExitCandidateGenerator,
|
||||
etc.) are speculative-decoding helpers voicebox's TTS engines do not
|
||||
use; a no-op stub module satisfies the imports.
|
||||
"""
|
||||
|
||||
_STUBBED_MODULES = frozenset(
|
||||
{
|
||||
"transformers.utils.auto_docstring",
|
||||
"transformers.generation.candidate_generator",
|
||||
}
|
||||
)
|
||||
|
||||
def find_spec(self, fullname, path=None, target=None):
|
||||
if fullname not in self._STUBBED_MODULES:
|
||||
return None
|
||||
from importlib.machinery import ModuleSpec
|
||||
|
||||
return ModuleSpec(fullname, _NoopStubLoader(), is_package=False)
|
||||
|
||||
|
||||
class _NoopStubLoader:
|
||||
def create_module(self, spec):
|
||||
return _NoopDynamoModule(spec.name)
|
||||
|
||||
def exec_module(self, module):
|
||||
# _NoopDynamoModule.__getattr__ already answers every non-dunder
|
||||
# attribute with a pass-through callable, which satisfies
|
||||
# `from stubbed_module import X` for any X.
|
||||
pass
|
||||
|
||||
|
||||
def _patch_scipy_distn_source(source: str) -> str:
|
||||
"""Replace the unsafe `del obj` with a no-op that survives when obj is unbound.
|
||||
|
||||
Returns the input unchanged if the target line isn't found (e.g. scipy
|
||||
version has changed).
|
||||
"""
|
||||
target = "\ndel obj\n"
|
||||
replacement = "\nglobals().pop('obj', None)\n"
|
||||
if target in source:
|
||||
return source.replace(target, replacement, 1)
|
||||
return source
|
||||
|
||||
|
||||
def _patch_masking_utils_source(source: str) -> str:
|
||||
"""Force torch<2.6 code path in transformers.masking_utils.
|
||||
|
||||
The torch>=2.6 path uses `with TransformGetItemToIndex():` to allow
|
||||
`.item()` calls inside vmap. That context manager is implemented via
|
||||
torch._dynamo graph transforms, which our stub doesn't reproduce — it's
|
||||
a no-op. The inner `_vmap_for_bhqkv` then crashes with:
|
||||
|
||||
RuntimeError: vmap: It looks like you're calling .item() on a Tensor.
|
||||
|
||||
Forcing the torch<2.6 flag off selects sdpa_mask_older_torch which uses
|
||||
a different vmap pattern that does not hit .item() and does not need
|
||||
TransformGetItemToIndex.
|
||||
"""
|
||||
target = 'is_torch_greater_or_equal("2.6", accept_dev=True)'
|
||||
# Find the specific line that assigns _is_torch_greater_or_equal_than_2_6
|
||||
if "_is_torch_greater_or_equal_than_2_6 = " + target in source:
|
||||
return source.replace(
|
||||
"_is_torch_greater_or_equal_than_2_6 = " + target,
|
||||
"_is_torch_greater_or_equal_than_2_6 = False",
|
||||
1,
|
||||
)
|
||||
return source
|
||||
|
||||
|
||||
class _SourcePatchingFinder:
|
||||
"""Generic delegate-and-wrap meta-path finder that patches a module's
|
||||
source before exec'ing.
|
||||
|
||||
Subclasses declare `target` (module fullname) and `patch` (str->str).
|
||||
Requires the target module's .py source to be bundled (use a PyInstaller
|
||||
hook setting module_collection_mode = "pyz+py").
|
||||
"""
|
||||
|
||||
target: str
|
||||
patch_fn: callable = None
|
||||
|
||||
def find_spec(self, fullname, path=None, target=None):
|
||||
if fullname != self.target:
|
||||
return None
|
||||
for finder in sys.meta_path:
|
||||
if finder is self:
|
||||
continue
|
||||
find = getattr(finder, "find_spec", None)
|
||||
if find is None:
|
||||
continue
|
||||
try:
|
||||
real_spec = find(fullname, path, target)
|
||||
except Exception:
|
||||
continue
|
||||
if real_spec is None or real_spec.loader is None:
|
||||
continue
|
||||
real_spec.loader = _SourcePatchLoader(real_spec.loader, self.patch_fn)
|
||||
return real_spec
|
||||
return None
|
||||
|
||||
|
||||
class _SourcePatchLoader:
|
||||
"""Delegate loader that reads source via get_source, applies a patch, and
|
||||
compile/exec's the patched text into module.__dict__.
|
||||
"""
|
||||
|
||||
def __init__(self, inner, patch_fn):
|
||||
self._inner = inner
|
||||
self._patch_fn = patch_fn
|
||||
|
||||
def __getattr__(self, name):
|
||||
return getattr(self._inner, name)
|
||||
|
||||
def create_module(self, spec):
|
||||
return self._inner.create_module(spec)
|
||||
|
||||
def exec_module(self, module):
|
||||
source = None
|
||||
try:
|
||||
source = self._inner.get_source(module.__name__)
|
||||
except Exception as e:
|
||||
_diag(f"[source-patch] get_source({module.__name__}) failed: {e!r}")
|
||||
|
||||
if not source:
|
||||
_diag(
|
||||
f"[source-patch] no source for {module.__name__}; "
|
||||
"falling back to inner exec_module (patch NOT applied)"
|
||||
)
|
||||
self._inner.exec_module(module)
|
||||
return
|
||||
|
||||
patched = self._patch_fn(source)
|
||||
_diag(
|
||||
f"[source-patch] {module.__name__}: "
|
||||
f"patched={patched is not source}, len={len(patched)}"
|
||||
)
|
||||
spec = module.__spec__
|
||||
if spec is not None and spec.submodule_search_locations is not None:
|
||||
module.__path__ = spec.submodule_search_locations
|
||||
filename = getattr(self._inner, "path", module.__name__)
|
||||
exec(compile(patched, filename, "exec"), module.__dict__)
|
||||
_diag(f"[source-patch] {module.__name__} OK")
|
||||
|
||||
|
||||
class _MaskingUtilsFinder(_SourcePatchingFinder):
|
||||
target = "transformers.masking_utils"
|
||||
patch_fn = staticmethod(_patch_masking_utils_source)
|
||||
|
||||
|
||||
class _ScipyDistnPatchingFinder:
|
||||
"""Delegate-and-wrap finder for scipy.stats._distn_infrastructure.
|
||||
|
||||
That module ends with:
|
||||
|
||||
for obj in [s for s in dir() if s.startswith('_doc_')]:
|
||||
exec('del ' + obj)
|
||||
del obj
|
||||
|
||||
In the PyInstaller bundle the list comprehension evaluates to empty
|
||||
(module-level dir() under the frozen importer returns a different scope
|
||||
than CPython's normal module-exec path — same class of bug as the
|
||||
torch._numpy._ufuncs crash). The for loop body doesn't run, `obj` is
|
||||
never bound, and the trailing `del obj` raises NameError at module load.
|
||||
|
||||
This kills every downstream module: librosa (needed by nearly every TTS
|
||||
engine for mel filters) -> scipy.signal -> scipy.stats -> here.
|
||||
|
||||
Workaround: delegate to the real loader, but pre-bind `obj = None` in the
|
||||
module namespace before its bytecode runs. If the for loop executes, each
|
||||
iteration overwrites the sentinel via STORE_NAME (normal behaviour). If it
|
||||
doesn't, `del obj` removes the sentinel and module load succeeds. The
|
||||
`_doc_*` cleanup this line was meant to do is purely cosmetic — those vars
|
||||
stay in the module namespace but nothing references them after this point.
|
||||
"""
|
||||
|
||||
_TARGET = "scipy.stats._distn_infrastructure"
|
||||
|
||||
def find_spec(self, fullname, path=None, target=None):
|
||||
if fullname != self._TARGET:
|
||||
return None
|
||||
_diag(f"[scipy-finder] match: {fullname}, path={path!r}")
|
||||
# Delegate to the other finders to locate the real spec
|
||||
for finder in sys.meta_path:
|
||||
if finder is self:
|
||||
continue
|
||||
find = getattr(finder, "find_spec", None)
|
||||
if find is None:
|
||||
continue
|
||||
try:
|
||||
real_spec = find(fullname, path, target)
|
||||
except Exception as e:
|
||||
_diag(f"[scipy-finder] inner finder {type(finder).__name__} raised: {e}")
|
||||
continue
|
||||
if real_spec is None:
|
||||
continue
|
||||
if real_spec.loader is None:
|
||||
_diag(f"[scipy-finder] {type(finder).__name__} returned spec with loader=None")
|
||||
continue
|
||||
_diag(
|
||||
f"[scipy-finder] wrapped loader from "
|
||||
f"{type(finder).__name__} -> {type(real_spec.loader).__name__}"
|
||||
)
|
||||
real_spec.loader = _ScipyDistnPrebindLoader(real_spec.loader)
|
||||
return real_spec
|
||||
_diag("[scipy-finder] NO inner finder returned a spec")
|
||||
return None
|
||||
|
||||
|
||||
class _ScipyDistnPrebindLoader:
|
||||
"""Thin wrapper that pre-binds `obj = None` before delegating to the
|
||||
real PyInstaller loader.
|
||||
|
||||
Every other attribute/method delegates to the inner loader — PyiFrozenLoader
|
||||
is a rich FileLoader/ExecutionLoader with get_code/get_source/get_filename/
|
||||
is_package/get_resource_reader/etc., any of which Python's import machinery
|
||||
or 3rd-party code may call on spec.loader. Forwarding via __getattr__
|
||||
avoids breaking any of those paths (and preserves @_check_name contracts
|
||||
because the decorated methods run on the inner instance where self.name
|
||||
matches spec.name).
|
||||
"""
|
||||
|
||||
def __init__(self, inner):
|
||||
self._inner = inner
|
||||
|
||||
def __getattr__(self, name):
|
||||
# __getattr__ fires only for attrs not already on self, so delegate
|
||||
# everything that isn't create_module/exec_module (or __getattr__/init).
|
||||
return getattr(self._inner, name)
|
||||
|
||||
def create_module(self, spec):
|
||||
return self._inner.create_module(spec)
|
||||
|
||||
def exec_module(self, module):
|
||||
# Compile scipy's module source with the problematic line patched.
|
||||
#
|
||||
# The real module ends with:
|
||||
# for obj in [s for s in dir() if s.startswith('_doc_')]:
|
||||
# exec('del ' + obj)
|
||||
# del obj
|
||||
#
|
||||
# Under PyInstaller's frozen importer, `del obj` raises NameError
|
||||
# even when we pre-populate module.__dict__['obj'] — the pre-compiled
|
||||
# .pyc bytecode interacts with the frame setup differently than a
|
||||
# fresh compile() from source. Easiest robust fix: read the source
|
||||
# and replace `del obj` with a safe variant before compiling.
|
||||
#
|
||||
# Requires the .py source to be bundled alongside the .pyc — see
|
||||
# backend/pyi_hooks/hook-scipy.stats._distn_infrastructure.py.
|
||||
source = None
|
||||
try:
|
||||
source = self._inner.get_source(module.__name__)
|
||||
except Exception as e:
|
||||
_diag(f"[scipy-loader] get_source failed: {e!r}")
|
||||
|
||||
if source:
|
||||
patched = _patch_scipy_distn_source(source)
|
||||
_diag(
|
||||
f"[scipy-loader] source-patch path: patched={patched is not source}, "
|
||||
f"len={len(patched)}"
|
||||
)
|
||||
spec = module.__spec__
|
||||
if spec is not None and spec.submodule_search_locations is not None:
|
||||
module.__path__ = spec.submodule_search_locations
|
||||
filename = getattr(self._inner, "path", module.__name__)
|
||||
bytecode = compile(patched, filename, "exec")
|
||||
try:
|
||||
exec(bytecode, module.__dict__)
|
||||
except Exception as e:
|
||||
_diag(f"[scipy-loader] patched exec raised {type(e).__name__}: {e!r}")
|
||||
raise
|
||||
_diag(f"[scipy-loader] exec_module {module.__name__} OK (source-patched)")
|
||||
return
|
||||
|
||||
# No source available — fall back to the pre-bind approach. This is
|
||||
# best-effort; if the frozen .pyc really does see a different `obj`
|
||||
# slot, this will still crash, but we've done all we can without
|
||||
# source.
|
||||
_diag("[scipy-loader] no source available; falling back to pre-bind")
|
||||
module.__dict__["obj"] = None
|
||||
self._inner.exec_module(module)
|
||||
|
||||
|
||||
def _install_dynamo_stub() -> None:
|
||||
stub = _NoopDynamoModule("torch._dynamo")
|
||||
# Mark as a package so `from torch._dynamo.X import Y` imports work
|
||||
# (Python's import machinery checks parent.__path__ before looking up
|
||||
# the child).
|
||||
stub.__path__ = []
|
||||
# torch._dynamo.config is accessed as a nested attribute namespace
|
||||
# (e.g. `torch._dynamo.config.capture_scalar_outputs = True`), so use
|
||||
# a permissive module so any attr read returns a no-op and sets succeed.
|
||||
stub.config = _NoopDynamoModule("torch._dynamo.config")
|
||||
stub.config.__path__ = []
|
||||
sys.modules["torch._dynamo"] = stub
|
||||
sys.modules["torch._dynamo.config"] = stub.config
|
||||
|
||||
# Finders:
|
||||
# - torch._dynamo.* submodules -> no-op stubs
|
||||
# - transformers.utils.auto_docstring and
|
||||
# transformers.generation.candidate_generator -> no-op stubs (both
|
||||
# paths reach sklearn -> scipy.stats which trips a separate crash)
|
||||
# - scipy.stats._distn_infrastructure -> real load with `obj` pre-bound,
|
||||
# so librosa -> scipy.signal -> scipy.stats loads cleanly
|
||||
for _FinderCls in (
|
||||
_DynamoMetaPathFinder,
|
||||
_TransformersStubFinder,
|
||||
_ScipyDistnPatchingFinder,
|
||||
_MaskingUtilsFinder,
|
||||
):
|
||||
try:
|
||||
sys.meta_path.insert(0, _FinderCls())
|
||||
_diag(f"installed finder: {_FinderCls.__name__}")
|
||||
except Exception as e:
|
||||
_diag(f"FAILED to install {_FinderCls.__name__}: {e!r}")
|
||||
_diag(
|
||||
"final sys.meta_path head: "
|
||||
+ ", ".join(type(f).__name__ for f in sys.meta_path[:6])
|
||||
)
|
||||
|
||||
# If torch is already imported, also set the attribute on the package so
|
||||
# `torch._dynamo` resolves to our stub without triggering torch.__getattr__
|
||||
# (which would lazy-import the real module and crash).
|
||||
torch_mod = sys.modules.get("torch")
|
||||
if torch_mod is not None:
|
||||
torch_mod._dynamo = stub
|
||||
|
||||
|
||||
try:
|
||||
_install_dynamo_stub()
|
||||
except Exception as _e:
|
||||
# Best effort. If this fails the original NameError will surface when
|
||||
# transformers imports — no worse than not patching at all.
|
||||
_diag(f"_install_dynamo_stub FAILED: {_e!r}")
|
||||
|
||||
# NOTE: we deliberately do NOT import torch or torch.compiler here.
|
||||
# Runtime hooks run before the app starts and before pyi_rth_numpy_compat
|
||||
# has had a chance to patch torch.from_numpy (it runs in a background
|
||||
# thread, waiting for torch to appear in sys.modules). Importing torch
|
||||
# eagerly at hook time would trip the numpy ABI issue and kill the
|
||||
# server process at startup.
|
||||
#
|
||||
# torch.compiler.disable does not need an explicit stub: its
|
||||
# implementation is effectively `import torch._dynamo; return
|
||||
# torch._dynamo.disable(fn, recursive, reason=reason)`, and since our
|
||||
# stub is installed in sys.modules, that call resolves to our no-op
|
||||
# _NoopDecorator pass-through.
|
||||
@@ -5,7 +5,7 @@ from PyInstaller.utils.hooks import copy_metadata
|
||||
|
||||
datas = []
|
||||
binaries = []
|
||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.services.profiles', 'backend.services.history', 'backend.services.tts', 'backend.services.transcribe', 'backend.utils.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.backends.qwen_custom_voice_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.services.cuda', 'backend.services.effects', 'backend.utils.effects', 'backend.services.versions', 'pedalboard', 'chatterbox', 'chatterbox.tts_turbo', 'chatterbox.mtl_tts', 'backend.backends.chatterbox_backend', 'backend.backends.chatterbox_turbo_backend', 'backend.backends.luxtts_backend', 'zipvoice', 'zipvoice.luxvoice', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'requests', 'pkg_resources.extern', 'backend.backends.hume_backend', 'tada', 'tada.modules', 'tada.modules.tada', 'tada.modules.encoder', 'tada.modules.decoder', 'tada.modules.aligner', 'tada.modules.acoustic_spkr_verf', 'tada.nn', 'tada.nn.vibevoice', 'tada.utils', 'tada.utils.gray_code', 'tada.utils.text', 'backend.utils.dac_shim', 'torchaudio', 'backend.backends.kokoro_backend', 'kokoro', 'kokoro.pipeline', 'kokoro.model', 'kokoro.istftnet', 'kokoro.modules', 'kokoro.custom_stft', 'en_core_web_sm', 'loguru', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
||||
hiddenimports = ['backend', 'backend.main', 'backend.config', 'backend.database', 'backend.models', 'backend.services.profiles', 'backend.services.history', 'backend.services.tts', 'backend.services.transcribe', 'backend.utils.platform_detect', 'backend.backends', 'backend.backends.pytorch_backend', 'backend.backends.qwen_custom_voice_backend', 'backend.utils.audio', 'backend.utils.cache', 'backend.utils.progress', 'backend.utils.hf_progress', 'backend.services.cuda', 'backend.services.effects', 'backend.utils.effects', 'backend.services.versions', 'pedalboard', 'chatterbox', 'chatterbox.tts_turbo', 'chatterbox.mtl_tts', 'backend.backends.chatterbox_backend', 'backend.backends.chatterbox_turbo_backend', 'backend.backends.luxtts_backend', 'zipvoice', 'zipvoice.luxvoice', 'torch', 'transformers', 'fastapi', 'uvicorn', 'sqlalchemy', 'soundfile', 'qwen_tts', 'qwen_tts.inference', 'qwen_tts.inference.qwen3_tts_model', 'qwen_tts.inference.qwen3_tts_tokenizer', 'qwen_tts.core', 'qwen_tts.cli', 'requests', 'pkg_resources.extern', 'backend.backends.hume_backend', 'tada', 'tada.modules', 'tada.modules.tada', 'tada.modules.encoder', 'tada.modules.decoder', 'tada.modules.aligner', 'tada.modules.acoustic_spkr_verf', 'tada.nn', 'tada.nn.vibevoice', 'tada.utils', 'tada.utils.gray_code', 'tada.utils.text', 'backend.utils.dac_shim', 'torchaudio', 'backend.backends.kokoro_backend', 'en_core_web_sm', 'loguru', 'backend.backends.mlx_backend', 'mlx', 'mlx.core', 'mlx.nn', 'mlx_audio', 'mlx_audio.tts', 'mlx_audio.stt']
|
||||
datas += copy_metadata('qwen-tts')
|
||||
datas += copy_metadata('requests')
|
||||
datas += copy_metadata('transformers')
|
||||
@@ -18,6 +18,8 @@ hiddenimports += collect_submodules('jaraco')
|
||||
hiddenimports += collect_submodules('tada')
|
||||
hiddenimports += collect_submodules('mlx')
|
||||
hiddenimports += collect_submodules('mlx_audio')
|
||||
tmp_ret = collect_all('spacy_pkuseg')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('zipvoice')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('linacodec')
|
||||
@@ -34,6 +36,8 @@ tmp_ret = collect_all('perth')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('piper_phonemize')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('kokoro')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('misaki')
|
||||
datas += tmp_ret[0]; binaries += tmp_ret[1]; hiddenimports += tmp_ret[2]
|
||||
tmp_ret = collect_all('language_tags')
|
||||
@@ -54,9 +58,9 @@ a = Analysis(
|
||||
binaries=binaries,
|
||||
datas=datas,
|
||||
hiddenimports=hiddenimports,
|
||||
hookspath=[],
|
||||
hookspath=['pyi_hooks'],
|
||||
hooksconfig={},
|
||||
runtime_hooks=[],
|
||||
runtime_hooks=['pyi_rth_numpy_compat.py', 'pyi_rth_torch_compiler_disable.py'],
|
||||
excludes=['nvidia', 'nvidia.cublas', 'nvidia.cuda_cupti', 'nvidia.cuda_nvrtc', 'nvidia.cuda_runtime', 'nvidia.cudnn', 'nvidia.cufft', 'nvidia.curand', 'nvidia.cusolver', 'nvidia.cusparse', 'nvidia.nccl', 'nvidia.nvjitlink', 'nvidia.nvtx'],
|
||||
noarchive=False,
|
||||
optimize=0,
|
||||
|
||||
@@ -5,22 +5,17 @@ description: "How voice profile management works in Voicebox"
|
||||
|
||||
## Overview
|
||||
|
||||
Voice profiles are the unit of "a saved voice" in Voicebox. As of 0.4 they support two flavors backed by the same `profiles` table:
|
||||
|
||||
- **Cloned profiles** — store one or more reference audio samples; the cloning engine generates a voice embedding at use time
|
||||
- **Preset profiles** — store no audio; just a pointer to an engine-specific pre-built voice (e.g. Kokoro's `am_adam`, Qwen CustomVoice's `Ryan`)
|
||||
|
||||
The schema also reserves a third type, `designed`, for future text-described voices. Not currently used by any shipped engine.
|
||||
Voice profiles are the foundation of Voicebox's voice cloning capability. Each profile stores reference audio samples and metadata that the TTS model uses to clone a voice.
|
||||
|
||||
## Architecture
|
||||
|
||||
The voice profile system consists of three main components:
|
||||
|
||||
**Database Layer:** SQLite tables store profile metadata, sample references (cloned), and engine + voice ID (preset).
|
||||
**Database Layer:** SQLite tables store profile metadata and sample references.
|
||||
|
||||
**File Storage:** Audio samples are stored on disk in a structured directory format. Preset profiles have no on-disk audio.
|
||||
**File Storage:** Audio samples are stored on disk in a structured directory format.
|
||||
|
||||
**Profile Module:** `backend/services/profiles.py` provides the business logic for CRUD operations and dispatches to the appropriate engine based on `voice_type`.
|
||||
**Profile Module:** The `profiles.py` module provides the business logic for CRUD operations.
|
||||
|
||||
## Data Model
|
||||
|
||||
@@ -29,49 +24,27 @@ The voice profile system consists of three main components:
|
||||
```python
|
||||
class VoiceProfile(Base):
|
||||
__tablename__ = "profiles"
|
||||
|
||||
id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4()))
|
||||
|
||||
id = Column(String, primary_key=True)
|
||||
name = Column(String, unique=True, nullable=False)
|
||||
description = Column(Text)
|
||||
language = Column(String, default="en")
|
||||
avatar_path = Column(String, nullable=True)
|
||||
effects_chain = Column(Text, nullable=True)
|
||||
|
||||
# Voice type system — added v0.3.x
|
||||
voice_type = Column(String, default="cloned") # "cloned" | "preset" | "designed"
|
||||
preset_engine = Column(String, nullable=True) # e.g. "kokoro" — only for preset
|
||||
preset_voice_id = Column(String, nullable=True) # e.g. "am_adam" — only for preset
|
||||
design_prompt = Column(Text, nullable=True) # text description — only for designed (reserved)
|
||||
default_engine = Column(String, nullable=True) # auto-selected engine, locked for preset
|
||||
|
||||
created_at = Column(DateTime, default=datetime.utcnow)
|
||||
updated_at = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
|
||||
created_at = Column(DateTime)
|
||||
updated_at = Column(DateTime)
|
||||
```
|
||||
|
||||
The `voice_type` column discriminates the three flavors:
|
||||
|
||||
| `voice_type` | `preset_engine` | `preset_voice_id` | Samples in `profile_samples` |
|
||||
| ------------ | --------------- | ----------------- | ---------------------------- |
|
||||
| `cloned` | NULL | NULL | Required (≥1 row) |
|
||||
| `preset` | engine name | voice ID string | None |
|
||||
| `designed` | NULL | NULL | None (uses `design_prompt`) |
|
||||
|
||||
The `default_engine` column is set automatically when the profile is created. For preset profiles it's locked to the source engine — switching engines at generation time will skip the profile (and the UI auto-switches back when the user clicks a greyed-out card; see the floating generate box and profile grid).
|
||||
|
||||
### ProfileSample Table
|
||||
|
||||
```python
|
||||
class ProfileSample(Base):
|
||||
__tablename__ = "profile_samples"
|
||||
|
||||
id = Column(String, primary_key=True, default=lambda: str(uuid.uuid4()))
|
||||
|
||||
id = Column(String, primary_key=True)
|
||||
profile_id = Column(String, ForeignKey("profiles.id"))
|
||||
audio_path = Column(String, nullable=False)
|
||||
reference_text = Column(Text, nullable=False)
|
||||
```
|
||||
|
||||
Only populated for cloned profiles. Preset and designed profiles have zero rows in this table.
|
||||
|
||||
## File Structure
|
||||
|
||||
Profiles are stored in the data directory:
|
||||
|
||||
@@ -1,43 +1,32 @@
|
||||
---
|
||||
title: "Creating Voice Profiles"
|
||||
description: "How to create voice profiles, both cloning-based and preset-based"
|
||||
description: "Advanced guide to creating high-quality voice profiles"
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
A **voice profile** is a saved voice you can reuse across generations, stories, and the API. As of 0.4, Voicebox profiles come in two flavors that map to two different ways of getting a voice:
|
||||
Voice profiles are the foundation of voice cloning in Voicebox. This guide covers best practices for creating professional-quality voice profiles.
|
||||
|
||||
| Profile type | What it stores | Use when… |
|
||||
| -------------- | ---------------------------------------------------- | -------------------------------------------------------- |
|
||||
| **Cloned** | One or more reference audio samples + a voice embedding | You want to replicate a specific person's voice |
|
||||
| **Preset** | A reference to a pre-built voice in a specific engine | You want a curated, production-ready voice with no audio prep |
|
||||
|
||||
Both types live in the same Profiles tab and behave the same way at generation time — pick the type that matches your goal and follow the workflow below.
|
||||
|
||||
<Callout type="info">
|
||||
Not sure which to use? Cloning gives you a *specific* voice but needs clean audio. Preset gives you *good* voices instantly but you don't get to choose who they sound like.
|
||||
</Callout>
|
||||
|
||||
## Workflow A — Cloned Profiles
|
||||
|
||||
Use this when you want to replicate a specific person's voice from a recording.
|
||||
## Quick Start
|
||||
|
||||
<Steps>
|
||||
<Step title="Prepare Audio">
|
||||
10-30 seconds of clear speech, minimal background noise. See [Voice Cloning](/overview/voice-cloning) for the engine catalog.
|
||||
10-30 seconds of clear speech
|
||||
</Step>
|
||||
<Step title="Create Profile">
|
||||
**Profiles** → **+ New Profile** → choose a cloning engine (Qwen3-TTS, Chatterbox, LuxTTS, or TADA)
|
||||
**Profiles** → **+ New Profile**
|
||||
</Step>
|
||||
<Step title="Upload or Record Sample">
|
||||
Drag in an audio file, or record directly with the in-app recorder
|
||||
<Step title="Upload Sample">
|
||||
Add your audio file
|
||||
</Step>
|
||||
<Step title="Generate to Test">
|
||||
Use the profile to generate a test phrase. If quality is poor, add more samples
|
||||
<Step title="Generate">
|
||||
Use the profile to generate speech
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
### Audio Requirements (Cloning Only)
|
||||
## Audio Requirements
|
||||
|
||||
### Ideal Sample Characteristics
|
||||
|
||||
<Cards>
|
||||
<Card title="Duration">
|
||||
@@ -55,7 +44,7 @@ Use this when you want to replicate a specific person's voice from a recording.
|
||||
<Card title="Quality">
|
||||
**High fidelity**
|
||||
|
||||
44.1 kHz or 48 kHz sample rate
|
||||
44.1kHz or 48kHz sample rate
|
||||
Minimal compression
|
||||
</Card>
|
||||
<Card title="Content">
|
||||
@@ -69,16 +58,18 @@ Use this when you want to replicate a specific person's voice from a recording.
|
||||
### File Formats
|
||||
|
||||
Supported formats:
|
||||
- **WAV** (recommended) — Lossless quality
|
||||
- **MP3** — Acceptable, minimal compression
|
||||
- **M4A** — Acceptable
|
||||
- **FLAC** — Lossless alternative
|
||||
- **WAV** (recommended) - Lossless quality
|
||||
- **MP3** - Acceptable, minimal compression
|
||||
- **M4A** - Acceptable
|
||||
- **FLAC** - Lossless alternative
|
||||
|
||||
<Callout type="info">
|
||||
Use WAV for best results. Avoid heavily compressed formats.
|
||||
</Callout>
|
||||
|
||||
### Recording Tips
|
||||
## Recording Tips
|
||||
|
||||
### Environment
|
||||
|
||||
<AccordionGroup>
|
||||
<Accordion title="Quiet Space">
|
||||
@@ -96,25 +87,27 @@ Supported formats:
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Recording Settings">
|
||||
- 44.1 kHz or 48 kHz sample rate
|
||||
- 44.1kHz or 48kHz sample rate
|
||||
- 16-bit or 24-bit depth
|
||||
- Mono is fine (stereo will be converted)
|
||||
- Avoid automatic gain control
|
||||
</Accordion>
|
||||
</AccordionGroup>
|
||||
|
||||
### Speaking Style
|
||||
### Speaking
|
||||
|
||||
- **Natural pace** — Don't rush or speak too slowly
|
||||
- **Clear articulation** — Pronounce words clearly
|
||||
- **Consistent volume** — Maintain steady loudness
|
||||
- **Normal tone** — Speak as you normally would
|
||||
- **Complete sentences** — Avoid fragments or "ums"
|
||||
- **Natural pace** - Don't rush or speak too slowly
|
||||
- **Clear articulation** - Pronounce words clearly
|
||||
- **Consistent volume** - Maintain steady loudness
|
||||
- **Normal tone** - Speak as you normally would
|
||||
- **Complete sentences** - Avoid fragments or "ums"
|
||||
|
||||
### Multiple Samples
|
||||
## Multiple Samples
|
||||
|
||||
Adding multiple samples can significantly improve quality:
|
||||
|
||||
### Why Multiple Samples?
|
||||
|
||||
<Cards>
|
||||
<Card title="Robustness">
|
||||
Model learns a more complete representation
|
||||
@@ -130,57 +123,110 @@ Adding multiple samples can significantly improve quality:
|
||||
</Card>
|
||||
</Cards>
|
||||
|
||||
### Sample Variety
|
||||
|
||||
Consider adding samples with:
|
||||
|
||||
1. **Different tones** — casual, formal, excited, calm
|
||||
2. **Different content** — narratives, questions, statements
|
||||
3. **Different recording conditions** — studio quality, room acoustics
|
||||
1. **Different tones**
|
||||
- Casual conversation
|
||||
- Professional/formal
|
||||
- Excited/enthusiastic
|
||||
- Calm/serious
|
||||
|
||||
2. **Different content**
|
||||
- Narratives
|
||||
- Questions
|
||||
- Statements
|
||||
- Emotions (happy, sad, neutral)
|
||||
|
||||
3. **Different recording conditions**
|
||||
- Studio quality
|
||||
- Phone call quality (if needed)
|
||||
- Room acoustics
|
||||
|
||||
<Callout type="warn">
|
||||
All samples should be from the **same speaker**. Mixing voices will produce poor results.
|
||||
</Callout>
|
||||
|
||||
### Processing Existing Audio
|
||||
## Processing Existing Audio
|
||||
|
||||
If you have existing audio (podcasts, videos, etc.):
|
||||
|
||||
### Extracting Clean Segments
|
||||
|
||||
<Steps>
|
||||
<Step title="Find Clean Speech">
|
||||
Look for segments with just the target speaker, no background music, minimal noise
|
||||
Look for segments with:
|
||||
- Just the target speaker
|
||||
- No background music
|
||||
- Minimal noise
|
||||
</Step>
|
||||
|
||||
<Step title="Use Audio Editor">
|
||||
Tools like Audacity or Adobe Audition: cut clean 10-30s segments, remove silence at start/end, normalize volume
|
||||
Tools like Audacity or Adobe Audition:
|
||||
- Cut out clean 10-30s segments
|
||||
- Remove silence at start/end
|
||||
- Normalize volume if needed
|
||||
</Step>
|
||||
|
||||
<Step title="Export as WAV">
|
||||
Save as high-quality WAV file
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
For light background noise, use Audacity's noise reduction (gentle settings — over-processing introduces artifacts).
|
||||
### Noise Reduction
|
||||
|
||||
### Testing & Iteration
|
||||
If you have light background noise:
|
||||
|
||||
After creating a cloned profile:
|
||||
```
|
||||
1. Use noise reduction in Audacity:
|
||||
- Select noise-only section
|
||||
- Get Noise Profile
|
||||
- Select full audio
|
||||
- Apply noise reduction (gentle settings)
|
||||
|
||||
2. Avoid over-processing:
|
||||
- Can introduce artifacts
|
||||
- May reduce voice quality
|
||||
```
|
||||
|
||||
## Testing & Iteration
|
||||
|
||||
### Test Your Profile
|
||||
|
||||
After creating a profile:
|
||||
|
||||
<Steps>
|
||||
<Step title="Generate Test">
|
||||
Try a simple phrase: `"Hello, this is a test of my voice profile."`
|
||||
Generate a simple phrase:
|
||||
```
|
||||
"Hello, this is a test of my voice profile."
|
||||
```
|
||||
</Step>
|
||||
|
||||
<Step title="Evaluate Quality">
|
||||
Listen for natural tone, clear pronunciation, proper prosody, lack of artifacts
|
||||
Listen for:
|
||||
- Natural tone
|
||||
- Clear pronunciation
|
||||
- Proper prosody
|
||||
- Lack of artifacts
|
||||
</Step>
|
||||
|
||||
<Step title="Iterate">
|
||||
If quality is poor: add more samples, try different source audio, check sample quality
|
||||
If quality is poor:
|
||||
- Add more samples
|
||||
- Try different source audio
|
||||
- Check sample quality
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
#### Common Issues
|
||||
### Common Issues
|
||||
|
||||
<AccordionGroup>
|
||||
<Accordion title="Robotic Voice">
|
||||
**Cause**: Poor quality samples or too short
|
||||
|
||||
**Fix**: Use longer, higher-quality samples
|
||||
**Fix**: Use longer, higher quality samples
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Wrong Tone">
|
||||
@@ -196,89 +242,51 @@ After creating a cloned profile:
|
||||
</Accordion>
|
||||
</AccordionGroup>
|
||||
|
||||
## Workflow B — Preset Profiles
|
||||
|
||||
Use this when you want a ready-made voice without recording anything. Available engines: **Kokoro 82M** (50 voices) and **Qwen CustomVoice** (9 voices). See [Preset Voices](/overview/preset-voices) for the full catalog.
|
||||
|
||||
<Steps>
|
||||
<Step title="Create Profile">
|
||||
**Profiles** → **+ New Profile** → choose **Kokoro** or **Qwen CustomVoice** as the engine
|
||||
</Step>
|
||||
<Step title="Pick a Voice">
|
||||
The engine's voice catalog appears. Click any voice to preview it
|
||||
</Step>
|
||||
<Step title="Name and Save">
|
||||
Give the profile a name. No audio sample required
|
||||
</Step>
|
||||
<Step title="Generate">
|
||||
The profile is ready immediately — use it in the floating generate box or Generate page
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
<Callout type="info">
|
||||
Preset profiles are **locked to their source engine**. Switching to a different engine in the floating generate box greys out the profile, since the voice only exists in that engine. Clicking a greyed profile auto-switches the engine back.
|
||||
</Callout>
|
||||
|
||||
### Qwen CustomVoice + Instruct
|
||||
|
||||
Preset voices in Qwen CustomVoice support **delivery instructions** — natural-language style control over tone, pace, and emotion. The floating generate box shows a slider icon next to the generate button when a Qwen CustomVoice profile is selected; click it to reveal the instruct textarea.
|
||||
|
||||
See [Preset Voices → Using Instruct Mode](/overview/preset-voices#using-instruct-mode) for examples.
|
||||
|
||||
## Advanced Tips
|
||||
|
||||
### Celebrity / Character Voices (Cloning)
|
||||
### Celebrity/Character Voices
|
||||
|
||||
For cloning public figures or characters:
|
||||
|
||||
1. **Legal considerations** — Ensure you have rights or it's clearly fair use
|
||||
2. **Source quality** — Find high-quality interview audio or clean clips
|
||||
3. **Consistency** — Use clips where they speak similarly
|
||||
4. **Multiple samples** — Very important for recognizable voices
|
||||
1. **Legal considerations** - Ensure you have rights or it's fair use
|
||||
2. **Source quality** - Find high-quality interview audio or clean clips
|
||||
3. **Consistency** - Use clips where they speak similarly
|
||||
4. **Multiple samples** - Very important for recognizable voices
|
||||
|
||||
### Accent & Dialect (Cloning)
|
||||
### Accent & Dialect
|
||||
|
||||
Cloning models preserve accent and dialect:
|
||||
The model will preserve accent and dialect:
|
||||
|
||||
- British English samples generate British English output
|
||||
- Southern accent samples produce Southern accent output
|
||||
- Regional pronunciations are maintained
|
||||
- British English will generate British English
|
||||
- Southern accent will produce Southern accent
|
||||
- Regional pronunciations will be maintained
|
||||
|
||||
### Emotion Transfer (Cloning)
|
||||
### Emotion Transfer
|
||||
|
||||
The emotional tone of samples affects generation:
|
||||
|
||||
- Energetic samples → energetic output
|
||||
- Calm samples → calm output
|
||||
- Mix samples for a more versatile profile
|
||||
|
||||
For Qwen CustomVoice presets, use the **instruct** field instead of relying on sample emotion — that's exactly what it controls.
|
||||
- Energetic samples → Energetic output
|
||||
- Calm samples → Calm output
|
||||
- Mix samples for versatile profile
|
||||
|
||||
## Managing Profiles
|
||||
|
||||
### Organization
|
||||
|
||||
- **Descriptive names** — "John Smith - Professional Narrator"
|
||||
- **Add descriptions** — Note recording conditions, use cases, or which preset voice
|
||||
- **Language tags** — Mark the primary language
|
||||
- **Archive unused** — Keep profile list manageable
|
||||
- **Descriptive names** - "John Smith - Professional Narrator"
|
||||
- **Add descriptions** - Note recording conditions, use cases
|
||||
- **Language tags** - Mark the primary language
|
||||
- **Archive unused** - Keep profile list manageable
|
||||
|
||||
### Export / Import
|
||||
### Export/Import
|
||||
|
||||
- **Export** profiles to share or backup
|
||||
- **Import** from colleagues or teammates
|
||||
- **Cloned profiles** export with their voice embeddings (not the original audio)
|
||||
- **Preset profiles** export as engine + voice ID metadata only — the importer must have that engine's model installed
|
||||
- Profiles include voice embeddings, not original audio
|
||||
|
||||
## Next Steps
|
||||
|
||||
<Cards>
|
||||
<Card title="Voice Cloning" href="/overview/voice-cloning">
|
||||
Engine catalog and best practices for cloning
|
||||
</Card>
|
||||
<Card title="Preset Voices" href="/overview/preset-voices">
|
||||
Full catalog of Kokoro and Qwen CustomVoice voices
|
||||
</Card>
|
||||
<Card title="Generate Speech" href="/overview/generating-speech">
|
||||
Use your profile to generate speech
|
||||
</Card>
|
||||
|
||||
@@ -1,236 +0,0 @@
|
||||
---
|
||||
title: "GPU Acceleration"
|
||||
description: "How Voicebox uses your GPU — auto-detection, manual setup, troubleshooting"
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Voicebox auto-detects available accelerators on first launch and picks the fastest backend it can use. For most people this just works — open the app and you're already on the right backend.
|
||||
|
||||
This page is for the cases where it doesn't:
|
||||
|
||||
- You have a GPU but Voicebox is running on CPU
|
||||
- You upgraded GPUs (especially to RTX 50-series / Blackwell) and generation broke
|
||||
- You want to switch backends manually (e.g. force MLX over PyTorch on Apple Silicon)
|
||||
- You see `[UNSUPPORTED - see logs]` next to your GPU in Settings
|
||||
|
||||
## Backend Matrix
|
||||
|
||||
| Platform | Auto-selected backend | Notes |
|
||||
| --------------------------- | ------------------------- | ---------------------------------------------------- |
|
||||
| **macOS Apple Silicon** | MLX (Metal) | 4-5x faster than PyTorch via Apple Neural Engine |
|
||||
| **macOS Intel** | PyTorch CPU | No GPU acceleration available; PyTorch ≥ 2.2 only |
|
||||
| **Windows + NVIDIA** | PyTorch CUDA (cu128) | Auto-downloads the CUDA backend binary on first use |
|
||||
| **Windows + Intel Arc** | PyTorch XPU (IPEX) | New in 0.4 — works with Arc A-series and B-series |
|
||||
| **Windows generic GPU** | DirectML | Universal Windows GPU support; slower than CUDA |
|
||||
| **Linux + NVIDIA** | PyTorch CUDA (cu128) | Same auto-download flow as Windows |
|
||||
| **Linux + AMD** | PyTorch ROCm | Auto-configures `HSA_OVERRIDE_GFX_VERSION` |
|
||||
| **Linux + Intel Arc** | PyTorch XPU (IPEX) | |
|
||||
| **Any (no GPU)** | PyTorch CPU | Works everywhere; expect 5-50x slower than GPU |
|
||||
|
||||
The detected backend is shown in Settings → GPU. Logs at startup also print the chosen backend and the device name.
|
||||
|
||||
## Apple Silicon — MLX vs PyTorch
|
||||
|
||||
On M-series Macs, Voicebox ships an MLX-optimized backend that uses the Apple Neural Engine. It's **4-5x faster** than the PyTorch (CPU/Metal) path for supported engines.
|
||||
|
||||
| Engine | MLX support | Notes |
|
||||
| -------------------- | ----------- | ------------------------------------------- |
|
||||
| Qwen3-TTS | ✅ Native | Uses MLX exclusively when available |
|
||||
| Chatterbox / Turbo | PyTorch MPS | Falls back to Metal via PyTorch |
|
||||
| LuxTTS | PyTorch MPS | |
|
||||
| TADA | PyTorch MPS | |
|
||||
| Kokoro | PyTorch MPS | Requires `PYTORCH_ENABLE_MPS_FALLBACK=1` |
|
||||
| Qwen CustomVoice | PyTorch MPS | |
|
||||
| Whisper (transcribe) | ✅ Native | MLX-Whisper is the default on Apple Silicon |
|
||||
|
||||
The Whisper Turbo + MLX combo dropped transcription latency from ~20s to ~2-3s on M-series chips (see CHANGELOG entry for v0.1.10).
|
||||
|
||||
## Windows / Linux + NVIDIA — The CUDA Backend Swap
|
||||
|
||||
Voicebox doesn't bundle CUDA into the main installer (it would balloon downloads to multi-gigabyte territory for users who don't have an NVIDIA GPU). Instead, when you first need it, the app downloads a separate **CUDA backend binary** that contains the PyTorch + CUDA runtime.
|
||||
|
||||
<Steps>
|
||||
<Step title="Open Settings → GPU">
|
||||
If an NVIDIA GPU is detected, you'll see "Install CUDA backend" in the GPU panel
|
||||
</Step>
|
||||
<Step title="Click Install">
|
||||
The app downloads two archives separately:
|
||||
- **Server core** (~200-400 MB) — versioned with each Voicebox release
|
||||
- **CUDA libs** (~4 GB) — the heavy PyTorch + CUDA DLLs, versioned independently
|
||||
</Step>
|
||||
<Step title="Restart">
|
||||
Voicebox restarts to swap in the CUDA backend
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
<Callout type="info">
|
||||
The split-archive design (added in v0.4) means most Voicebox upgrades only redownload the small server-core archive. The 4 GB libs archive is only refreshed when the underlying CUDA toolkit or torch major version changes.
|
||||
</Callout>
|
||||
|
||||
### Auto-update
|
||||
|
||||
When a new Voicebox release ships, the GPU panel checks if the bundled server-core matches the installed CUDA version. If only the core changed (typical), it pulls the new core in the background. If the libs version changed (rare — only happens on cu126 → cu128 type bumps), you'll be prompted to confirm the larger download.
|
||||
|
||||
## RTX 50-series / Blackwell
|
||||
|
||||
Voicebox 0.4 added explicit RTX 50-series support:
|
||||
|
||||
- CUDA toolkit upgraded to **cu128** (previous releases used cu126 which lacks Blackwell kernels)
|
||||
- Build pinned with `TORCH_CUDA_ARCH_LIST=...12.0+PTX` for forward-compatibility
|
||||
|
||||
If you're on an RTX 5070 / 5080 / 5090 and you see "no kernel image is available" errors:
|
||||
|
||||
1. Make sure you're on Voicebox **≥ 0.4.0** (Settings → About)
|
||||
2. Reinstall the CUDA backend (Settings → GPU → Reinstall CUDA backend) — older installs may have stale cu126 libs
|
||||
3. If errors persist, see the GPU compatibility warnings section below
|
||||
|
||||
## Intel Arc (XPU)
|
||||
|
||||
New in 0.4. Works with both Arc A-series (Alchemist: A380, A580, A750, A770) and B-series (Battlemage).
|
||||
|
||||
### Setup
|
||||
|
||||
Voicebox auto-detects Arc GPUs and routes through Intel's PyTorch XPU backend (powered by IPEX — Intel Extension for PyTorch). No extra installation step beyond the standard Voicebox install.
|
||||
|
||||
Verify it's working:
|
||||
- Settings → GPU should show **XPU** followed by your Arc model name (e.g. `XPU (Intel Arc A770)`)
|
||||
- Startup logs print `Backend: PYTORCH` and `GPU: XPU (Intel Arc ...)`
|
||||
|
||||
### Engines on XPU
|
||||
|
||||
All PyTorch-based engines work on XPU. Performance is generally between CPU and CUDA — expect ~2-3x speedup over CPU for the larger models.
|
||||
|
||||
## DirectML
|
||||
|
||||
The fallback for Windows users with non-NVIDIA, non-Intel-Arc GPUs (older AMD discrete, integrated GPUs, etc.). Slower than CUDA and XPU but provides some acceleration over CPU.
|
||||
|
||||
Auto-selected when no other GPU backend is available.
|
||||
|
||||
## AMD ROCm (Linux)
|
||||
|
||||
ROCm provides PyTorch GPU acceleration on AMD discrete GPUs. Voicebox auto-configures `HSA_OVERRIDE_GFX_VERSION` for common cards that need the override.
|
||||
|
||||
### Verifying
|
||||
|
||||
```bash
|
||||
# In a terminal
|
||||
echo $HSA_OVERRIDE_GFX_VERSION
|
||||
# Should show e.g. 10.3.0 for RX 6000 series
|
||||
```
|
||||
|
||||
If detection fails, set the variable manually before launching Voicebox:
|
||||
|
||||
```bash
|
||||
export HSA_OVERRIDE_GFX_VERSION=10.3.0
|
||||
voicebox
|
||||
```
|
||||
|
||||
Common values:
|
||||
- `10.3.0` — RX 6000 series (RDNA 2)
|
||||
- `11.0.0` — RX 7000 series (RDNA 3)
|
||||
- `9.0.0` — Older Vega cards
|
||||
|
||||
## GPU Compatibility Warnings
|
||||
|
||||
Voicebox 0.4 added a runtime check that compares your GPU's compute capability against the architectures the bundled PyTorch was compiled for. If they don't match, you'll see:
|
||||
|
||||
- A startup log line: `WARNING: GPU COMPATIBILITY: <your GPU> is not supported by this PyTorch build...`
|
||||
- The GPU label in Settings shows `[UNSUPPORTED - see logs]`
|
||||
- The `/health` API returns a populated `gpu_compatibility_warning` field
|
||||
|
||||
### What to do
|
||||
|
||||
The most common trigger is a brand-new GPU architecture that pre-built PyTorch wheels don't yet cover natively. In order of preference:
|
||||
|
||||
1. **Update Voicebox** — newer releases ship newer PyTorch with broader arch support
|
||||
2. **Reinstall the CUDA backend** — Settings → GPU → Reinstall CUDA backend
|
||||
3. **For bleeding-edge GPUs (newer than current Blackwell):** install PyTorch nightly manually:
|
||||
```bash
|
||||
pip install torch --index-url https://download.pytorch.org/whl/nightly/cu128 --force-reinstall
|
||||
```
|
||||
Then point Voicebox at that environment via [Remote Mode](/overview/remote-mode) until stable PyTorch catches up.
|
||||
4. **Fall back to CPU** temporarily — set `VOICEBOX_FORCE_CPU=1` before launching
|
||||
|
||||
## CPU-Only Fallback
|
||||
|
||||
When no GPU is available (or you've forced it off), Voicebox runs the PyTorch CPU backend. Expect:
|
||||
|
||||
- 5-50x slower generation depending on engine and text length
|
||||
- Heavy CPU usage during generation
|
||||
- Some engines work better than others on CPU:
|
||||
- **Kokoro 82M** — runs at realtime on modern CPUs
|
||||
- **LuxTTS** — exceeds 150x realtime on CPU
|
||||
- **Chatterbox Turbo (350M)** — usable but slow
|
||||
- Larger models (Qwen 1.7B, Chatterbox Multilingual, TADA 3B) — painful
|
||||
|
||||
For CPU-bound use cases, prefer the smaller, lighter engines.
|
||||
|
||||
## Verifying Your Setup
|
||||
|
||||
Three places to check that the right backend is being used:
|
||||
|
||||
<Steps>
|
||||
<Step title="Settings → GPU">
|
||||
Shows the detected backend, GPU model, and VRAM (when applicable). Look for the `[UNSUPPORTED - see logs]` suffix
|
||||
</Step>
|
||||
<Step title="Settings → Logs">
|
||||
The "Server logs" tab shows the startup banner with `Backend: <type>` and `GPU: <name>`
|
||||
</Step>
|
||||
<Step title="Health endpoint">
|
||||
`curl http://localhost:17493/health` returns a JSON payload with `backend_type`, `backend_variant`, and `gpu_compatibility_warning` (when applicable)
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
<AccordionGroup>
|
||||
<Accordion title="Settings shows CPU instead of my GPU">
|
||||
- On NVIDIA: install the CUDA backend (Settings → GPU)
|
||||
- On Intel Arc: confirm IPEX detection in startup logs; restart the app after a driver update
|
||||
- On AMD Linux: check `HSA_OVERRIDE_GFX_VERSION` is set
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="'no kernel image is available' / 'CUDA error'">
|
||||
Almost always means the bundled PyTorch doesn't have kernels for your GPU's compute capability.
|
||||
|
||||
1. Update to Voicebox ≥ 0.4.0 (Blackwell support added there)
|
||||
2. Reinstall the CUDA backend
|
||||
3. If still broken, install PyTorch nightly via Remote Mode
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Out of memory (CUDA)">
|
||||
- Switch to a smaller model size (e.g. Qwen3 0.6B instead of 1.7B)
|
||||
- Use Settings → Models to unload other engines you're not using
|
||||
- Enable `low_cpu_mem_usage` is already on for CPU; for CUDA, the engine's `device_map` handles offload automatically
|
||||
- Close other GPU applications
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="MPS fallback errors on macOS">
|
||||
Some operations don't have a Metal implementation. Voicebox sets `PYTORCH_ENABLE_MPS_FALLBACK=1` for engines that need it (notably Kokoro), but if you launch from a custom env, set it manually:
|
||||
```bash
|
||||
export PYTORCH_ENABLE_MPS_FALLBACK=1
|
||||
```
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Generation works but is slow on my GPU">
|
||||
- Check Settings → GPU shows your GPU (not CPU)
|
||||
- Check VRAM usage — you may be paging to system memory
|
||||
- Try a smaller model
|
||||
- For NVIDIA: confirm cu128 is installed (Settings → GPU → version)
|
||||
</Accordion>
|
||||
</AccordionGroup>
|
||||
|
||||
## Next Steps
|
||||
|
||||
<Cards>
|
||||
<Card title="Remote Mode" href="/overview/remote-mode">
|
||||
Run the backend on a different machine with a stronger GPU
|
||||
</Card>
|
||||
<Card title="Model Management" href="/developer/model-management">
|
||||
Unload models to free GPU memory
|
||||
</Card>
|
||||
<Card title="Troubleshooting" href="/overview/troubleshooting">
|
||||
General troubleshooting beyond GPU
|
||||
</Card>
|
||||
</Cards>
|
||||
@@ -6,9 +6,7 @@
|
||||
"installation",
|
||||
"docker",
|
||||
"quick-start",
|
||||
"gpu-acceleration",
|
||||
"voice-cloning",
|
||||
"preset-voices",
|
||||
"stories-editor",
|
||||
"recording-transcription",
|
||||
"generation-history",
|
||||
|
||||
@@ -1,202 +0,0 @@
|
||||
---
|
||||
title: "Preset Voices"
|
||||
description: "Use built-in, ready-made voices without recording audio samples"
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Some Voicebox engines ship with a curated set of pre-built voices. Instead of cloning from your own audio sample, you pick a voice from a fixed catalog and the model speaks in that voice. No recording, no upload, no per-voice training required.
|
||||
|
||||
Two engines in 0.4 ship preset voices:
|
||||
|
||||
| Engine | Voices | Languages | Strengths |
|
||||
| --------------------- | ----------------------- | --------- | ------------------------------------------------------- |
|
||||
| **Kokoro 82M** | 50 | 9 | Tiny model, CPU-friendly, lowest VRAM of any engine |
|
||||
| **Qwen CustomVoice** | 9 (premium curated) | 4 | Natural-language style control over tone, emotion, pace |
|
||||
|
||||
<Callout type="info">
|
||||
Looking for cloning a specific person's voice instead? See [Voice Cloning](/overview/voice-cloning).
|
||||
</Callout>
|
||||
|
||||
## When to Use Preset Voices
|
||||
|
||||
<Cards>
|
||||
<Card title="No reference audio">
|
||||
You don't have (or don't want to provide) a recording of the target voice
|
||||
</Card>
|
||||
<Card title="Production reliability">
|
||||
Curated voices have predictable quality across any text input
|
||||
</Card>
|
||||
<Card title="Speed">
|
||||
Skip the audio cleanup, sample preparation, and quality iteration loop
|
||||
</Card>
|
||||
<Card title="Lightweight setup">
|
||||
Kokoro runs at CPU realtime with ~150 MB on disk — no GPU needed
|
||||
</Card>
|
||||
</Cards>
|
||||
|
||||
## Creating a Preset-Voice Profile
|
||||
|
||||
<Steps>
|
||||
<Step title="Open Profiles → New Profile">
|
||||
Same entry point as cloning profiles
|
||||
</Step>
|
||||
<Step title="Choose the engine">
|
||||
Select **Kokoro** or **Qwen CustomVoice** from the engine dropdown
|
||||
</Step>
|
||||
<Step title="Pick a preset voice">
|
||||
The voice catalog for the chosen engine appears — preview each by clicking it
|
||||
</Step>
|
||||
<Step title="Name and save">
|
||||
Give the profile a name. No audio sample needed — just save
|
||||
</Step>
|
||||
<Step title="Generate">
|
||||
Use the profile like any other in the floating generate box or the Generate page
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
<Callout type="info">
|
||||
Preset profiles are locked to their source engine — switching engines won't work since the voice exists only for that model. The profile grid greys out preset profiles when you switch to a different engine, and clicking one auto-switches the engine back to the right one.
|
||||
</Callout>
|
||||
|
||||
## Kokoro 82M — 50 Voices Across 9 Languages
|
||||
|
||||
Kokoro is the smallest engine in Voicebox at 82M parameters. It runs at CPU realtime with negligible VRAM, making it the best option for lightweight local inference. Voices are pre-built style vectors trained into the model — there's no concept of cloning here.
|
||||
|
||||
**Repository:** [`hexgrad/Kokoro-82M`](https://huggingface.co/hexgrad/Kokoro-82M) · Apache 2.0 licensed
|
||||
|
||||
### American English
|
||||
|
||||
| Female | Male |
|
||||
| ------- | ------- |
|
||||
| Alloy | Adam |
|
||||
| Aoede | Echo |
|
||||
| Bella | Eric |
|
||||
| Heart | Fenrir |
|
||||
| Jessica | Liam |
|
||||
| Kore | Michael |
|
||||
| Nicole | Onyx |
|
||||
| Nova | Puck |
|
||||
| River | Santa |
|
||||
| Sarah | |
|
||||
| Sky | |
|
||||
|
||||
### British English
|
||||
|
||||
| Female | Male |
|
||||
| -------- | ------ |
|
||||
| Alice | Daniel |
|
||||
| Emma | Fable |
|
||||
| Isabella | George |
|
||||
| Lily | Lewis |
|
||||
|
||||
### Other Languages
|
||||
|
||||
| Language | Voices |
|
||||
| ----------------- | ------------------------------------------- |
|
||||
| Spanish (`es`) | Dora (f), Alex (m), Santa (m) |
|
||||
| French (`fr`) | Siwis (f) |
|
||||
| Hindi (`hi`) | Alpha (f), Beta (f), Omega (m), Psi (m) |
|
||||
| Italian (`it`) | Sara (f), Nicola (m) |
|
||||
| Japanese (`ja`) | Alpha (f), Gongitsune (f), Nezumi (f), Tebukuro (f), Kumo (m) |
|
||||
| Portuguese (`pt`) | Dora (f), Alex (m), Santa (m) |
|
||||
| Chinese (`zh`) | Xiaobei (f), Xiaoni (f), Xiaoxiao (f), Xiaoyi (f) |
|
||||
|
||||
### Kokoro at a Glance
|
||||
|
||||
| Property | Value |
|
||||
| --------------- | -------------------------------------------- |
|
||||
| Parameters | 82M |
|
||||
| Sample rate | 24 kHz |
|
||||
| VRAM | ~150 MB (negligible on CPU) |
|
||||
| Speed | Realtime on CPU, faster on GPU |
|
||||
| Instruct | Not supported (preset voice carries the style) |
|
||||
| License | Apache 2.0 |
|
||||
|
||||
## Qwen CustomVoice — 9 Premium Voices with Instruct Control
|
||||
|
||||
Qwen CustomVoice ships with 9 curated speakers and supports **natural-language style control** — you tell the model how to deliver the line ("speak slowly with warmth", "authoritative and clear") and it adapts tone, emotion, and pace.
|
||||
|
||||
Two model sizes:
|
||||
- **1.7B** — full quality, recommended default
|
||||
- **0.6B** — lighter, faster, lower-end hardware
|
||||
|
||||
**Repository:** [`Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice`](https://huggingface.co/Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice) (and 0.6B variant) · by Alibaba
|
||||
|
||||
### Voice Catalog
|
||||
|
||||
| Speaker | Gender | Language | Description |
|
||||
| --------- | ------ | -------- | ------------------------------------------------------------ |
|
||||
| Vivian | female | Chinese | Bright, slightly edgy young female voice |
|
||||
| Serena | female | Chinese | Warm, gentle young female voice |
|
||||
| Uncle Fu | male | Chinese | Seasoned male voice with a low, mellow timbre |
|
||||
| Dylan | male | Chinese | Youthful Beijing male voice with a clear, natural timbre |
|
||||
| Eric | male | Chinese | Lively Chengdu male voice with a slightly husky brightness |
|
||||
| Ryan | male | English | Dynamic male voice with strong rhythmic drive (default) |
|
||||
| Aiden | male | English | Sunny American male voice with a clear midrange |
|
||||
| Ono Anna | female | Japanese | Playful Japanese female voice with a light, nimble timbre |
|
||||
| Sohee | female | Korean | Warm Korean female voice with rich emotion |
|
||||
|
||||
### Using Instruct Mode
|
||||
|
||||
In the floating generate box, switch to a Qwen CustomVoice profile and click the **delivery instructions** toggle (slider icon, left of the generate button). A second textarea appears below the main text:
|
||||
|
||||
- Main text → what you want the voice to say
|
||||
- Instruct text → how you want it delivered
|
||||
|
||||
Examples of effective instruct prompts:
|
||||
|
||||
```
|
||||
Speak slowly with emphasis, like reading bedtime stories
|
||||
Warm and friendly, conversational tone
|
||||
Professional and authoritative, broadcast quality
|
||||
Whisper, intimate and close
|
||||
Excited and energetic, like sports commentary
|
||||
```
|
||||
|
||||
The full Generate page also surfaces the instruct field as a separate input.
|
||||
|
||||
### Qwen CustomVoice at a Glance
|
||||
|
||||
| Property | Value |
|
||||
| --------------- | -------------------------------------------------- |
|
||||
| Parameters | 1.7B / 0.6B |
|
||||
| Languages | Chinese, English, Japanese, Korean (10 supported) |
|
||||
| Voices | 9 curated preset speakers |
|
||||
| VRAM | ~3.5 GB (1.7B), ~1.2 GB (0.6B) |
|
||||
| Instruct | Yes — natural-language style control |
|
||||
| Cloning | No — paired Base Qwen3-TTS engine handles cloning |
|
||||
|
||||
## Cloning vs Preset — Quick Decision
|
||||
|
||||
| You want… | Use |
|
||||
| -------------------------------------------------- | ----------------------------------------- |
|
||||
| To replicate a specific person's voice | [Voice Cloning](/overview/voice-cloning) |
|
||||
| Production-ready voices with no audio prep | Kokoro or Qwen CustomVoice |
|
||||
| The smallest possible footprint (CPU-only) | Kokoro |
|
||||
| Fine control over delivery (tone, pace, emotion) | Qwen CustomVoice |
|
||||
| The broadest language coverage | [Voice Cloning](/overview/voice-cloning) via Chatterbox Multilingual (23 langs) |
|
||||
|
||||
## Limitations
|
||||
|
||||
<Callout type="warn">
|
||||
Preset voices are fixed — you can't fine-tune or modify the underlying voice. If you want a specific voice that isn't in the catalog, use a cloning engine and provide a reference sample.
|
||||
</Callout>
|
||||
|
||||
- Preset voices can't be exported to use in other Voicebox installations as audio (only as profile metadata pointing to the same engine + voice ID)
|
||||
- The Kokoro voice catalog is set by the upstream model — new voices appear only when hexgrad publishes new model releases
|
||||
- Qwen CustomVoice's 9 speakers are part of the model checkpoint — same constraint
|
||||
|
||||
## Next Steps
|
||||
|
||||
<Cards>
|
||||
<Card title="Voice Cloning" href="/overview/voice-cloning">
|
||||
Clone a specific voice from your own audio
|
||||
</Card>
|
||||
<Card title="Generate Speech" href="/overview/generating-speech">
|
||||
Use a profile to generate audio
|
||||
</Card>
|
||||
<Card title="Build Stories" href="/overview/building-stories">
|
||||
Compose multi-voice narratives
|
||||
</Card>
|
||||
</Cards>
|
||||
@@ -1,25 +1,11 @@
|
||||
---
|
||||
title: "Voice Cloning"
|
||||
description: "Clone any voice from a few seconds of reference audio"
|
||||
description: "Clone any voice from just a few seconds of audio"
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Voicebox can replicate a specific person's voice from a short audio sample — known as **zero-shot voice cloning**. You provide 10-30 seconds of clear speech, the model extracts a voice embedding, and from then on you can generate any text in that voice.
|
||||
|
||||
Five engines in 0.4 support cloning:
|
||||
|
||||
| Engine | Languages | Strengths |
|
||||
| --------------------------- | --------- | -------------------------------------------------------------------------- |
|
||||
| **Qwen3-TTS** (0.6B / 1.7B) | 10 | High-quality multilingual, supports delivery instructions on the same kwarg |
|
||||
| **Chatterbox Multilingual** | 23 | Broadest language coverage — Arabic, Hindi, Swahili, Hebrew, more |
|
||||
| **Chatterbox Turbo** | English | Fast 350M model with paralinguistic emotion tags (`[laugh]`, `[sigh]`) |
|
||||
| **LuxTTS** | English | Lightweight (~1 GB VRAM), 48 kHz output, 150x realtime on CPU |
|
||||
| **TADA** (1B / 3B) | 10 | Speech-language model with 700s+ coherent long-form generation |
|
||||
|
||||
<Callout type="info">
|
||||
Don't want to record audio? Use a curated voice from Kokoro or Qwen CustomVoice instead — see [Preset Voices](/overview/preset-voices).
|
||||
</Callout>
|
||||
Voicebox uses **Qwen3-TTS** from Alibaba to achieve near-perfect voice cloning from just a few seconds of audio. The model captures prosody, emotion, and natural cadence.
|
||||
|
||||
## How It Works
|
||||
|
||||
@@ -27,30 +13,17 @@ Five engines in 0.4 support cloning:
|
||||
<Step title="Upload or Record Sample">
|
||||
Provide 10-30 seconds of clear speech from the target voice
|
||||
</Step>
|
||||
<Step title="Engine Analysis">
|
||||
The selected engine analyzes vocal characteristics, tone, and speaking patterns
|
||||
<Step title="Model Analysis">
|
||||
Qwen3-TTS analyzes vocal characteristics, tone, and speaking patterns
|
||||
</Step>
|
||||
<Step title="Voice Profile Created">
|
||||
A voice embedding is generated and stored with your profile
|
||||
The model generates a voice embedding for synthesis
|
||||
</Step>
|
||||
<Step title="Generate Speech">
|
||||
Use the profile to generate any text in the cloned voice
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
## Choosing an Engine for Cloning
|
||||
|
||||
Different engines suit different use cases. The profile grid greys out unsupported engines so you can switch easily.
|
||||
|
||||
| If you want… | Pick |
|
||||
| -------------------------------------------------- | --------------------- |
|
||||
| Best overall quality on a few common languages | **Qwen3-TTS 1.7B** |
|
||||
| Faster generation, slightly lower quality | **Qwen3-TTS 0.6B** |
|
||||
| Languages outside Qwen's 10 (Arabic, Hindi, etc.) | **Chatterbox Multilingual** |
|
||||
| Expressive English with `[laugh]` `[sigh]` tags | **Chatterbox Turbo** |
|
||||
| CPU-only or GPU-light setup, English | **LuxTTS** |
|
||||
| Long-form generation (audiobooks, full chapters) | **TADA 3B** |
|
||||
|
||||
## Best Practices
|
||||
|
||||
### Sample Quality
|
||||
@@ -79,40 +52,24 @@ Adding multiple samples from the same speaker can improve quality:
|
||||
- Different recording conditions
|
||||
|
||||
<Callout type="info">
|
||||
The model will learn a more robust representation from diverse samples. Especially helpful for distinctive voices the model might otherwise smooth over.
|
||||
The model will learn a more robust representation from diverse samples.
|
||||
</Callout>
|
||||
|
||||
## Supported Languages by Engine
|
||||
## Supported Languages
|
||||
|
||||
- **Qwen3-TTS** — English, Chinese, Japanese, Korean, German, French, Russian, Portuguese, Spanish, Italian (10)
|
||||
- **Chatterbox Multilingual** — Arabic, Chinese, Danish, Dutch, English, Finnish, French, German, Greek, Hebrew, Hindi, Italian, Japanese, Korean, Malay, Norwegian, Polish, Portuguese, Russian, Spanish, Swahili, Swedish, Turkish (23)
|
||||
- **Chatterbox Turbo** — English
|
||||
- **LuxTTS** — English
|
||||
- **TADA 3B** — 10 multilingual; **TADA 1B** — English
|
||||
Currently supported:
|
||||
- English
|
||||
- Chinese (Mandarin)
|
||||
|
||||
For complete language tables and engine-specific notes, see the [TTS Engines developer guide](/developer/tts-engines).
|
||||
More languages coming soon.
|
||||
|
||||
## Limitations
|
||||
|
||||
<Callout type="warn">
|
||||
Voice cloning should only be used with consent. Ensure you have permission to clone someone's voice. See the project's [SECURITY.md](https://github.com/jamiepine/voicebox/blob/main/SECURITY.md) and your local laws on synthetic voice content.
|
||||
Voice cloning should only be used with consent. Ensure you have permission to clone someone's voice.
|
||||
</Callout>
|
||||
|
||||
- Quality depends on sample clarity — noisy samples produce noisy clones
|
||||
- Works best with consistent speaking tone within a sample
|
||||
- Quality depends on sample clarity
|
||||
- Works best with consistent speaking tone
|
||||
- May struggle with extreme accents or speech impediments
|
||||
- Background noise reduces quality and can introduce artifacts
|
||||
|
||||
## Next Steps
|
||||
|
||||
<Cards>
|
||||
<Card title="Creating Voice Profiles" href="/overview/creating-voice-profiles">
|
||||
Step-by-step guide to creating profiles
|
||||
</Card>
|
||||
<Card title="Preset Voices" href="/overview/preset-voices">
|
||||
Use built-in voices instead of cloning
|
||||
</Card>
|
||||
<Card title="Generating Speech" href="/overview/generating-speech">
|
||||
Use a profile to generate audio
|
||||
</Card>
|
||||
</Cards>
|
||||
- Background noise reduces quality
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
'use client';
|
||||
|
||||
import { Github, Globe, Languages, MessageSquare, SlidersHorizontal, Zap } from 'lucide-react';
|
||||
import { Github, Globe, Languages, MessageSquare, Zap } from 'lucide-react';
|
||||
import { useEffect, useState } from 'react';
|
||||
import { ControlUI } from '@/components/ControlUI';
|
||||
import { Features } from '@/components/Features';
|
||||
@@ -236,101 +236,6 @@ export default function Home() {
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Qwen CustomVoice */}
|
||||
<div className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-6 transition-colors hover:border-accent/30">
|
||||
<div className="flex items-start justify-between mb-3">
|
||||
<div>
|
||||
<h3 className="text-base font-semibold text-foreground">Qwen CustomVoice</h3>
|
||||
<span className="text-xs text-muted-foreground/60">by Alibaba</span>
|
||||
</div>
|
||||
<div className="flex gap-1.5">
|
||||
<span className="text-[10px] px-2 py-0.5 rounded-full border border-border bg-background text-muted-foreground">
|
||||
1.7B
|
||||
</span>
|
||||
<span className="text-[10px] px-2 py-0.5 rounded-full border border-border bg-background text-muted-foreground">
|
||||
0.6B
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
<p className="text-sm text-muted-foreground leading-relaxed mb-4">
|
||||
Nine premium preset speakers with natural-language style control. Tell the model how
|
||||
to deliver — "speak slowly with warmth", "authoritative and clear" — and it adapts
|
||||
tone, emotion, and pace.
|
||||
</p>
|
||||
<div className="flex flex-wrap gap-2">
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
<SlidersHorizontal className="h-3 w-3" />
|
||||
Instruct control
|
||||
</span>
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
<Globe className="h-3 w-3" />
|
||||
10 languages
|
||||
</span>
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
9 preset voices
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* HumeAI TADA */}
|
||||
<div className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-6 transition-colors hover:border-accent/30">
|
||||
<div className="flex items-start justify-between mb-3">
|
||||
<div>
|
||||
<h3 className="text-base font-semibold text-foreground">TADA</h3>
|
||||
<span className="text-xs text-muted-foreground/60">by Hume AI</span>
|
||||
</div>
|
||||
<div className="flex gap-1.5">
|
||||
<span className="text-[10px] px-2 py-0.5 rounded-full border border-border bg-background text-muted-foreground">
|
||||
3B
|
||||
</span>
|
||||
<span className="text-[10px] px-2 py-0.5 rounded-full border border-border bg-background text-muted-foreground">
|
||||
1B
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
<p className="text-sm text-muted-foreground leading-relaxed mb-4">
|
||||
Speech-language model with text-acoustic dual alignment. Built for long-form
|
||||
generation — produces 700s+ of coherent audio without drift. Multilingual at 3B,
|
||||
English-focused at 1B.
|
||||
</p>
|
||||
<div className="flex flex-wrap gap-2">
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
<Globe className="h-3 w-3" />
|
||||
10 languages
|
||||
</span>
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
Long-form coherent
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Kokoro 82M */}
|
||||
<div className="rounded-xl border border-border bg-card/60 backdrop-blur-sm p-6 transition-colors hover:border-accent/30">
|
||||
<div className="flex items-start justify-between mb-3">
|
||||
<div>
|
||||
<h3 className="text-base font-semibold text-foreground">Kokoro</h3>
|
||||
<span className="text-xs text-muted-foreground/60">by hexgrad · Apache 2.0</span>
|
||||
</div>
|
||||
<span className="text-[10px] px-2 py-0.5 rounded-full border border-border bg-background text-muted-foreground">
|
||||
82M
|
||||
</span>
|
||||
</div>
|
||||
<p className="text-sm text-muted-foreground leading-relaxed mb-4">
|
||||
Tiny 82M-parameter TTS that runs at CPU realtime with negligible VRAM. Pre-built
|
||||
voice styles instead of cloning — pick a voice, type, generate. Smallest footprint
|
||||
of any engine.
|
||||
</p>
|
||||
<div className="flex flex-wrap gap-2">
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
<Zap className="h-3 w-3" />
|
||||
CPU realtime
|
||||
</span>
|
||||
<span className="flex items-center gap-1 text-[11px] text-muted-foreground/70">
|
||||
Preset voices
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import { Coffee } from 'lucide-react';
|
||||
import Image from 'next/image';
|
||||
import Link from 'next/link';
|
||||
import { DONATE_URL, GITHUB_REPO } from '@/lib/constants';
|
||||
import { GITHUB_REPO } from '@/lib/constants';
|
||||
|
||||
export function Footer() {
|
||||
return (
|
||||
@@ -20,19 +19,9 @@ export function Footer() {
|
||||
/>
|
||||
<span className="text-sm font-semibold">Voicebox</span>
|
||||
</div>
|
||||
<p className="text-sm text-muted-foreground leading-relaxed mb-4">
|
||||
<p className="text-sm text-muted-foreground leading-relaxed">
|
||||
Open source voice cloning studio. Local-first, free forever.
|
||||
</p>
|
||||
<a
|
||||
href={DONATE_URL}
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="inline-flex items-center gap-2 rounded-lg border border-border/60 bg-card/60 px-3 py-2 text-sm text-muted-foreground transition-colors hover:text-foreground hover:border-[#FFDD00]/40"
|
||||
aria-label="Donate via Buy Me a Coffee"
|
||||
>
|
||||
<Coffee className="h-4 w-4 text-[#FFDD00]" />
|
||||
<span className="text-[13px] font-medium">Donate</span>
|
||||
</a>
|
||||
</div>
|
||||
|
||||
{/* Product */}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
'use client';
|
||||
|
||||
import { Coffee, Github } from 'lucide-react';
|
||||
import { Github } from 'lucide-react';
|
||||
import Image from 'next/image';
|
||||
import { useEffect, useState } from 'react';
|
||||
import { DONATE_URL, GITHUB_REPO } from '@/lib/constants';
|
||||
import { GITHUB_REPO } from '@/lib/constants';
|
||||
|
||||
function formatStarCount(count: number): string {
|
||||
if (count >= 1000) {
|
||||
@@ -75,33 +75,21 @@ export function Navbar() {
|
||||
</a>
|
||||
</div>
|
||||
|
||||
{/* Donate + GitHub star buttons */}
|
||||
<div className="flex items-center gap-2 justify-self-end">
|
||||
<a
|
||||
href={DONATE_URL}
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="hidden sm:flex items-center gap-2 rounded-lg border border-border/60 bg-card/60 px-3 py-1.5 text-sm text-muted-foreground transition-colors hover:text-foreground hover:border-[#FFDD00]/40"
|
||||
aria-label="Donate via Buy Me a Coffee"
|
||||
>
|
||||
<Coffee className="h-4 w-4 text-[#FFDD00]" />
|
||||
<span className="text-[13px] font-medium">Donate</span>
|
||||
</a>
|
||||
<a
|
||||
href={GITHUB_REPO}
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="flex items-center gap-2 rounded-lg border border-border/60 bg-card/60 px-3 py-1.5 text-sm text-muted-foreground transition-colors hover:text-foreground hover:border-border"
|
||||
>
|
||||
<Github className="h-4 w-4" />
|
||||
<span className="text-[13px] font-medium">Star</span>
|
||||
{starCount !== null && (
|
||||
<span className="border-l border-border/60 pl-2 text-[13px] font-semibold text-foreground">
|
||||
{formatStarCount(starCount)}
|
||||
</span>
|
||||
)}
|
||||
</a>
|
||||
</div>
|
||||
{/* GitHub star button */}
|
||||
<a
|
||||
href={GITHUB_REPO}
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="flex items-center gap-2 justify-self-end rounded-lg border border-border/60 bg-card/60 px-3 py-1.5 text-sm text-muted-foreground transition-colors hover:text-foreground hover:border-border"
|
||||
>
|
||||
<Github className="h-4 w-4" />
|
||||
<span className="text-[13px] font-medium">Star</span>
|
||||
{starCount !== null && (
|
||||
<span className="border-l border-border/60 pl-2 text-[13px] font-semibold text-foreground">
|
||||
{formatStarCount(starCount)}
|
||||
</span>
|
||||
)}
|
||||
</a>
|
||||
</div>
|
||||
</nav>
|
||||
);
|
||||
|
||||
@@ -4,7 +4,6 @@ export const LATEST_VERSION = 'v0.1.0';
|
||||
|
||||
export const GITHUB_REPO = 'https://github.com/jamiepine/voicebox';
|
||||
export const GITHUB_RELEASES_PAGE = `${GITHUB_REPO}/releases`;
|
||||
export const DONATE_URL = 'https://buymeacoffee.com/jamiepine';
|
||||
|
||||
export const DOWNLOAD_LINKS = {
|
||||
macArm: GITHUB_RELEASES_PAGE,
|
||||
|
||||
Generated
+1
-1
@@ -5041,7 +5041,7 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||
|
||||
[[package]]
|
||||
name = "voicebox"
|
||||
version = "0.4.0"
|
||||
version = "0.3.1"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"core-foundation-sys",
|
||||
|
||||
Reference in New Issue
Block a user