[build-system] requires = ["hatchling"] build-backend = "hatchling.build" [project] name = "omnivoice" version = "0.2.7" description = "OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models" readme = "README.md" license = "Apache-2.0" requires-python = ">=3.11" authors = [{name = "Han Zhu"}] keywords = [ "tts", "text-to-speech", "speech-synthesis", "zero-shot", "multilingual", "diffusion", "voice-cloning", ] classifiers = [ "Intended Audience :: Science/Research", "Intended Audience :: Developers", "Topic :: Scientific/Engineering :: Artificial Intelligence", "Topic :: Multimedia :: Sound/Audio :: Speech", "Operating System :: OS Independent", "Programming Language :: Python :: 3", ] dependencies = [ "torch>=2.4", "torchaudio>=2.4", "transformers>=5.3.0", "accelerate", "pydub", "gradio", "tensorboardX", "webdataset", "numpy", "soundfile", # pkg_resources is used by whisperx/faster-whisper at runtime. # On Python 3.12+ it's no longer bundled by default — pin a modern # setuptools to guarantee it's present (fixes #58). "setuptools>=75.0", "psutil>=7.2.2", # Pinned to 3.x — pyannote 4.x removed `use_auth_token` from `Inference` # which whisperx 3.4.2 still passes, blowing up `whisperx.load_model()` # with TypeError. whisperx tests against pyannote 3.3.2+, so we track # that range and revisit when whisperx releases a 4-compatible build. "pyannote-audio>=3.3.2,<4.0", "pyinstaller>=6.19.0", "imageio-ffmpeg>=0.6.0", "pedalboard>=0.9.14", # Primary ASR — cross-platform, CTranslate2-based under the hood. WhisperX # adds wav2vec2 forced alignment (±10-30 ms word timing vs Whisper's own # ±100-300 ms) which directly improves lip-sync on the dub pipeline. # Pulls `faster-whisper` transitively, so a WhisperX install also provides # the plain faster-whisper backend as a fallback for rare-language audio # where no wav2vec2 alignment model exists. "whisperx>=3.1.0", "faster-whisper>=1.0.0", # Apple Silicon-only speedup; skipped everywhere else so `uv sync` can # succeed on Linux/Windows/mac-Intel (no mlx wheels exist for those). "mlx-whisper>=0.2.1 ; sys_platform == 'darwin' and platform_machine == 'arm64'", # Apple Silicon-only rich TTS library — 14+ engines (Kokoro, CSM, Dia, # Qwen3-TTS, Chatterbox, MeloTTS, OuteTTS, Spark, Higgs-Audio, Voxtral, # …). Gives mac-ARM users a broad engine picker in Settings. Also gated # by platform markers because it depends on mlx (Apple Silicon only). "mlx-audio>=0.3.0 ; sys_platform == 'darwin' and platform_machine == 'arm64'", "demucs>=4.0.1", "yt-dlp>=2024.12.13", "alembic>=1.13", # Lightweight English TTS "Turbo" tier — 25-80 MB ONNX model, 8 preset # voices (Bella, Jasper, Luna, Bruno, Rosie, Hugo, Kiki, Leo), CPU # realtime on any platform. Complements OmniVoice's 2.4 GB multilingual # zero-shot clone: when the caller just needs fast English narration with # no reference sample, this is ~100× smaller + ~10× faster. Pinned to # the exact wheel because the project is in developer preview and the # 0.x API is explicitly unstable. "kittentts @ https://github.com/KittenML/KittenTTS/releases/download/0.8.1/kittentts-0.8.1-py3-none-any.whl", # Invisible audio watermarking — embeds imperceptible neural watermarks # in AI-generated speech for provenance detection (SynthID-like). # MIT license, ~5ms per segment on CPU, 16-bit message payload. "audioseal>=0.1.3", # API server — always needed for the Studio UI. "fastapi", "scalar-fastapi", "uvicorn", "python-multipart", # Required by uvicorn for WebSocket support (real-time sidebar events). "websockets", # Offline translation — listed as builtin in the engine registry so the # "Argos (Local, Fast)" option in the Dub tab works out-of-the-box. "argostranslate>=1.9.0", # Phase 1 AUTH-02: Fernet symmetric encryption + scrypt KDF for the # at-rest HF token in the SQLite settings store. Pulled in directly so # we don't depend on a transitive arrival via pyannote/huggingface_hub # (Assumption A1 in RESEARCH.md was checked at execute-time and proved # false — `cryptography` is not on the install path today). "cryptography>=41", ] [project.optional-dependencies] eval = [ "jiwer==3.1.0", # WER "librosa", # Audio processing "s3prl", # Speech representation (HuBERT etc.) "funasr", # ASR models "zhconv", # Chinese character normalization "zhon", # Chinese punctuation "unidecode", # Unicode normalization ] ui = [ "gradio", "gradio_client", "requests", ] # Phase 3 Plan 03-01 — Supertonic-3 opt-in engine. CPU-only ONNX TTS, # 31 languages, ~99M params, ~400 MB model on first use. Default # `uv sync --no-dev` does NOT install this; users opt in with # `uv sync --extra supertonic` after accepting the OpenRAIL-M model # license in Settings → Engines. # # Publisher verified per Plan 03-01 Task 1 (Package Legitimacy Audit): # • PyPI maintainers = Yu Yechan / Juheon Lee / Hyeongju Kim (Supertone Inc.) # • Repository = github.com/supertone-inc/supertonic-py # • Same publisher ships supertonic-js on npm (same maintainer email) # • Wheel inspected: pure-Python, no postinstall scripts, no subprocess/exec # at module top level. ``supertonic.config.MODEL_CONFIGS["supertonic-3"]`` # itself pins the HF model revision by SHA — we re-pin to the same # SHA in backend/engines/supertonic3/constants.py for TTS-03. supertonic = [ "supertonic==1.3.1", ] [project.scripts] omnivoice-infer = "omnivoice.cli.infer:main" omnivoice-infer-batch = "omnivoice.cli.infer_batch:main" omnivoice-demo = "omnivoice.cli.demo:main" omnivoice-dub = "omnivoice.cli.dub:main" [project.urls] Homepage = "https://github.com/k2-fsa/OmniVoice" Repository = "https://github.com/k2-fsa/OmniVoice" "Bug Tracker" = "https://github.com/k2-fsa/OmniVoice/issues" [tool.uv.sources] # Install PyTorch with CUDA support on Linux/Windows (CUDA doesn't exist for Mac). # NOTE: We must explicitly request them as `dependencies` above. These improved # versions will not be selected if they're only third-party dependencies. torch = [ { index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" }, ] torchaudio = [ { index = "pytorch-cuda", marker = "sys_platform == 'linux' or sys_platform == 'win32'" }, ] [[tool.uv.index]] name = "pytorch-cuda" # Use PyTorch built for NVIDIA Toolkit version 12.8. # Available versions: https://pytorch.org/get-started/locally/ url = "https://download.pytorch.org/whl/cu128" # Only use this index when explicitly requested by `tool.uv.sources`. explicit = true [tool.uv] constraint-dependencies = [ "torch==2.8.0", "torchaudio==2.8.0", ] [tool.hatch.metadata] # Needed so the KittenTTS wheel-URL dep in `project.dependencies` is accepted # by hatchling's metadata validator. KittenTTS isn't on PyPI (dev preview), # so pulling it via GH Releases URL is the only option today. allow-direct-references = true [tool.hatch.build.targets.sdist] include = ["omnivoice"] [tool.hatch.build.targets.wheel] packages = ["omnivoice"] [dependency-groups] dev = [ "httpx>=0.28.1", "pytest>=9.0.3", "pytest-asyncio>=1.3.0", "pytest-cov>=6.0", ] [tool.pytest.ini_options] # Bare `pytest` would otherwise walk into `research/` (1.2 GB of vendored # upstream projects, each with its own test_*.py that calls sys.exit at # module level) and INTERNALERROR. `backend/tests/` runs separately because # it stubs `core.config` in sys.modules, which pollutes import state for # other modules in the same session. testpaths = ["tests"] norecursedirs = [ "research", "omnivoice/training", "omnivoice/eval", "frontend", "deploy", ".venv", "node_modules", "omnivoice_data", "backend/omnivoice_data", ]