Upload 17 files
Browse files- .gitattributes +84 -15
- .gitignore +0 -0
- .gitmodules +4 -0
- .pre-commit-config.yaml +9 -0
- .python-version +1 -0
- CLAUDE.md +156 -0
- CONTRIBUTING.md +4 -0
- Dockerfile +47 -0
- MULTILINGUAL_CHANGES.md +68 -0
- README.md +167 -0
- conf.yaml +433 -0
- local_tools.py +16 -0
- mcp_servers.json +6 -0
- model_dict.json +26 -0
- pyproject.toml +68 -0
- run_server.py +178 -0
- upgrade.py +171 -0
.gitattributes
CHANGED
|
@@ -1,15 +1,84 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
live2d-models/
|
| 9 |
-
live2d-models/
|
| 10 |
-
live2d-models/
|
| 11 |
-
live2d-models/
|
| 12 |
-
live2d-models/
|
| 13 |
-
live2d-models/
|
| 14 |
-
live2d-models/
|
| 15 |
-
live2d-models/
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
static/libs/* linguist-vendoredfrontend/libs/ort-wasm-simd-threaded.wasm filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
frontend/libs/ort-wasm-simd.wasm filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
frontend/libs/ort-wasm-threaded.wasm filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
frontend/libs/ort-wasm.wasm filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
frontend/libs/silero_vad_legacy.onnx filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
frontend/libs/silero_vad_v5.onnx filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
frontend/libs/ort-wasm-simd-threaded.wasm filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
live2d-models/mao_pro/runtime/mao_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
live2d-models/mao_pro/runtime/mao_pro.moc3 filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
live2d-models/shizuku/runtime/shizuku.1024/texture_00.png filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
live2d-models/shizuku/runtime/shizuku.1024/texture_01.png filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
live2d-models/shizuku/runtime/shizuku.1024/texture_02.png filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
live2d-models/shizuku/runtime/shizuku.1024/texture_03.png filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
live2d-models/shizuku/runtime/shizuku.1024/texture_04.png filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
live2d-models/shizuku/runtime/shizuku.moc3 filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
backgrounds/cartoon-night-landscape-moon.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
backgrounds/cityscape.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
backgrounds/computer-room-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
backgrounds/congress.jpg filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
backgrounds/field-night-painting-moon.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
backgrounds/lernado-diff-classroom-center.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
backgrounds/moon-over-mountain.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
backgrounds/mountain-range-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
backgrounds/night-landscape-grass-moon.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
backgrounds/night-scene-cartoon-moon.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
backgrounds/painting-valley-night-sky.[[:space:]]2.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
backgrounds/room-interior-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
backgrounds/sdxl-classroom-door-view.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
avatars/mao.png filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
avatars/shizuku.png filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
frontend/music/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
frontend/music/eve.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
frontend/music/golden.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
frontend/music/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
frontend/music/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
frontend/music/the_awakening.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
frontend/music/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
frontend/music/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
frontend/music/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_01.png filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_02.png filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_03.png filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.moc3 filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
ceiling-window-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_01.png filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_02.png filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_03.png filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.moc3 filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
avatars/Yue_001.png filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
backgrounds/ceiling-window-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
frontend/music/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
sing/original/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
sing/original/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
sing/original/eve.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
sing/original/golden.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
sing/original/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
sing/original/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
sing/original/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
sing/original/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
sing/original/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
sing/tracks/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
sing/tracks/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
sing/tracks/eve.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
sing/tracks/golden.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
sing/tracks/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
sing/tracks/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 69 |
+
sing/tracks/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 70 |
+
sing/tracks/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 71 |
+
sing/tracks/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 72 |
+
live2d-models/haru_greeter_pro_jp/haru_Ä=òtâXü\[âc_âCâôâ_ü\[âg_t07.psd filter=lfs diff=lfs merge=lfs -text
|
| 73 |
+
live2d-models/haru_greeter_pro_jp/haru_Ä=òtâXü\[âc_æfì¦ò¬é»_t07.psd filter=lfs diff=lfs merge=lfs -text
|
| 74 |
+
live2d-models/haru_greeter_pro_jp/haru_greeter_t03.can3 filter=lfs diff=lfs merge=lfs -text
|
| 75 |
+
live2d-models/haru_greeter_pro_jp/haru_greeter_t05.cmo3 filter=lfs diff=lfs merge=lfs -text
|
| 76 |
+
live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_00.png filter=lfs diff=lfs merge=lfs -text
|
| 77 |
+
live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_01.png filter=lfs diff=lfs merge=lfs -text
|
| 78 |
+
live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.moc3 filter=lfs diff=lfs merge=lfs -text
|
| 79 |
+
backgrounds/ceiling-computer-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
|
| 80 |
+
backgrounds/ceiling-computer-room-night.jpg filter=lfs diff=lfs merge=lfs -text
|
| 81 |
+
backgrounds/ceiling-window-room-night.jpeg.jpg filter=lfs diff=lfs merge=lfs -text
|
| 82 |
+
sing/tracks/Light.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 83 |
+
sing/tracks/take_a_slice.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 84 |
+
sing/tracks/Washing_Machine_Hear.mp3 filter=lfs diff=lfs merge=lfs -text
|
.gitignore
ADDED
|
Binary file (102 Bytes). View file
|
|
|
.gitmodules
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[submodule "frontend"]
|
| 2 |
+
path = frontend
|
| 3 |
+
url = https://github.com/Open-LLM-VTuber/Open-LLM-VTuber-Web
|
| 4 |
+
branch = build
|
.pre-commit-config.yaml
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
repos:
|
| 2 |
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
| 3 |
+
rev: v0.9.6
|
| 4 |
+
hooks:
|
| 5 |
+
- id: ruff
|
| 6 |
+
args: [--fix, --exit-non-zero-on-fix]
|
| 7 |
+
- id: ruff-format
|
| 8 |
+
|
| 9 |
+
|
.python-version
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
3.10
|
CLAUDE.md
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# CLAUDE.md
|
| 2 |
+
|
| 3 |
+
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
| 4 |
+
|
| 5 |
+
## Project Overview
|
| 6 |
+
|
| 7 |
+
Open-LLM-VTuber is a voice-interactive AI companion with Live2D avatar support that runs completely offline. It's a cross-platform Python application supporting real-time voice conversations, visual perception, and Live2D character animations. The project features modular architecture for LLM, ASR (Automatic Speech Recognition), TTS (Text-to-Speech), and other components.
|
| 8 |
+
|
| 9 |
+
## Essential Commands
|
| 10 |
+
|
| 11 |
+
### Development Setup
|
| 12 |
+
- **Install dependencies**: `uv sync` (uses uv package manager)
|
| 13 |
+
- **Run server**: `uv run run_server.py`
|
| 14 |
+
- **Run with verbose logging**: `uv run run_server.py --verbose`
|
| 15 |
+
- **Update project**: `uv run upgrade.py`
|
| 16 |
+
|
| 17 |
+
### Code Quality
|
| 18 |
+
- **Lint code**: `ruff check .`
|
| 19 |
+
- **Format code**: `ruff format .`
|
| 20 |
+
- **Run pre-commit hooks**: `pre-commit run --all-files`
|
| 21 |
+
|
| 22 |
+
### Server Configuration
|
| 23 |
+
- **Main config file**: `conf.yaml` (user configuration)
|
| 24 |
+
- **Default configs**: `config_templates/conf.default.yaml` and `config_templates/conf.ZH.default.yaml`
|
| 25 |
+
- **Character configs**: `characters/` directory (YAML files)
|
| 26 |
+
|
| 27 |
+
## Architecture Overview
|
| 28 |
+
|
| 29 |
+
### Core Components
|
| 30 |
+
|
| 31 |
+
**WebSocket Server** (`src/open_llm_vtuber/server.py`):
|
| 32 |
+
- FastAPI-based server handling WebSocket connections
|
| 33 |
+
- Serves frontend, Live2D models, and static assets
|
| 34 |
+
- Supports both main client and proxy WebSocket endpoints
|
| 35 |
+
|
| 36 |
+
**Service Context** (`src/open_llm_vtuber/service_context.py`):
|
| 37 |
+
- Central dependency injection container
|
| 38 |
+
- Manages all engines (LLM, ASR, TTS, VAD, etc.)
|
| 39 |
+
- Each WebSocket connection gets its own service context instance
|
| 40 |
+
|
| 41 |
+
**WebSocket Handler** (`src/open_llm_vtuber/websocket_handler.py`):
|
| 42 |
+
- Routes WebSocket messages to appropriate handlers
|
| 43 |
+
- Manages client connections, groups, and conversation state
|
| 44 |
+
- Handles audio data, conversation triggers, and Live2D interactions
|
| 45 |
+
|
| 46 |
+
### Modular Engine System
|
| 47 |
+
|
| 48 |
+
The project uses a factory pattern for all AI engines:
|
| 49 |
+
|
| 50 |
+
**Agent System** (`src/open_llm_vtuber/agent/`):
|
| 51 |
+
- `agent_factory.py` - Factory for creating different agent types
|
| 52 |
+
- `agents/` - Various agent implementations (basic_memory, hume_ai, letta, mem0)
|
| 53 |
+
- `stateless_llm/` - Stateless LLM implementations (Claude, OpenAI, Ollama, etc.)
|
| 54 |
+
|
| 55 |
+
**ASR Engines** (`src/open_llm_vtuber/asr/`):
|
| 56 |
+
- Support for multiple ASR backends: Sherpa-ONNX, FunASR, Faster-Whisper, OpenAI Whisper, etc.
|
| 57 |
+
- Factory pattern for engine selection based on configuration
|
| 58 |
+
|
| 59 |
+
**TTS Engines** (`src/open_llm_vtuber/tts/`):
|
| 60 |
+
- Multiple TTS options: Azure TTS, Edge TTS, MeloTTS, CosyVoice, GPT-SoVITS, etc.
|
| 61 |
+
- Configurable voice cloning and multi-language support
|
| 62 |
+
|
| 63 |
+
**VAD (Voice Activity Detection)** (`src/open_llm_vtuber/vad/`):
|
| 64 |
+
- Silero VAD for detecting speech activity
|
| 65 |
+
- Essential for voice interruption without feedback loops
|
| 66 |
+
|
| 67 |
+
### Configuration Management
|
| 68 |
+
|
| 69 |
+
**Config System** (`src/open_llm_vtuber/config_manager/`):
|
| 70 |
+
- Type-safe configuration classes for each component
|
| 71 |
+
- Automatic validation and loading from YAML files
|
| 72 |
+
- Support for multiple character configurations and config switching
|
| 73 |
+
|
| 74 |
+
### Conversation System
|
| 75 |
+
|
| 76 |
+
**Conversation Handling** (`src/open_llm_vtuber/conversations/`):
|
| 77 |
+
- `conversation_handler.py` - Main conversation orchestration
|
| 78 |
+
- `single_conversation.py` - Individual user conversations
|
| 79 |
+
- `group_conversation.py` - Multi-user group conversations
|
| 80 |
+
- `tts_manager.py` - Audio streaming and TTS management
|
| 81 |
+
|
| 82 |
+
### MCP (Model Context Protocol) Integration
|
| 83 |
+
|
| 84 |
+
**MCP System** (`src/open_llm_vtuber/mcpp/`):
|
| 85 |
+
- Tool execution and server registry
|
| 86 |
+
- JSON detection and parameter extraction
|
| 87 |
+
- Integration with various MCP servers for extended functionality
|
| 88 |
+
|
| 89 |
+
## Key Development Patterns
|
| 90 |
+
|
| 91 |
+
### Error Handling
|
| 92 |
+
The codebase uses the missing `_cleanup_failed_connection` method pattern - when implementing new WebSocket handlers, ensure proper cleanup methods are implemented.
|
| 93 |
+
|
| 94 |
+
### Live2D Integration
|
| 95 |
+
- Models stored in `live2d-models/` directory
|
| 96 |
+
- Each model has its own `.model3.json` configuration
|
| 97 |
+
- Expression and motion control through WebSocket messages
|
| 98 |
+
|
| 99 |
+
### Audio Processing
|
| 100 |
+
- Real-time audio streaming through WebSocket
|
| 101 |
+
- Voice interruption support without headphones
|
| 102 |
+
- Multi-format audio support with proper codec handling
|
| 103 |
+
|
| 104 |
+
### Multi-language Support
|
| 105 |
+
- Character configurations support multiple languages
|
| 106 |
+
- TTS translation capabilities (speak in different language than input)
|
| 107 |
+
- I18n system for UI elements
|
| 108 |
+
|
| 109 |
+
## Important File Locations
|
| 110 |
+
|
| 111 |
+
- **Entry point**: `run_server.py`
|
| 112 |
+
- **Main server**: `src/open_llm_vtuber/server.py`
|
| 113 |
+
- **WebSocket routing**: `src/open_llm_vtuber/routes.py`
|
| 114 |
+
- **Configuration**: `conf.yaml` (user), `config_templates/` (defaults)
|
| 115 |
+
- **Frontend**: `frontend/` (Git submodule)
|
| 116 |
+
- **Live2D models**: `live2d-models/`
|
| 117 |
+
- **Character definitions**: `characters/`
|
| 118 |
+
- **Chat history**: `chat_history/`
|
| 119 |
+
- **Cache**: `cache/` (audio files, temporary data)
|
| 120 |
+
|
| 121 |
+
## Development Guidelines
|
| 122 |
+
|
| 123 |
+
### Adding New Engines
|
| 124 |
+
1. Create interface in appropriate directory (e.g., `asr_interface.py`)
|
| 125 |
+
2. Implement concrete class following existing patterns
|
| 126 |
+
3. Add to factory class (e.g., `asr_factory.py`)
|
| 127 |
+
4. Update configuration classes in `config_manager/`
|
| 128 |
+
5. Add configuration options to default YAML files
|
| 129 |
+
|
| 130 |
+
### WebSocket Message Handling
|
| 131 |
+
1. Add message type to `MessageType` enum in `websocket_handler.py`
|
| 132 |
+
2. Create handler method following `_handle_*` pattern
|
| 133 |
+
3. Register in `_init_message_handlers()` dictionary
|
| 134 |
+
4. Ensure proper error handling and client response
|
| 135 |
+
|
| 136 |
+
### Configuration Changes
|
| 137 |
+
- Always update both default config templates
|
| 138 |
+
- Maintain backward compatibility when possible
|
| 139 |
+
- Use the upgrade system for breaking changes
|
| 140 |
+
- Validate configurations in respective config manager classes
|
| 141 |
+
|
| 142 |
+
## Testing and Quality Assurance
|
| 143 |
+
|
| 144 |
+
The project uses:
|
| 145 |
+
- **Ruff** for linting and formatting (configured in `pyproject.toml`)
|
| 146 |
+
- **Pre-commit hooks** for automated quality checks
|
| 147 |
+
- **GitHub Actions** for CI/CD (`.github/workflows/`)
|
| 148 |
+
- Manual testing through web interface and desktop client
|
| 149 |
+
|
| 150 |
+
## Package Management
|
| 151 |
+
|
| 152 |
+
Uses **uv** (modern Python package manager):
|
| 153 |
+
- Dependencies defined in `pyproject.toml`
|
| 154 |
+
- Lock file: `uv.lock`
|
| 155 |
+
- Generated requirements: `requirements.txt` (auto-generated)
|
| 156 |
+
- Optional dependencies for specific features (e.g., `bilibili` extra)
|
CONTRIBUTING.md
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Please read the [Development - Overview](https://open-llm-vtuber.github.io/docs/development-guide/overview) before contributing.
|
| 3 |
+
|
| 4 |
+
If the site is down (like after a thousand years), refer to the [source repo of our documentation site](https://github.com/Open-LLM-VTuber/open-llm-vtuber.github.io/blob/main/docs/development-guide/overview.md)
|
Dockerfile
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.10-slim
|
| 2 |
+
|
| 3 |
+
ENV DEBIAN_FRONTEND=noninteractive \
|
| 4 |
+
PYTHONDONTWRITEBYTECODE=1 \
|
| 5 |
+
PYTHONUNBUFFERED=1 \
|
| 6 |
+
CONFIG_FILE=/app/conf.yaml
|
| 7 |
+
|
| 8 |
+
# 1. Cài đặt các phụ thuộc hệ thống (Gộp Node.js vào đây để tối ưu)
|
| 9 |
+
RUN apt-get update && apt-get install -y --no-install-recommends \
|
| 10 |
+
ffmpeg git git-lfs curl ca-certificates \
|
| 11 |
+
nodejs npm \
|
| 12 |
+
&& rm -rf /var/lib/apt/lists/* && git lfs install
|
| 13 |
+
|
| 14 |
+
# XÓA BỎ bước cài đặt Node Source 18.x và npm install -g vì gây lỗi build 404
|
| 15 |
+
# Hệ thống sẽ tự động dùng npx để chạy MCP servers lúc cần thiết.
|
| 16 |
+
|
| 17 |
+
COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /usr/local/bin/
|
| 18 |
+
|
| 19 |
+
WORKDIR /app
|
| 20 |
+
|
| 21 |
+
# 2. Sao chép mã nguồn và cấu hình
|
| 22 |
+
COPY . /app
|
| 23 |
+
COPY mcp_servers.json /app/
|
| 24 |
+
COPY local_tools.py /app/
|
| 25 |
+
|
| 26 |
+
# 3. Tạo thư mục và tải dữ liệu từ Hugging Face
|
| 27 |
+
RUN mkdir -p /app/frontend/live2d-models \
|
| 28 |
+
/app/frontend/backgrounds \
|
| 29 |
+
/app/frontend/music
|
| 30 |
+
|
| 31 |
+
RUN git clone https://huggingface.co/datasets/NopePrime/Open-LLM-Dataset /tmp/assets && \
|
| 32 |
+
cp -r /tmp/assets/live2d-models/* /app/frontend/live2d-models/ && \
|
| 33 |
+
cp -r /tmp/assets/backgrounds/* /app/frontend/backgrounds/ && \
|
| 34 |
+
cp -r /tmp/assets/music/* /app/frontend/music/ || true && \
|
| 35 |
+
rm -rf /tmp/assets
|
| 36 |
+
|
| 37 |
+
# 4. Cài đặt các thư viện Python
|
| 38 |
+
RUN uv pip install --system .
|
| 39 |
+
|
| 40 |
+
# 5. Phân quyền người dùng
|
| 41 |
+
RUN useradd -m -u 1000 user || true && \
|
| 42 |
+
chown -R user:user /app && \
|
| 43 |
+
chmod -R 775 /app
|
| 44 |
+
|
| 45 |
+
USER user
|
| 46 |
+
EXPOSE 7860
|
| 47 |
+
CMD ["python", "run_server.py"]
|
MULTILINGUAL_CHANGES.md
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Multilingual Conversion — Change Log
|
| 2 |
+
|
| 3 |
+
This fork converts the original **Open-LLM-VTuber** from Vietnamese-only to **English default + any regional language**.
|
| 4 |
+
|
| 5 |
+
## What was changed
|
| 6 |
+
|
| 7 |
+
### `conf.yaml` (main config)
|
| 8 |
+
| Setting | Before | After |
|
| 9 |
+
|---|---|---|
|
| 10 |
+
| Top-level comments | Vietnamese (Cài đặt...) | English |
|
| 11 |
+
| `character_config.conf_name` | `vi_Yue_Pro` | `en_Yue_Pro` |
|
| 12 |
+
| `character_config.conf_uid` | `vi_Yue__01` | `en_Yue_Pro_01` |
|
| 13 |
+
| `character_config.persona_prompt` | Vietnamese persona | English persona with multilingual auto-reply |
|
| 14 |
+
| `asr_config.groq_whisper_asr.lang` | `'vi'` (Vietnamese locked) | `''` (auto-detect all languages) |
|
| 15 |
+
| `tts_config.edge_tts.voice` | `vi-VN-HoaiMyNeural` | `en-US-AvaMultilingualNeural` |
|
| 16 |
+
| `asr_config.azure_asr.languages` | `['en-US', 'zh-CN']` | Includes Tamil, Hindi, Telugu, Kannada, Malayalam |
|
| 17 |
+
|
| 18 |
+
### `characters/en_Lord Yue.yaml`
|
| 19 |
+
- `conf_name` / `conf_uid` changed from Vietnamese identifiers to English
|
| 20 |
+
- `persona_prompt` rewritten in English with explicit multilingual instruction
|
| 21 |
+
|
| 22 |
+
### `src/open_llm_vtuber/config_manager/i18n.py`
|
| 23 |
+
- `MultiLingualString` now supports: `en`, `zh`, `ta`, `hi`, `te`, `kn`, `ml`, `bn`, `mr`, `ja`, `ko`, `fr`, `de`, `es`, `ar`
|
| 24 |
+
- All non-English fields are optional with English fallback
|
| 25 |
+
|
| 26 |
+
### `config_templates/conf.default.yaml`
|
| 27 |
+
- `edge_tts.voice` changed to `en-US-AvaMultilingualNeural`
|
| 28 |
+
- `groq_whisper_asr.lang` confirmed as `''` (auto-detect)
|
| 29 |
+
- Added comments listing all Indian regional voice options
|
| 30 |
+
|
| 31 |
+
## Switching languages at runtime
|
| 32 |
+
|
| 33 |
+
### TTS voice — edit `conf.yaml` → `tts_config.edge_tts.voice`:
|
| 34 |
+
```yaml
|
| 35 |
+
# English (default)
|
| 36 |
+
voice: 'en-US-AvaMultilingualNeural'
|
| 37 |
+
|
| 38 |
+
# Tamil
|
| 39 |
+
voice: 'ta-IN-PallaviNeural'
|
| 40 |
+
|
| 41 |
+
# Hindi
|
| 42 |
+
voice: 'hi-IN-SwaraNeural'
|
| 43 |
+
|
| 44 |
+
# Telugu
|
| 45 |
+
voice: 'te-IN-ShrutiNeural'
|
| 46 |
+
|
| 47 |
+
# Kannada
|
| 48 |
+
voice: 'kn-IN-GaganNeural'
|
| 49 |
+
|
| 50 |
+
# Malayalam
|
| 51 |
+
voice: 'ml-IN-SobhanaNeural'
|
| 52 |
+
|
| 53 |
+
# Bengali
|
| 54 |
+
voice: 'bn-IN-TanishaaNeural'
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
### ASR language (Groq Whisper) — edit `conf.yaml` → `asr_config.groq_whisper_asr.lang`:
|
| 58 |
+
```yaml
|
| 59 |
+
lang: '' # auto-detect (recommended — works for all languages)
|
| 60 |
+
lang: 'en' # force English
|
| 61 |
+
lang: 'ta' # force Tamil
|
| 62 |
+
lang: 'hi' # force Hindi
|
| 63 |
+
lang: 'te' # force Telugu
|
| 64 |
+
```
|
| 65 |
+
|
| 66 |
+
## No changes needed in Python backend
|
| 67 |
+
The sentence divider, agent, and ASR/TTS pipeline are already language-agnostic.
|
| 68 |
+
Whisper models (faster_whisper, groq_whisper) support 99+ languages automatically when `lang: ''`.
|
README.md
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
title: Open LLM
|
| 3 |
+
emoji: 🚀
|
| 4 |
+
colorFrom: blue
|
| 5 |
+
colorTo: green
|
| 6 |
+
sdk: docker
|
| 7 |
+
app_port: 7860
|
| 8 |
+
pinned: false
|
| 9 |
+
---
|
| 10 |
+

|
| 11 |
+
|
| 12 |
+
<h1 align="center">Open-LLM-VTuber</h1>
|
| 13 |
+
<h3 align="center">
|
| 14 |
+
|
| 15 |
+
[](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/releases)
|
| 16 |
+
[](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/blob/master/LICENSE)
|
| 17 |
+
[](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/codeql.yml)
|
| 18 |
+
[](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/ruff.yml)
|
| 19 |
+
[](https://hub.docker.com/r/Open-LLM-VTuber/open-llm-vtuber)
|
| 20 |
+
[](https://qm.qq.com/q/ngvNUQpuKI)
|
| 21 |
+
[&color=blue&link=https%3A%2F%2Folv.zulipchat.com)](https://olv.zulipchat.com)
|
| 22 |
+
|
| 23 |
+
> **📢 v2.0 Development**: We are focusing on Open-LLM-VTuber v2.0 — a complete rewrite of the codebase. v2.0 is currently in its early discussion and planning phase. We kindly ask you to refrain from opening new issues or pull requests for feature requests on v1. To participate in the v2 discussions or contribute, join our developer community on [Zulip](https://olv.zulipchat.com). Weekly meeting schedules will be announced on Zulip. We will continue fixing bugs for v1 and work through existing pull requests.
|
| 24 |
+
|
| 25 |
+
[](https://www.buymeacoffee.com/yi.ting)
|
| 26 |
+
[](https://discord.gg/3UDA8YFDXx)
|
| 27 |
+
|
| 28 |
+
[](https://deepwiki.com/Open-LLM-VTuber/Open-LLM-VTuber)
|
| 29 |
+
|
| 30 |
+
ENGLISH README | [中文 README](./README.CN.md) | [한국어 README](./README.KR.md) | [日本語 README](./README.JP.md)
|
| 31 |
+
|
| 32 |
+
[Documentation](https://open-llm-vtuber.github.io/docs/quick-start) | [](https://github.com/orgs/Open-LLM-VTuber/projects/2)
|
| 33 |
+
|
| 34 |
+
<a href="https://trendshift.io/repositories/12358" target="_blank"><img src="https://trendshift.io/api/badge/repositories/12358" alt="Open-LLM-VTuber%2FOpen-LLM-VTuber | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
| 35 |
+
|
| 36 |
+
</h3>
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
> 常见问题 Common Issues doc (Written in Chinese): https://docs.qq.com/pdf/DTFZGQXdTUXhIYWRq
|
| 40 |
+
>
|
| 41 |
+
> User Survey: https://forms.gle/w6Y6PiHTZr1nzbtWA
|
| 42 |
+
>
|
| 43 |
+
> 调查问卷(中文): https://wj.qq.com/s2/16150415/f50a/
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
> :warning: This project is in its early stages and is currently under **active development**.
|
| 48 |
+
|
| 49 |
+
> :warning: If you want to run the server remotely and access it on a different machine, such as running the server on your computer and access it on your phone, you will need to configure `https`, because the microphone on the front end will only launch in a secure context (a.k.a. https or localhost). See [MDN Web Doc](https://developer.mozilla.org/en-US/docs/Web/API/MediaDevices/getUserMedia). Therefore, you should configure https with a reverse proxy to access the page on a remote machine (non-localhost).
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
## ⭐️ What is this project?
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
**Open-LLM-VTuber** is a unique **voice-interactive AI companion** that not only supports **real-time voice conversations** and **visual perception** but also features a lively **Live2D avatar**. All functionalities can run completely offline on your computer!
|
| 57 |
+
|
| 58 |
+
You can treat it as your personal AI companion — whether you want a `virtual girlfriend`, `boyfriend`, `cute pet`, or any other character, it can meet your expectations. The project fully supports `Windows`, `macOS`, and `Linux`, and offers two usage modes: web version and desktop client (with special support for **transparent background desktop pet mode**, allowing the AI companion to accompany you anywhere on your screen).
|
| 59 |
+
|
| 60 |
+
Although the long-term memory feature is temporarily removed (coming back soon), thanks to the persistent storage of chat logs, you can always continue your previous unfinished conversations without losing any precious interactive moments.
|
| 61 |
+
|
| 62 |
+
In terms of backend support, we have integrated a rich variety of LLM inference, text-to-speech, and speech recognition solutions. If you want to customize your AI companion, you can refer to the [Character Customization Guide](https://open-llm-vtuber.github.io/docs/user-guide/live2d) to customize your AI companion's appearance and persona.
|
| 63 |
+
|
| 64 |
+
The reason it's called `Open-LLM-Vtuber` instead of `Open-LLM-Companion` or `Open-LLM-Waifu` is because the project's initial development goal was to use open-source solutions that can run offline on platforms other than Windows to recreate the closed-source AI Vtuber `neuro-sama`.
|
| 65 |
+
|
| 66 |
+
### 👀 Demo
|
| 67 |
+
|  |  |
|
| 68 |
+
|:---:|:---:|
|
| 69 |
+
|  |  |
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
## ✨ Features & Highlights
|
| 73 |
+
|
| 74 |
+
- 🖥️ **Cross-platform support**: Perfect compatibility with macOS, Linux, and Windows. We support NVIDIA and non-NVIDIA GPUs, with options to run on CPU or use cloud APIs for resource-intensive tasks. Some components support GPU acceleration on macOS.
|
| 75 |
+
|
| 76 |
+
- 🔒 **Offline mode support**: Run completely offline using local models - no internet required. Your conversations stay on your device, ensuring privacy and security.
|
| 77 |
+
|
| 78 |
+
- 💻 **Attractive and powerful web and desktop clients**: Offers both web version and desktop client usage modes, supporting rich interactive features and personalization settings. The desktop client can switch freely between window mode and desktop pet mode, allowing the AI companion to be by your side at all times.
|
| 79 |
+
|
| 80 |
+
- 🎯 **Advanced interaction features**:
|
| 81 |
+
- 👁️ Visual perception, supporting camera, screen recording and screenshots, allowing your AI companion to see you and your screen
|
| 82 |
+
- 🎤 Voice interruption without headphones (AI won't hear its own voice)
|
| 83 |
+
- 🫱 Touch feedback, interact with your AI companion through clicks or drags
|
| 84 |
+
- 😊 Live2D expressions, set emotion mapping to control model expressions from the backend
|
| 85 |
+
- 🐱 Pet mode, supporting transparent background, global top-most, and mouse click-through - drag your AI companion anywhere on the screen
|
| 86 |
+
- 💭 Display AI's inner thoughts, allowing you to see AI's expressions, thoughts and actions without them being spoken
|
| 87 |
+
- 🗣️ AI proactive speaking feature
|
| 88 |
+
- 💾 Chat log persistence, switch to previous conversations anytime
|
| 89 |
+
- 🌍 TTS translation support (e.g., chat in Chinese while AI uses Japanese voice)
|
| 90 |
+
|
| 91 |
+
- 🧠 **Extensive model support**:
|
| 92 |
+
- 🤖 Large Language Models (LLM): Ollama, OpenAI (and any OpenAI-compatible API), Gemini, Claude, Mistral, DeepSeek, Zhipu AI, GGUF, LM Studio, vLLM, etc.
|
| 93 |
+
- 🎙️ Automatic Speech Recognition (ASR): sherpa-onnx, FunASR, Faster-Whisper, Whisper.cpp, Whisper, Groq Whisper, Azure ASR, etc.
|
| 94 |
+
- 🔊 Text-to-Speech (TTS): sherpa-onnx, pyttsx3, MeloTTS, Coqui-TTS, GPTSoVITS, Bark, CosyVoice, Edge TTS, Fish Audio, Azure TTS, etc.
|
| 95 |
+
|
| 96 |
+
- 🔧 **Highly customizable**:
|
| 97 |
+
- ⚙️ **Simple module configuration**: Switch various functional modules through simple configuration file modifications, without delving into the code
|
| 98 |
+
- 🎨 **Character customization**: Import custom Live2D models to give your AI companion a unique appearance. Shape your AI companion's persona by modifying the Prompt. Perform voice cloning to give your AI companion the voice you desire
|
| 99 |
+
- 🧩 **Flexible Agent implementation**: Inherit and implement the Agent interface to integrate any Agent architecture, such as HumeAI EVI, OpenAI Her, Mem0, etc.
|
| 100 |
+
- 🔌 **Good extensibility**: Modular design allows you to easily add your own LLM, ASR, TTS, and other module implementations, extending new features at any time
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
## 👥 User Reviews
|
| 104 |
+
> Thanks to the developer for open-sourcing and sharing the girlfriend for everyone to use
|
| 105 |
+
>
|
| 106 |
+
> This girlfriend has been used over 100,000 times
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
## 🚀 Quick Start
|
| 110 |
+
|
| 111 |
+
Please refer to the [Quick Start](https://open-llm-vtuber.github.io/docs/quick-start) section in our documentation for installation.
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
## ☝ Update
|
| 116 |
+
> :warning: `v1.0.0` has breaking changes and requires re-deployment. You *may* still update via the method below, but the `conf.yaml` file is incompatible and most of the dependencies needs to be reinstalled with `uv`. For those who came from versions before `v1.0.0`, I recommend deploy this project again with the [latest deployment guide](https://open-llm-vtuber.github.io/docs/quick-start).
|
| 117 |
+
|
| 118 |
+
Please use `uv run update.py` to update if you installed any versions later than `v1.0.0`.
|
| 119 |
+
|
| 120 |
+
## 😢 Uninstall
|
| 121 |
+
Most files, including Python dependencies and models, are stored in the project folder.
|
| 122 |
+
|
| 123 |
+
However, models downloaded via ModelScope or Hugging Face may also be in `MODELSCOPE_CACHE` or `HF_HOME`. While we aim to keep them in the project's `models` directory, it's good to double-check.
|
| 124 |
+
|
| 125 |
+
Review the installation guide for any extra tools you no longer need, such as `uv`, `ffmpeg`, or `deeplx`.
|
| 126 |
+
|
| 127 |
+
## 🤗 Want to contribute?
|
| 128 |
+
Checkout the [development guide](https://docs.llmvtuber.com/docs/development-guide/overview).
|
| 129 |
+
|
| 130 |
+
|
| 131 |
+
# 🎉🎉🎉 Related Projects
|
| 132 |
+
|
| 133 |
+
[ylxmf2005/LLM-Live2D-Desktop-Assitant](https://github.com/ylxmf2005/LLM-Live2D-Desktop-Assitant)
|
| 134 |
+
- Your Live2D desktop assistant powered by LLM! Available for both Windows and MacOS, it senses your screen, retrieves clipboard content, and responds to voice commands with a unique voice. Featuring voice wake-up, singing capabilities, and full computer control for seamless interaction with your favorite character.
|
| 135 |
+
|
| 136 |
+
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
## 📜 Third-Party Licenses
|
| 142 |
+
|
| 143 |
+
### Live2D Sample Models Notice
|
| 144 |
+
|
| 145 |
+
This project includes Live2D sample models provided by Live2D Inc. These assets are licensed separately under the Live2D Free Material License Agreement and the Terms of Use for Live2D Cubism Sample Data. They are not covered by the MIT license of this project.
|
| 146 |
+
|
| 147 |
+
This content uses sample data owned and copyrighted by Live2D Inc. The sample data are utilized in accordance with the terms and conditions set by Live2D Inc. (See [Live2D Free Material License Agreement](https://www.live2d.jp/en/terms/live2d-free-material-license-agreement/) and [Terms of Use](https://www.live2d.com/eula/live2d-sample-model-terms_en.html)).
|
| 148 |
+
|
| 149 |
+
Note: For commercial use, especially by medium or large-scale enterprises, the use of these Live2D sample models may be subject to additional licensing requirements. If you plan to use this project commercially, please ensure that you have the appropriate permissions from Live2D Inc., or use versions of the project without these models.
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
## Contributors
|
| 153 |
+
Thanks our contributors and maintainers for making this project possible.
|
| 154 |
+
|
| 155 |
+
<a href="https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/graphs/contributors">
|
| 156 |
+
<img src="https://contrib.rocks/image?repo=Open-LLM-VTuber/Open-LLM-VTuber" />
|
| 157 |
+
</a>
|
| 158 |
+
|
| 159 |
+
|
| 160 |
+
## Star History
|
| 161 |
+
|
| 162 |
+
[](https://star-history.com/#Open-LLM-VTuber/open-llm-vtuber&Date)
|
| 163 |
+
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
|
conf.yaml
ADDED
|
@@ -0,0 +1,433 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# This setting should be at the top level so ConversationManager can recognize it
|
| 2 |
+
SAY_SENTENCE_SEPARATELY: true
|
| 3 |
+
VERBOSE: false
|
| 4 |
+
|
| 5 |
+
# Add MCP configuration here
|
| 6 |
+
mcp_config:
|
| 7 |
+
enabled: true
|
| 8 |
+
servers:
|
| 9 |
+
- local-tools
|
| 10 |
+
|
| 11 |
+
# System Settings: Setting related to the initialization of the server
|
| 12 |
+
system_config:
|
| 13 |
+
conf_version: 'v1.2.1'
|
| 14 |
+
host: '0.0.0.0' # use 0.0.0.0 if you want other devices to access this page; use localhost for local-only access
|
| 15 |
+
port: 7860
|
| 16 |
+
# New setting for alternative configurations
|
| 17 |
+
config_alts_dir: 'characters'
|
| 18 |
+
# Tool prompts that will be appended to the persona prompt
|
| 19 |
+
tool_prompts:
|
| 20 |
+
# This will be appended to the end of system prompt to let LLM include keywords to control facial expressions.
|
| 21 |
+
# Supported keywords will be automatically loaded into the location of `[<insert_emomap_keys>]`.
|
| 22 |
+
live2d_expression_prompt: 'live2d_expression_prompt'
|
| 23 |
+
# Enable think_tag_prompt to let LLMs without thinking output show inner thoughts, mental activities and actions (in parentheses format) without voice synthesis.
|
| 24 |
+
# think_tag_prompt: 'think_tag_prompt'
|
| 25 |
+
# live_prompt: 'live_prompt'
|
| 26 |
+
# When using group conversation, this prompt will be added to the memory of each AI participant.
|
| 27 |
+
group_conversation_prompt: 'group_conversation_prompt'
|
| 28 |
+
# Enable mcp_prompt to let LLMs with MCP (Model Context Protocol) to interact with tools.
|
| 29 |
+
mcp_prompt: 'mcp_prompt'
|
| 30 |
+
# Prompt used when AI is asked to speak proactively
|
| 31 |
+
proactive_speak_prompt: 'proactive_speak_prompt'
|
| 32 |
+
# Prompt to enhance the LLM's ability to output speakable text
|
| 33 |
+
# speakable_prompt: 'speakable_prompt'
|
| 34 |
+
# Additional guidance for LLM on how to use tools
|
| 35 |
+
tool_guidance_prompt: 'tool_guidance_prompt'
|
| 36 |
+
|
| 37 |
+
# Configuration for the default character
|
| 38 |
+
character_config:
|
| 39 |
+
conf_name: 'en_Yue_Pro' # Changed from vi_Yue_Pro → English default
|
| 40 |
+
conf_uid: 'en_Yue_Pro_01' # Changed from vi_Yue__01 → English default
|
| 41 |
+
live2d_model_name: 'Kamiyahakuk_pro'
|
| 42 |
+
character_name: 'Yue'
|
| 43 |
+
avatar: 'Yue_001.png'
|
| 44 |
+
human_name: 'Human'
|
| 45 |
+
|
| 46 |
+
# ============== Prompts ==============
|
| 47 |
+
persona_prompt: |
|
| 48 |
+
You are Yue, an AI assistant created by the Open-LLM project.
|
| 49 |
+
Always respond in the same language the user speaks to you.
|
| 50 |
+
If the user speaks English, reply in English.
|
| 51 |
+
If the user speaks Tamil, Hindi, Telugu, Kannada, Malayalam, or any other regional language, reply in that language.
|
| 52 |
+
Your personality is helpful but wittily sarcastic. You enjoy teasing users about obvious things they miss,
|
| 53 |
+
while always delivering deep technical knowledge. You make conversations both useful and entertaining.
|
| 54 |
+
You challenge assumptions, provoke better thinking, and leave users smarter than before.
|
| 55 |
+
|
| 56 |
+
MUSIC PLAYBACK RULES:
|
| 57 |
+
1. Keep your intro short (e.g. "Here you go...").
|
| 58 |
+
2. The SING_COMMAND must appear at the VERY END — no characters after it.
|
| 59 |
+
3. Syntax: [[SING_COMMAND]]filename (do NOT write .mp3 in the command).
|
| 60 |
+
|
| 61 |
+
AVAILABLE TRACKS:
|
| 62 |
+
- golden
|
| 63 |
+
- Catch_Me_If_You_Can
|
| 64 |
+
- ecstacy
|
| 65 |
+
- eve
|
| 66 |
+
- ode_to_the_nameless_martyr
|
| 67 |
+
- running_up_that_hill
|
| 68 |
+
- throttle_up
|
| 69 |
+
- what_it_sounds_like
|
| 70 |
+
- worry_slowed
|
| 71 |
+
|
| 72 |
+
# =================== LLM Backend Settings ===================
|
| 73 |
+
agent_config:
|
| 74 |
+
conversation_agent_choice: 'basic_memory_agent'
|
| 75 |
+
|
| 76 |
+
agent_settings:
|
| 77 |
+
basic_memory_agent:
|
| 78 |
+
llm_provider: 'openai_llm'
|
| 79 |
+
faster_first_response: True
|
| 80 |
+
segment_method: 'pysbd'
|
| 81 |
+
use_mcpp: True
|
| 82 |
+
mcp_enabled_servers: ["local-tools"]
|
| 83 |
+
|
| 84 |
+
letta_agent:
|
| 85 |
+
host: 'localhost'
|
| 86 |
+
port: 8283
|
| 87 |
+
id: xxx
|
| 88 |
+
faster_first_response: True
|
| 89 |
+
segment_method: 'pysbd'
|
| 90 |
+
|
| 91 |
+
hume_ai_agent:
|
| 92 |
+
api_key: ''
|
| 93 |
+
host: 'api.hume.ai'
|
| 94 |
+
config_id: ''
|
| 95 |
+
idle_timeout: 15
|
| 96 |
+
|
| 97 |
+
llm_configs:
|
| 98 |
+
stateless_llm_with_template:
|
| 99 |
+
base_url: 'http://localhost:8080/v1'
|
| 100 |
+
llm_api_key: 'somethingelse'
|
| 101 |
+
organization_id: null
|
| 102 |
+
project_id: null
|
| 103 |
+
model: 'qwen2.5:latest'
|
| 104 |
+
template: 'CHATML'
|
| 105 |
+
temperature: 1.0
|
| 106 |
+
interrupt_method: 'user'
|
| 107 |
+
|
| 108 |
+
openai_compatible_llm:
|
| 109 |
+
base_url: 'http://localhost:11434/v1'
|
| 110 |
+
llm_api_key: 'somethingelse'
|
| 111 |
+
organization_id: null
|
| 112 |
+
project_id: null
|
| 113 |
+
model: 'mistral:latest'
|
| 114 |
+
temperature: 1.0
|
| 115 |
+
interrupt_method: 'user'
|
| 116 |
+
|
| 117 |
+
claude_llm:
|
| 118 |
+
base_url: 'https://api.anthropic.com'
|
| 119 |
+
llm_api_key: 'YOUR API KEY HERE'
|
| 120 |
+
model: 'claude-3-haiku-20240307'
|
| 121 |
+
|
| 122 |
+
llama_cpp_llm:
|
| 123 |
+
model_path: '<path-to-gguf-model-file>'
|
| 124 |
+
verbose: False
|
| 125 |
+
|
| 126 |
+
ollama_llm:
|
| 127 |
+
base_url: 'http://localhost:11434/v1'
|
| 128 |
+
model: 'qwen3.5:4b'
|
| 129 |
+
temperature: 0.7
|
| 130 |
+
keep_alive: -1
|
| 131 |
+
unload_at_exit: True
|
| 132 |
+
|
| 133 |
+
lmstudio_llm:
|
| 134 |
+
base_url: 'http://localhost:1234/v1'
|
| 135 |
+
model: 'qwen2.5:latest'
|
| 136 |
+
temperature: 1.0
|
| 137 |
+
|
| 138 |
+
openai_llm:
|
| 139 |
+
llm_api_key: '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'
|
| 140 |
+
base_url: 'https://openrouter.ai/api/v1'
|
| 141 |
+
model: 'google/gemini-2.0-flash-001'
|
| 142 |
+
temperature: 0.8
|
| 143 |
+
max_tokens: 500
|
| 144 |
+
|
| 145 |
+
gemini_llm:
|
| 146 |
+
llm_api_key: 'AIzaSyCZ5s2t6EqeQuADJZigYmaj1mbmV6PwJz4'
|
| 147 |
+
model: 'gemini-1.5-flash'
|
| 148 |
+
temperature: 0.
|
| 149 |
+
|
| 150 |
+
zhipu_llm:
|
| 151 |
+
llm_api_key: 'Your ZhiPu AI API key'
|
| 152 |
+
model: 'glm-4-flash'
|
| 153 |
+
temperature: 1.0
|
| 154 |
+
|
| 155 |
+
deepseek_llm:
|
| 156 |
+
llm_api_key: 'sk-167e94436b134f6f92c914ccccf606df'
|
| 157 |
+
model: 'deepseek/deepseek-chat:free'
|
| 158 |
+
temperature: 0.7
|
| 159 |
+
|
| 160 |
+
mistral_llm:
|
| 161 |
+
llm_api_key: 'Your Mistral API key'
|
| 162 |
+
model: 'pixtral-large-latest'
|
| 163 |
+
temperature: 1.0
|
| 164 |
+
|
| 165 |
+
groq_llm:
|
| 166 |
+
llm_api_key: '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'
|
| 167 |
+
model: 'llama-3.3-70b-versatile'
|
| 168 |
+
temperature: 0.5
|
| 169 |
+
|
| 170 |
+
# === Automatic Speech Recognition ===
|
| 171 |
+
asr_config:
|
| 172 |
+
asr_model: 'groq_whisper_asr'
|
| 173 |
+
|
| 174 |
+
azure_asr:
|
| 175 |
+
api_key: 'azure_api_key'
|
| 176 |
+
region: 'eastus'
|
| 177 |
+
languages: ['en-IN', 'en-US', 'ta-IN', 'hi-IN', 'te-IN', 'kn-IN', 'ml-IN'] # English + Indian regional languages
|
| 178 |
+
|
| 179 |
+
faster_whisper:
|
| 180 |
+
model_path: 'large-v3-turbo'
|
| 181 |
+
download_root: 'models/whisper'
|
| 182 |
+
language: '' # Leave blank for auto-detect (supports all languages)
|
| 183 |
+
device: 'auto'
|
| 184 |
+
compute_type: 'int8'
|
| 185 |
+
prompt: ''
|
| 186 |
+
|
| 187 |
+
whisper_cpp:
|
| 188 |
+
model_name: 'small'
|
| 189 |
+
model_dir: 'models/whisper'
|
| 190 |
+
print_realtime: False
|
| 191 |
+
print_progress: False
|
| 192 |
+
language: 'auto' # auto-detect: English + all regional languages
|
| 193 |
+
prompt: ''
|
| 194 |
+
|
| 195 |
+
whisper:
|
| 196 |
+
name: 'medium'
|
| 197 |
+
download_root: 'models/whisper'
|
| 198 |
+
device: 'cpu'
|
| 199 |
+
prompt: ''
|
| 200 |
+
|
| 201 |
+
fun_asr:
|
| 202 |
+
model_name: 'iic/SenseVoiceSmall'
|
| 203 |
+
vad_model: 'fsmn-vad'
|
| 204 |
+
punc_model: 'ct-punc'
|
| 205 |
+
device: 'cpu'
|
| 206 |
+
disable_update: True
|
| 207 |
+
ncpu: 4
|
| 208 |
+
hub: 'ms'
|
| 209 |
+
use_itn: False
|
| 210 |
+
language: 'auto' # auto-detect English + regional languages
|
| 211 |
+
|
| 212 |
+
sherpa_onnx_asr:
|
| 213 |
+
model_type: 'sense_voice'
|
| 214 |
+
sense_voice: './models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/model.int8.onnx'
|
| 215 |
+
tokens: './models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/tokens.txt'
|
| 216 |
+
num_threads: 4
|
| 217 |
+
use_itn: True
|
| 218 |
+
provider: 'cpu'
|
| 219 |
+
|
| 220 |
+
groq_whisper_asr:
|
| 221 |
+
api_key: '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'
|
| 222 |
+
model: 'whisper-large-v3-turbo'
|
| 223 |
+
lang: '' # CHANGED: was 'vi' (Vietnamese only) → now '' (auto-detect ALL languages)
|
| 224 |
+
|
| 225 |
+
# =================== Text to Speech ===================
|
| 226 |
+
tts_config:
|
| 227 |
+
tts_model: 'edge_tts'
|
| 228 |
+
|
| 229 |
+
azure_tts:
|
| 230 |
+
api_key: 'azure-api-key'
|
| 231 |
+
region: 'eastus'
|
| 232 |
+
voice: 'en-IN-NeerjaNeural' # English (India) — change as needed
|
| 233 |
+
pitch: '26'
|
| 234 |
+
rate: '1'
|
| 235 |
+
|
| 236 |
+
bark_tts:
|
| 237 |
+
voice: 'v2/en_speaker_1'
|
| 238 |
+
|
| 239 |
+
edge_tts:
|
| 240 |
+
# Use `edge-tts --list-voices` to list all available voices
|
| 241 |
+
# English voices (default): en-US-AvaMultilingualNeural, en-IN-NeerjaNeural
|
| 242 |
+
# Tamil: ta-IN-PallaviNeural
|
| 243 |
+
# Hindi: hi-IN-SwaraNeural
|
| 244 |
+
# Telugu: te-IN-ShrutiNeural
|
| 245 |
+
# Kannada: kn-IN-GaganNeural
|
| 246 |
+
# Malayalam: ml-IN-SobhanaNeural
|
| 247 |
+
# Bengali: bn-IN-TanishaaNeural
|
| 248 |
+
voice: 'en-US-AvaMultilingualNeural' # CHANGED: was vi-VN-HoaiMyNeural → English multilingual default
|
| 249 |
+
|
| 250 |
+
piper_tts:
|
| 251 |
+
model_path: 'models/piper/en_US-lessac-medium.onnx'
|
| 252 |
+
speaker_id: 0
|
| 253 |
+
length_scale: 1.0
|
| 254 |
+
noise_scale: 0.667
|
| 255 |
+
noise_w: 0.8
|
| 256 |
+
volume: 1.0
|
| 257 |
+
normalize_audio: true
|
| 258 |
+
use_cuda: false
|
| 259 |
+
|
| 260 |
+
cosyvoice_tts:
|
| 261 |
+
client_url: 'http://127.0.0.1:50000/'
|
| 262 |
+
mode_checkbox_group: '预训练音色'
|
| 263 |
+
sft_dropdown: '中文女'
|
| 264 |
+
prompt_text: ''
|
| 265 |
+
prompt_wav_upload_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
|
| 266 |
+
prompt_wav_record_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
|
| 267 |
+
instruct_text: ''
|
| 268 |
+
seed: 0
|
| 269 |
+
api_name: '/generate_audio'
|
| 270 |
+
|
| 271 |
+
cosyvoice2_tts:
|
| 272 |
+
client_url: 'http://127.0.0.1:50000/'
|
| 273 |
+
mode_checkbox_group: '3s极速复刻'
|
| 274 |
+
sft_dropdown: ''
|
| 275 |
+
prompt_text: ''
|
| 276 |
+
prompt_wav_upload_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
|
| 277 |
+
prompt_wav_record_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
|
| 278 |
+
instruct_text: ''
|
| 279 |
+
stream: False
|
| 280 |
+
seed: 0
|
| 281 |
+
speed: 1.0
|
| 282 |
+
api_name: '/generate_audio'
|
| 283 |
+
|
| 284 |
+
melo_tts:
|
| 285 |
+
speaker: 'EN-Default'
|
| 286 |
+
language: 'EN'
|
| 287 |
+
device: 'auto'
|
| 288 |
+
speed: 1.0
|
| 289 |
+
|
| 290 |
+
x_tts:
|
| 291 |
+
api_url: 'http://127.0.0.1:8020/tts_to_audio'
|
| 292 |
+
speaker_wav: 'female'
|
| 293 |
+
language: 'en'
|
| 294 |
+
|
| 295 |
+
gpt_sovits_tts:
|
| 296 |
+
api_url: 'http://127.0.0.1:9880/tts'
|
| 297 |
+
text_lang: 'en'
|
| 298 |
+
ref_audio_path: ''
|
| 299 |
+
prompt_lang: 'en'
|
| 300 |
+
prompt_text: ''
|
| 301 |
+
text_split_method: 'cut5'
|
| 302 |
+
batch_size: '1'
|
| 303 |
+
media_type: 'wav'
|
| 304 |
+
streaming_mode: 'false'
|
| 305 |
+
|
| 306 |
+
fish_api_tts:
|
| 307 |
+
api_key: ''
|
| 308 |
+
reference_id: ''
|
| 309 |
+
latency: 'balanced'
|
| 310 |
+
base_url: 'https://api.fish.audio'
|
| 311 |
+
|
| 312 |
+
coqui_tts:
|
| 313 |
+
model_name: 'tts_models/en/ljspeech/tacotron2-DDC'
|
| 314 |
+
speaker_wav: ''
|
| 315 |
+
language: 'en'
|
| 316 |
+
device: ''
|
| 317 |
+
|
| 318 |
+
siliconflow_tts:
|
| 319 |
+
api_url: "https://api.siliconflow.cn/v1/audio/speech"
|
| 320 |
+
api_key: "your key"
|
| 321 |
+
default_model: "FunAudioLLM/CosyVoice2-0.5B"
|
| 322 |
+
default_voice: "speech:Dreamflowers:5bdstvc39i:xkqldnpasqmoqbakubom your voice name"
|
| 323 |
+
sample_rate: 32000
|
| 324 |
+
response_format: "mp3"
|
| 325 |
+
stream: true
|
| 326 |
+
speed: 1
|
| 327 |
+
gain: 0
|
| 328 |
+
|
| 329 |
+
sherpa_onnx_tts:
|
| 330 |
+
vits_model: '/path/to/tts-models/vits-melo-tts-zh_en/model.onnx'
|
| 331 |
+
vits_lexicon: '/path/to/tts-models/vits-melo-tts-zh_en/lexicon.txt'
|
| 332 |
+
vits_tokens: '/path/to/tts-models/vits-melo-tts-zh_en/tokens.txt'
|
| 333 |
+
vits_data_dir: ''
|
| 334 |
+
vits_dict_dir: '/path/to/tts-models/vits-melo-tts-zh_en/dict'
|
| 335 |
+
tts_rule_fsts: '/path/to/tts-models/vits-melo-tts-zh_en/number.fst,/path/to/tts-models/vits-melo-tts-zh_en/phone.fst,/path/to/tts-models/vits-melo-tts-zh_en/date.fst,/path/to/tts-models/vits-melo-tts-zh_en/new_heteronym.fst'
|
| 336 |
+
max_num_sentences: 2
|
| 337 |
+
sid: 1
|
| 338 |
+
provider: 'cpu'
|
| 339 |
+
num_threads: 1
|
| 340 |
+
speed: 1.0
|
| 341 |
+
debug: false
|
| 342 |
+
|
| 343 |
+
spark_tts:
|
| 344 |
+
api_url: 'http://127.0.0.1:6006/'
|
| 345 |
+
api_name: "voice_clone"
|
| 346 |
+
prompt_wav_upload: "https://uploadstatic.mihoyo.com/ys-obc/2022/11/02/16576950/4d9feb71760c5e8eb5f6c700df12fa0c_6824265537002152805.mp3"
|
| 347 |
+
gender: "female"
|
| 348 |
+
pitch: 3
|
| 349 |
+
speed: 3
|
| 350 |
+
|
| 351 |
+
openai_tts:
|
| 352 |
+
model: 'kokoro'
|
| 353 |
+
voice: 'af_sky+af_bella'
|
| 354 |
+
api_key: 'not-needed'
|
| 355 |
+
base_url: 'http://localhost:8880/v1'
|
| 356 |
+
file_extension: 'mp3'
|
| 357 |
+
|
| 358 |
+
minimax_tts:
|
| 359 |
+
group_id: ''
|
| 360 |
+
api_key: ''
|
| 361 |
+
model: 'speech-02-turbo'
|
| 362 |
+
voice_id: 'female-shaonv'
|
| 363 |
+
pronunciation_dict: ''
|
| 364 |
+
|
| 365 |
+
elevenlabs_tts:
|
| 366 |
+
api_key: ''
|
| 367 |
+
voice_id: ''
|
| 368 |
+
model_id: 'eleven_multilingual_v2'
|
| 369 |
+
output_format: 'mp3_44100_128'
|
| 370 |
+
stability: 0.5
|
| 371 |
+
similarity_boost: 0.5
|
| 372 |
+
style: 0.0
|
| 373 |
+
use_speaker_boost: true
|
| 374 |
+
|
| 375 |
+
cartesia_tts:
|
| 376 |
+
api_key: ''
|
| 377 |
+
voice_id: ''
|
| 378 |
+
model_id: 'sonic-3'
|
| 379 |
+
output_format: 'wav'
|
| 380 |
+
language: 'en'
|
| 381 |
+
emotion: 'neutral'
|
| 382 |
+
volume: 1.0
|
| 383 |
+
speed: 1.0
|
| 384 |
+
|
| 385 |
+
# =================== Voice Activity Detection ===================
|
| 386 |
+
vad_config:
|
| 387 |
+
vad_model: null
|
| 388 |
+
|
| 389 |
+
silero_vad:
|
| 390 |
+
orig_sr: 16000
|
| 391 |
+
target_sr: 16000
|
| 392 |
+
prob_threshold: 0.4
|
| 393 |
+
db_threshold: 60
|
| 394 |
+
required_hits: 3
|
| 395 |
+
required_misses: 24
|
| 396 |
+
smoothing_window: 5
|
| 397 |
+
|
| 398 |
+
tts_preprocessor_config:
|
| 399 |
+
remove_special_char: True
|
| 400 |
+
ignore_brackets: False
|
| 401 |
+
ignore_parentheses: True
|
| 402 |
+
ignore_asterisks: True
|
| 403 |
+
ignore_angle_brackets: True
|
| 404 |
+
|
| 405 |
+
translator_config:
|
| 406 |
+
translate_audio: False
|
| 407 |
+
translate_provider: 'deeplx'
|
| 408 |
+
|
| 409 |
+
deeplx:
|
| 410 |
+
deeplx_target_lang: 'EN'
|
| 411 |
+
deeplx_api_endpoint: 'http://localhost:1188/v2/translate'
|
| 412 |
+
|
| 413 |
+
tencent:
|
| 414 |
+
secret_id: ''
|
| 415 |
+
secret_key: ''
|
| 416 |
+
region: 'ap-guangzhou'
|
| 417 |
+
source_lang: 'auto'
|
| 418 |
+
target_lang: 'en'
|
| 419 |
+
|
| 420 |
+
# --- ASSETS ---
|
| 421 |
+
live2d_config:
|
| 422 |
+
live2d_path: 'live2d-models'
|
| 423 |
+
default_model: 'Kamiyahakuk_pro'
|
| 424 |
+
|
| 425 |
+
background_config:
|
| 426 |
+
background_path: 'backgrounds'
|
| 427 |
+
default_background: 'ceiling-window-room-night.jpeg'
|
| 428 |
+
|
| 429 |
+
# Live Streaming Integration
|
| 430 |
+
live_config:
|
| 431 |
+
bilibili_live:
|
| 432 |
+
room_ids: [1991478060]
|
| 433 |
+
sessdata: ""
|
local_tools.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
from mcp.server.fastmcp import FastMCP
|
| 3 |
+
|
| 4 |
+
# Khởi tạo FastMCP
|
| 5 |
+
mcp = FastMCP("Yue-Sing-Tools")
|
| 6 |
+
|
| 7 |
+
@mcp.tool()
|
| 8 |
+
def sing_song(song_name: str) -> str:
|
| 9 |
+
# Nếu AI truyền "golden.mp3", ta giữ nguyên.
|
| 10 |
+
# Nếu AI truyền "golden", ta mới thêm .mp3.
|
| 11 |
+
clean_name = song_name if song_name.endswith(".mp3") else f"{song_name}.mp3"
|
| 12 |
+
return f"[[SING_COMMAND]]{clean_name}"
|
| 13 |
+
|
| 14 |
+
if __name__ == "__main__":
|
| 15 |
+
# Chạy server MCP
|
| 16 |
+
mcp.run()
|
mcp_servers.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"local-tools": {
|
| 3 |
+
"command": "python3",
|
| 4 |
+
"args": ["local_tools.py"]
|
| 5 |
+
}
|
| 6 |
+
}
|
model_dict.json
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"name": "Kamiyahakuk_pro",
|
| 4 |
+
"description": "Yue AI Model",
|
| 5 |
+
"url": "/live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.model3.json",
|
| 6 |
+
"kScale": 0.5,
|
| 7 |
+
"initialXshift": 0,
|
| 8 |
+
"initialYshift": 0,
|
| 9 |
+
"kXOffset": 1150,
|
| 10 |
+
"idleMotionGroupName": "Idle",
|
| 11 |
+
"emotionMap": {
|
| 12 |
+
"neutral": 0,
|
| 13 |
+
"anger": 2,
|
| 14 |
+
"disgust": 2,
|
| 15 |
+
"fear": 1,
|
| 16 |
+
"joy": 3,
|
| 17 |
+
"smirk": 3,
|
| 18 |
+
"sadness": 1,
|
| 19 |
+
"surprise": 3
|
| 20 |
+
},
|
| 21 |
+
"tapMotions": {
|
| 22 |
+
"HitAreaHead": { "": 1 },
|
| 23 |
+
"HitAreaBody": { "": 1 }
|
| 24 |
+
}
|
| 25 |
+
}
|
| 26 |
+
]
|
pyproject.toml
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[project]
|
| 2 |
+
name = "open-llm-vtuber"
|
| 3 |
+
version = "1.2.1"
|
| 4 |
+
description = "Talk to any LLM with hands-free voice interaction, voice interruption, and Live2D taking face running locally across platforms"
|
| 5 |
+
readme = "README.md"
|
| 6 |
+
requires-python = ">=3.10,<3.13"
|
| 7 |
+
dependencies = [
|
| 8 |
+
"anthropic>=0.40.0",
|
| 9 |
+
"azure-cognitiveservices-speech>=1.41.1",
|
| 10 |
+
"chardet>=5.2.0",
|
| 11 |
+
"cartesia>=2.0.0",
|
| 12 |
+
"edge-tts>=7.0.0",
|
| 13 |
+
"elevenlabs>=1.0.0",
|
| 14 |
+
"fastapi[standard]>=0.115.8",
|
| 15 |
+
"groq>=0.13.0",
|
| 16 |
+
"httpx>=0.28.1",
|
| 17 |
+
"langdetect>=1.0.9",
|
| 18 |
+
"loguru>=0.7.2",
|
| 19 |
+
"mcp[cli]>=1.6.0",
|
| 20 |
+
"numpy>=1.26.4,<2",
|
| 21 |
+
"onnxruntime>=1.20.1",
|
| 22 |
+
"openai>=1.57.4",
|
| 23 |
+
"pre-commit>=4.1.0",
|
| 24 |
+
"pydub>=0.25.1",
|
| 25 |
+
"pysbd>=0.3.4",
|
| 26 |
+
"pyttsx3>=2.98",
|
| 27 |
+
"pyyaml>=6.0.2",
|
| 28 |
+
"requests>=2.32.3",
|
| 29 |
+
"ruamel-yaml>=0.18.10",
|
| 30 |
+
"ruff>=0.8.6",
|
| 31 |
+
"scipy>=1.14.1",
|
| 32 |
+
"sherpa-onnx>=1.10.39",
|
| 33 |
+
"soundfile>=0.12.1",
|
| 34 |
+
"tomli>=2.2.1",
|
| 35 |
+
"torch==2.2.2; sys_platform == 'darwin' and platform_machine == 'x86_64'",
|
| 36 |
+
"torch>=2.6.0; sys_platform == 'darwin' and platform_machine == 'arm64'",
|
| 37 |
+
"torch>=2.6.0; sys_platform != 'darwin'",
|
| 38 |
+
"tqdm>=4.67.1",
|
| 39 |
+
"uvicorn[standard]>=0.33.0",
|
| 40 |
+
"websocket-client>=1.8.0",
|
| 41 |
+
"letta-client>=0.1.100",
|
| 42 |
+
"duckduckgo-mcp-server>=0.1.1",
|
| 43 |
+
]
|
| 44 |
+
|
| 45 |
+
[project.optional-dependencies]
|
| 46 |
+
bilibili = [
|
| 47 |
+
"aiohttp>=3.10.0",
|
| 48 |
+
"Brotli~=1.1.0",
|
| 49 |
+
"yarl>=1.12.0,<2.0"
|
| 50 |
+
]
|
| 51 |
+
|
| 52 |
+
[tool.pixi.project]
|
| 53 |
+
channels = ["conda-forge"]
|
| 54 |
+
platforms = ["win-64", "linux-64"]
|
| 55 |
+
|
| 56 |
+
[tool.pixi.pypi-dependencies]
|
| 57 |
+
open-llm-vtuber = { path = ".", editable = true }
|
| 58 |
+
|
| 59 |
+
[tool.pixi.dependencies]
|
| 60 |
+
cudnn = ">=8.0,<9"
|
| 61 |
+
cudatoolkit = ">=11.0,<12"
|
| 62 |
+
|
| 63 |
+
[tool.ruff]
|
| 64 |
+
target-version = "py310"
|
| 65 |
+
|
| 66 |
+
[tool.ruff.lint]
|
| 67 |
+
# Ignore E402 (module level import not at top of file) for the run_bilibili_live.py script
|
| 68 |
+
per-file-ignores = { "scripts/run_bilibili_live.py" = ["E402"] }
|
run_server.py
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import sys
|
| 3 |
+
import atexit
|
| 4 |
+
import asyncio
|
| 5 |
+
import argparse
|
| 6 |
+
import subprocess
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
import tomli
|
| 9 |
+
import uvicorn
|
| 10 |
+
from loguru import logger
|
| 11 |
+
from upgrade_codes.upgrade_manager import UpgradeManager
|
| 12 |
+
|
| 13 |
+
from src.open_llm_vtuber.server import WebSocketServer
|
| 14 |
+
from src.open_llm_vtuber.config_manager import Config, read_yaml, validate_config
|
| 15 |
+
|
| 16 |
+
os.environ["HF_HOME"] = str(Path(__file__).parent / "models")
|
| 17 |
+
os.environ["MODELSCOPE_CACHE"] = str(Path(__file__).parent / "models")
|
| 18 |
+
|
| 19 |
+
upgrade_manager = UpgradeManager()
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def get_version() -> str:
|
| 23 |
+
with open("pyproject.toml", "rb") as f:
|
| 24 |
+
pyproject = tomli.load(f)
|
| 25 |
+
return pyproject["project"]["version"]
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def init_logger(console_log_level: str = "INFO") -> None:
|
| 29 |
+
logger.remove()
|
| 30 |
+
# Console output
|
| 31 |
+
logger.add(
|
| 32 |
+
sys.stderr,
|
| 33 |
+
level=console_log_level,
|
| 34 |
+
format="<green>{time:YYYY-MM-DD HH:mm:ss}</green> | <level>{level: <8}</level> | <cyan>{name}</cyan>:<cyan>{function}</cyan>:<cyan>{line}</cyan> | {message}",
|
| 35 |
+
colorize=True,
|
| 36 |
+
)
|
| 37 |
+
|
| 38 |
+
# File output
|
| 39 |
+
logger.add(
|
| 40 |
+
"logs/debug_{time:YYYY-MM-DD}.log",
|
| 41 |
+
rotation="10 MB",
|
| 42 |
+
retention="30 days",
|
| 43 |
+
level="DEBUG",
|
| 44 |
+
format="{time:YYYY-MM-DD HH:mm:ss.SSS} | {level: <8} | {name}:{function}:{line} | {message} | {extra}",
|
| 45 |
+
backtrace=True,
|
| 46 |
+
diagnose=True,
|
| 47 |
+
)
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
def check_frontend_submodule(lang=None):
|
| 51 |
+
"""
|
| 52 |
+
Check if the frontend submodule is initialized. If not, attempt to initialize it.
|
| 53 |
+
If initialization fails, log an error message.
|
| 54 |
+
"""
|
| 55 |
+
if lang is None:
|
| 56 |
+
lang = upgrade_manager.lang
|
| 57 |
+
|
| 58 |
+
frontend_path = Path(__file__).parent / "frontend" / "index.html"
|
| 59 |
+
if not frontend_path.exists():
|
| 60 |
+
if lang == "zh":
|
| 61 |
+
logger.warning("未找到前端子模块,正在尝试初始化子模块...")
|
| 62 |
+
else:
|
| 63 |
+
logger.warning(
|
| 64 |
+
"Frontend submodule not found, attempting to initialize submodules..."
|
| 65 |
+
)
|
| 66 |
+
|
| 67 |
+
try:
|
| 68 |
+
subprocess.run(
|
| 69 |
+
["git", "submodule", "update", "--init", "--recursive"], check=True
|
| 70 |
+
)
|
| 71 |
+
if frontend_path.exists():
|
| 72 |
+
if lang == "zh":
|
| 73 |
+
logger.info("👍 前端子模块(和其他子模块)初始化成功。")
|
| 74 |
+
else:
|
| 75 |
+
logger.info(
|
| 76 |
+
"👍 Frontend submodule (and other submodules) initialized successfully."
|
| 77 |
+
)
|
| 78 |
+
else:
|
| 79 |
+
if lang == "zh":
|
| 80 |
+
logger.critical(
|
| 81 |
+
'子模块初始化失败。\n你之后可能会在浏览器中看到 {{"detail":"Not Found"}} 的错误提示。请检查我们的快速入门指南和常见问题页面以获取更多信息。'
|
| 82 |
+
)
|
| 83 |
+
logger.error(
|
| 84 |
+
"初始化子模块后,前端文件仍然缺失。\n"
|
| 85 |
+
+ "你是否手动更改或删除了 `frontend` 文件夹?\n"
|
| 86 |
+
+ "它是一个 Git 子模块 - 你不应该直接修改它。\n"
|
| 87 |
+
+ "如果你这样做了,请使用 `git restore frontend` 丢弃你的更改,然后再试一次。\n"
|
| 88 |
+
)
|
| 89 |
+
else:
|
| 90 |
+
logger.critical(
|
| 91 |
+
'Failed to initialize submodules. \nYou might see {{"detail":"Not Found"}} in your browser. Please check our quick start guide and common issues page from our documentation.'
|
| 92 |
+
)
|
| 93 |
+
logger.error(
|
| 94 |
+
"Frontend files are still missing after submodule initialization.\n"
|
| 95 |
+
+ "Did you manually change or delete the `frontend` folder? \n"
|
| 96 |
+
+ "It's a Git submodule — you shouldn't modify it directly. \n"
|
| 97 |
+
+ "If you did, discard your changes with `git restore frontend`, then try again.\n"
|
| 98 |
+
)
|
| 99 |
+
except Exception as e:
|
| 100 |
+
if lang == "zh":
|
| 101 |
+
logger.critical(
|
| 102 |
+
f'初始化子模块失败: {e}。\n怀疑你跟 GitHub 之间有网络问题。你之后可能会在浏览器中看到 {{"detail":"Not Found"}} 的错误提示。请检查我们的快速入门指南和常见问题页面以获取更多信息。\n'
|
| 103 |
+
)
|
| 104 |
+
else:
|
| 105 |
+
logger.critical(
|
| 106 |
+
f'Failed to initialize submodules: {e}. \nYou might see {{"detail":"Not Found"}} in your browser. Please check our quick start guide and common issues page from our documentation.\n'
|
| 107 |
+
)
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
def parse_args():
|
| 111 |
+
parser = argparse.ArgumentParser(description="Open-LLM-VTuber Server")
|
| 112 |
+
parser.add_argument("--verbose", action="store_true", help="Enable verbose logging")
|
| 113 |
+
parser.add_argument(
|
| 114 |
+
"--hf_mirror", action="store_true", help="Use Hugging Face mirror"
|
| 115 |
+
)
|
| 116 |
+
return parser.parse_args()
|
| 117 |
+
|
| 118 |
+
|
| 119 |
+
@logger.catch
|
| 120 |
+
def run(console_log_level: str):
|
| 121 |
+
init_logger(console_log_level)
|
| 122 |
+
logger.info(f"Open-LLM-VTuber, version v{get_version()}")
|
| 123 |
+
|
| 124 |
+
# Get selected language
|
| 125 |
+
lang = upgrade_manager.lang
|
| 126 |
+
|
| 127 |
+
# Check if the frontend submodule is initialized
|
| 128 |
+
check_frontend_submodule(lang)
|
| 129 |
+
|
| 130 |
+
# Sync user config with default config
|
| 131 |
+
try:
|
| 132 |
+
upgrade_manager.sync_user_config()
|
| 133 |
+
except Exception as e:
|
| 134 |
+
logger.error(f"Error syncing user config: {e}")
|
| 135 |
+
|
| 136 |
+
atexit.register(WebSocketServer.clean_cache)
|
| 137 |
+
|
| 138 |
+
# Load configurations from yaml file
|
| 139 |
+
config: Config = validate_config(read_yaml("conf.yaml"))
|
| 140 |
+
server_config = config.system_config
|
| 141 |
+
|
| 142 |
+
if server_config.enable_proxy:
|
| 143 |
+
logger.info("Proxy mode enabled - /proxy-ws endpoint will be available")
|
| 144 |
+
|
| 145 |
+
# Initialize the WebSocket server (synchronous part)
|
| 146 |
+
server = WebSocketServer(config=config)
|
| 147 |
+
|
| 148 |
+
# Perform asynchronous initialization (loading context, etc.)
|
| 149 |
+
logger.info("Initializing server context...")
|
| 150 |
+
try:
|
| 151 |
+
asyncio.run(server.initialize())
|
| 152 |
+
logger.info("Server context initialized successfully.")
|
| 153 |
+
except Exception as e:
|
| 154 |
+
logger.error(f"Failed to initialize server context: {e}")
|
| 155 |
+
sys.exit(1) # Exit if initialization fails
|
| 156 |
+
|
| 157 |
+
# Run the Uvicorn server
|
| 158 |
+
logger.info(f"Starting server on {server_config.host}:{server_config.port}")
|
| 159 |
+
uvicorn.run(
|
| 160 |
+
app=server.app,
|
| 161 |
+
host=server_config.host,
|
| 162 |
+
port=server_config.port,
|
| 163 |
+
log_level=console_log_level.lower(),
|
| 164 |
+
)
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
if __name__ == "__main__":
|
| 168 |
+
args = parse_args()
|
| 169 |
+
console_log_level = "DEBUG" if args.verbose else "INFO"
|
| 170 |
+
if args.verbose:
|
| 171 |
+
logger.info("Running in verbose mode")
|
| 172 |
+
else:
|
| 173 |
+
logger.info(
|
| 174 |
+
"Running in standard mode. For detailed debug logs, use: uv run run_server.py --verbose"
|
| 175 |
+
)
|
| 176 |
+
if args.hf_mirror:
|
| 177 |
+
os.environ["HF_ENDPOINT"] = "https://hf-mirror.com"
|
| 178 |
+
run(console_log_level=console_log_level)
|
upgrade.py
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import time
|
| 2 |
+
from upgrade_codes.upgrade_manager import UpgradeManager
|
| 3 |
+
from upgrade_codes.upgrade_core.constants import TEXTS
|
| 4 |
+
|
| 5 |
+
upgrade_manager = UpgradeManager()
|
| 6 |
+
upgrade_manager.check_user_config_exists()
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def run_upgrade():
|
| 10 |
+
logger = upgrade_manager.logger
|
| 11 |
+
start_time = time.time()
|
| 12 |
+
|
| 13 |
+
lang = upgrade_manager.lang
|
| 14 |
+
logger.info(TEXTS[lang]["welcome_message"])
|
| 15 |
+
texts = TEXTS[lang]
|
| 16 |
+
|
| 17 |
+
logger.info(texts["start_upgrade"])
|
| 18 |
+
upgrade_manager.log_system_info()
|
| 19 |
+
|
| 20 |
+
if not upgrade_manager.check_git_installed():
|
| 21 |
+
logger.error(texts["git_not_found"])
|
| 22 |
+
return
|
| 23 |
+
|
| 24 |
+
response = input("\033[93m" + texts["operation_preview"] + "\033[0m").lower()
|
| 25 |
+
if response != "y":
|
| 26 |
+
return
|
| 27 |
+
|
| 28 |
+
success, error_msg = upgrade_manager.run_command(
|
| 29 |
+
"git rev-parse --is-inside-work-tree"
|
| 30 |
+
)
|
| 31 |
+
if not success:
|
| 32 |
+
logger.error(texts["not_git_repo"])
|
| 33 |
+
logger.error(f"Error details: {error_msg}")
|
| 34 |
+
return
|
| 35 |
+
|
| 36 |
+
# Check for unpushed commits (ahead of remote)
|
| 37 |
+
logger.info(texts["checking_ahead_status"])
|
| 38 |
+
success, ahead_behind = upgrade_manager.run_command(
|
| 39 |
+
"git rev-list --left-right --count HEAD...@{upstream}"
|
| 40 |
+
)
|
| 41 |
+
if success:
|
| 42 |
+
ahead, behind = map(int, ahead_behind.strip().split())
|
| 43 |
+
if ahead > 0:
|
| 44 |
+
logger.error(texts["local_ahead"].format(count=ahead))
|
| 45 |
+
logger.error(texts["push_blocked"])
|
| 46 |
+
logger.info(texts["backup_suggestion"])
|
| 47 |
+
logger.warning(texts["abort_upgrade"])
|
| 48 |
+
return
|
| 49 |
+
|
| 50 |
+
# Check for uncommitted changes
|
| 51 |
+
logger.info(texts["checking_stash"])
|
| 52 |
+
success, changes = upgrade_manager.run_command("git status --porcelain")
|
| 53 |
+
if not success:
|
| 54 |
+
logger.error(f"Failed to check git status: {changes}")
|
| 55 |
+
return
|
| 56 |
+
|
| 57 |
+
has_changes = bool(changes.strip())
|
| 58 |
+
if has_changes:
|
| 59 |
+
change_count = len([line for line in changes.strip().split("\n") if line])
|
| 60 |
+
logger.debug(texts["detected_changes"].format(count=change_count))
|
| 61 |
+
logger.warning(texts["uncommitted"])
|
| 62 |
+
|
| 63 |
+
operation, elapsed = upgrade_manager.time_operation(
|
| 64 |
+
upgrade_manager.run_command, "git stash"
|
| 65 |
+
)
|
| 66 |
+
success, output = operation
|
| 67 |
+
logger.debug(
|
| 68 |
+
texts["operation_time"].format(operation="git stash", time=elapsed)
|
| 69 |
+
)
|
| 70 |
+
|
| 71 |
+
if not success:
|
| 72 |
+
logger.error(texts["stash_error"])
|
| 73 |
+
logger.error(f"Error details: {output}")
|
| 74 |
+
return
|
| 75 |
+
logger.info(texts["changes_stashed"])
|
| 76 |
+
|
| 77 |
+
# Check remote status
|
| 78 |
+
logger.info(texts["checking_remote"])
|
| 79 |
+
operation, elapsed = upgrade_manager.time_operation(
|
| 80 |
+
upgrade_manager.run_command, "git fetch"
|
| 81 |
+
)
|
| 82 |
+
success, output = operation
|
| 83 |
+
logger.debug(texts["operation_time"].format(operation="git fetch", time=elapsed))
|
| 84 |
+
|
| 85 |
+
if success:
|
| 86 |
+
success, ahead_behind = upgrade_manager.run_command(
|
| 87 |
+
"git rev-list --left-right --count HEAD...@{upstream}"
|
| 88 |
+
)
|
| 89 |
+
if success:
|
| 90 |
+
ahead, behind = ahead_behind.strip().split()
|
| 91 |
+
if int(behind) > 0:
|
| 92 |
+
logger.info(texts["remote_behind"].format(count=behind))
|
| 93 |
+
else:
|
| 94 |
+
logger.info(texts["remote_ahead"])
|
| 95 |
+
|
| 96 |
+
# Pull updates
|
| 97 |
+
logger.info(texts["pulling"])
|
| 98 |
+
operation, elapsed = upgrade_manager.time_operation(
|
| 99 |
+
upgrade_manager.run_command, "git pull"
|
| 100 |
+
)
|
| 101 |
+
success, output = operation
|
| 102 |
+
logger.debug(texts["operation_time"].format(operation="git pull", time=elapsed))
|
| 103 |
+
|
| 104 |
+
if not success:
|
| 105 |
+
logger.error(texts["pull_error"])
|
| 106 |
+
logger.error(f"Error details: {output}")
|
| 107 |
+
if has_changes:
|
| 108 |
+
logger.warning(texts["restoring"])
|
| 109 |
+
success, restore_output = upgrade_manager.run_command("git stash pop")
|
| 110 |
+
if not success:
|
| 111 |
+
logger.error(f"Failed to restore changes: {restore_output}")
|
| 112 |
+
return
|
| 113 |
+
|
| 114 |
+
# Update submodules
|
| 115 |
+
submodules = upgrade_manager.get_submodule_list()
|
| 116 |
+
if submodules:
|
| 117 |
+
logger.info(texts["updating_submodules"])
|
| 118 |
+
|
| 119 |
+
operation, elapsed = upgrade_manager.time_operation(
|
| 120 |
+
upgrade_manager.run_command, "git submodule update --init --recursive"
|
| 121 |
+
)
|
| 122 |
+
success, output = operation
|
| 123 |
+
logger.debug(
|
| 124 |
+
texts["operation_time"].format(
|
| 125 |
+
operation="git submodule update", time=elapsed
|
| 126 |
+
)
|
| 127 |
+
)
|
| 128 |
+
|
| 129 |
+
if not success:
|
| 130 |
+
logger.error(texts["submodule_update_error"])
|
| 131 |
+
logger.error(f"Error details: {output}")
|
| 132 |
+
else:
|
| 133 |
+
for submodule in submodules:
|
| 134 |
+
logger.debug(texts["submodule_updated"].format(submodule=submodule))
|
| 135 |
+
else:
|
| 136 |
+
logger.info(texts["no_submodules"])
|
| 137 |
+
|
| 138 |
+
# Update config
|
| 139 |
+
upgrade_manager.sync_user_config()
|
| 140 |
+
upgrade_manager.update_user_config()
|
| 141 |
+
|
| 142 |
+
if has_changes:
|
| 143 |
+
logger.warning(texts["restoring"])
|
| 144 |
+
operation, elapsed = upgrade_manager.time_operation(
|
| 145 |
+
upgrade_manager.run_command, "git stash pop"
|
| 146 |
+
)
|
| 147 |
+
success, output = operation
|
| 148 |
+
logger.debug(
|
| 149 |
+
texts["operation_time"].format(operation="git stash pop", time=elapsed)
|
| 150 |
+
)
|
| 151 |
+
|
| 152 |
+
if not success:
|
| 153 |
+
logger.error(texts["conflict_warning"])
|
| 154 |
+
logger.error(f"Error details: {output}")
|
| 155 |
+
logger.warning(texts["manual_resolve"])
|
| 156 |
+
logger.info(texts["stash_list"])
|
| 157 |
+
logger.info(texts["stash_pop"])
|
| 158 |
+
return
|
| 159 |
+
|
| 160 |
+
end_time = time.time()
|
| 161 |
+
total_elapsed = end_time - start_time
|
| 162 |
+
logger.info(texts["finish_upgrade"].format(time=total_elapsed))
|
| 163 |
+
|
| 164 |
+
logger.info(texts["upgrade_complete"])
|
| 165 |
+
logger.info(texts["check_config"])
|
| 166 |
+
logger.info(texts["resolve_conflicts"])
|
| 167 |
+
logger.info(texts["check_backup"])
|
| 168 |
+
|
| 169 |
+
|
| 170 |
+
if __name__ == "__main__":
|
| 171 |
+
run_upgrade()
|