britto224 commited on
Commit
79592ba
·
verified ·
1 Parent(s): 3fde5f3

Upload 17 files

Browse files
.gitattributes CHANGED
@@ -1,15 +1,84 @@
1
- avatars/mao.png filter=lfs diff=lfs merge=lfs -text
2
- avatars/Yue_001.png filter=lfs diff=lfs merge=lfs -text
3
- backgrounds/ceiling-computer-room-night.jpg filter=lfs diff=lfs merge=lfs -text
4
- backgrounds/ceiling-window-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
5
- live2d-models/haru_greeter_pro_jp/haru_\#U00c4=\#U00f2t\#U00e2X\#U00fc\[\#U00e2c_\#U00e2C\#U00e2\#U00f4\#U00e2_\#U00fc\[\#U00e2g_t07.psd filter=lfs diff=lfs merge=lfs -text
6
- live2d-models/haru_greeter_pro_jp/haru_\#U00c4=\#U00f2t\#U00e2X\#U00fc\[\#U00e2c_\#U00e6f\#U00ec\#U00a6\#U00f2\#U00ac\#U00e9\#U00bb_t07.psd filter=lfs diff=lfs merge=lfs -text
7
- live2d-models/haru_greeter_pro_jp/haru_greeter_t03.can3 filter=lfs diff=lfs merge=lfs -text
8
- live2d-models/haru_greeter_pro_jp/haru_greeter_t05.cmo3 filter=lfs diff=lfs merge=lfs -text
9
- live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_00.png filter=lfs diff=lfs merge=lfs -text
10
- live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_01.png filter=lfs diff=lfs merge=lfs -text
11
- live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
12
- live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_01.png filter=lfs diff=lfs merge=lfs -text
13
- live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_02.png filter=lfs diff=lfs merge=lfs -text
14
- live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_03.png filter=lfs diff=lfs merge=lfs -text
15
- live2d-models/mao_pro/runtime/mao_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ static/libs/* linguist-vendoredfrontend/libs/ort-wasm-simd-threaded.wasm filter=lfs diff=lfs merge=lfs -text
2
+ frontend/libs/ort-wasm-simd.wasm filter=lfs diff=lfs merge=lfs -text
3
+ frontend/libs/ort-wasm-threaded.wasm filter=lfs diff=lfs merge=lfs -text
4
+ frontend/libs/ort-wasm.wasm filter=lfs diff=lfs merge=lfs -text
5
+ frontend/libs/silero_vad_legacy.onnx filter=lfs diff=lfs merge=lfs -text
6
+ frontend/libs/silero_vad_v5.onnx filter=lfs diff=lfs merge=lfs -text
7
+ frontend/libs/ort-wasm-simd-threaded.wasm filter=lfs diff=lfs merge=lfs -text
8
+ live2d-models/mao_pro/runtime/mao_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
9
+ live2d-models/mao_pro/runtime/mao_pro.moc3 filter=lfs diff=lfs merge=lfs -text
10
+ live2d-models/shizuku/runtime/shizuku.1024/texture_00.png filter=lfs diff=lfs merge=lfs -text
11
+ live2d-models/shizuku/runtime/shizuku.1024/texture_01.png filter=lfs diff=lfs merge=lfs -text
12
+ live2d-models/shizuku/runtime/shizuku.1024/texture_02.png filter=lfs diff=lfs merge=lfs -text
13
+ live2d-models/shizuku/runtime/shizuku.1024/texture_03.png filter=lfs diff=lfs merge=lfs -text
14
+ live2d-models/shizuku/runtime/shizuku.1024/texture_04.png filter=lfs diff=lfs merge=lfs -text
15
+ live2d-models/shizuku/runtime/shizuku.moc3 filter=lfs diff=lfs merge=lfs -text
16
+ backgrounds/cartoon-night-landscape-moon.jpeg filter=lfs diff=lfs merge=lfs -text
17
+ backgrounds/cityscape.jpeg filter=lfs diff=lfs merge=lfs -text
18
+ backgrounds/computer-room-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
19
+ backgrounds/congress.jpg filter=lfs diff=lfs merge=lfs -text
20
+ backgrounds/field-night-painting-moon.jpeg filter=lfs diff=lfs merge=lfs -text
21
+ backgrounds/lernado-diff-classroom-center.jpeg filter=lfs diff=lfs merge=lfs -text
22
+ backgrounds/moon-over-mountain.jpeg filter=lfs diff=lfs merge=lfs -text
23
+ backgrounds/mountain-range-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
24
+ backgrounds/night-landscape-grass-moon.jpeg filter=lfs diff=lfs merge=lfs -text
25
+ backgrounds/night-scene-cartoon-moon.jpeg filter=lfs diff=lfs merge=lfs -text
26
+ backgrounds/painting-valley-night-sky.[[:space:]]2.jpeg filter=lfs diff=lfs merge=lfs -text
27
+ backgrounds/room-interior-illustration.jpeg filter=lfs diff=lfs merge=lfs -text
28
+ backgrounds/sdxl-classroom-door-view.jpeg filter=lfs diff=lfs merge=lfs -text
29
+ avatars/mao.png filter=lfs diff=lfs merge=lfs -text
30
+ avatars/shizuku.png filter=lfs diff=lfs merge=lfs -text
31
+ frontend/music/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
32
+ frontend/music/eve.mp3 filter=lfs diff=lfs merge=lfs -text
33
+ frontend/music/golden.mp3 filter=lfs diff=lfs merge=lfs -text
34
+ frontend/music/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
35
+ frontend/music/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
36
+ frontend/music/the_awakening.mp3 filter=lfs diff=lfs merge=lfs -text
37
+ frontend/music/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
38
+ frontend/music/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
39
+ frontend/music/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
40
+ live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
41
+ live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_01.png filter=lfs diff=lfs merge=lfs -text
42
+ live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_02.png filter=lfs diff=lfs merge=lfs -text
43
+ live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.4096/texture_03.png filter=lfs diff=lfs merge=lfs -text
44
+ live2d-models/Kamiyahakuk_pro/Kamiyahakuk_pro.moc3 filter=lfs diff=lfs merge=lfs -text
45
+ ceiling-window-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
46
+ live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_00.png filter=lfs diff=lfs merge=lfs -text
47
+ live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_01.png filter=lfs diff=lfs merge=lfs -text
48
+ live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_02.png filter=lfs diff=lfs merge=lfs -text
49
+ live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.4096/texture_03.png filter=lfs diff=lfs merge=lfs -text
50
+ live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.moc3 filter=lfs diff=lfs merge=lfs -text
51
+ avatars/Yue_001.png filter=lfs diff=lfs merge=lfs -text
52
+ backgrounds/ceiling-window-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
53
+ frontend/music/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
54
+ sing/original/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
55
+ sing/original/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
56
+ sing/original/eve.mp3 filter=lfs diff=lfs merge=lfs -text
57
+ sing/original/golden.mp3 filter=lfs diff=lfs merge=lfs -text
58
+ sing/original/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
59
+ sing/original/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
60
+ sing/original/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
61
+ sing/original/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
62
+ sing/original/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
63
+ sing/tracks/Catch_Me_If_You_Can.mp3 filter=lfs diff=lfs merge=lfs -text
64
+ sing/tracks/ecstacy.mp3 filter=lfs diff=lfs merge=lfs -text
65
+ sing/tracks/eve.mp3 filter=lfs diff=lfs merge=lfs -text
66
+ sing/tracks/golden.mp3 filter=lfs diff=lfs merge=lfs -text
67
+ sing/tracks/ode_to_the_nameless_martyr.mp3 filter=lfs diff=lfs merge=lfs -text
68
+ sing/tracks/running_up_that_hill.mp3 filter=lfs diff=lfs merge=lfs -text
69
+ sing/tracks/throttle_up.mp3 filter=lfs diff=lfs merge=lfs -text
70
+ sing/tracks/what_it_sounds_like.mp3 filter=lfs diff=lfs merge=lfs -text
71
+ sing/tracks/worry_slowed.mp3 filter=lfs diff=lfs merge=lfs -text
72
+ live2d-models/haru_greeter_pro_jp/haru_Ä=òtâXü\[âc_âCâôâ_ü\[âg_t07.psd filter=lfs diff=lfs merge=lfs -text
73
+ live2d-models/haru_greeter_pro_jp/haru_Ä=òtâXü\[âc_æfì¦ò¬é»_t07.psd filter=lfs diff=lfs merge=lfs -text
74
+ live2d-models/haru_greeter_pro_jp/haru_greeter_t03.can3 filter=lfs diff=lfs merge=lfs -text
75
+ live2d-models/haru_greeter_pro_jp/haru_greeter_t05.cmo3 filter=lfs diff=lfs merge=lfs -text
76
+ live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_00.png filter=lfs diff=lfs merge=lfs -text
77
+ live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.2048/texture_01.png filter=lfs diff=lfs merge=lfs -text
78
+ live2d-models/haru_greeter_pro_jp/runtime/haru_greeter_t05.moc3 filter=lfs diff=lfs merge=lfs -text
79
+ backgrounds/ceiling-computer-room-night.jpeg filter=lfs diff=lfs merge=lfs -text
80
+ backgrounds/ceiling-computer-room-night.jpg filter=lfs diff=lfs merge=lfs -text
81
+ backgrounds/ceiling-window-room-night.jpeg.jpg filter=lfs diff=lfs merge=lfs -text
82
+ sing/tracks/Light.mp3 filter=lfs diff=lfs merge=lfs -text
83
+ sing/tracks/take_a_slice.mp3 filter=lfs diff=lfs merge=lfs -text
84
+ sing/tracks/Washing_Machine_Hear.mp3 filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
Binary file (102 Bytes). View file
 
.gitmodules ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ [submodule "frontend"]
2
+ path = frontend
3
+ url = https://github.com/Open-LLM-VTuber/Open-LLM-VTuber-Web
4
+ branch = build
.pre-commit-config.yaml ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ repos:
2
+ - repo: https://github.com/astral-sh/ruff-pre-commit
3
+ rev: v0.9.6
4
+ hooks:
5
+ - id: ruff
6
+ args: [--fix, --exit-non-zero-on-fix]
7
+ - id: ruff-format
8
+
9
+
.python-version ADDED
@@ -0,0 +1 @@
 
 
1
+ 3.10
CLAUDE.md ADDED
@@ -0,0 +1,156 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # CLAUDE.md
2
+
3
+ This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
4
+
5
+ ## Project Overview
6
+
7
+ Open-LLM-VTuber is a voice-interactive AI companion with Live2D avatar support that runs completely offline. It's a cross-platform Python application supporting real-time voice conversations, visual perception, and Live2D character animations. The project features modular architecture for LLM, ASR (Automatic Speech Recognition), TTS (Text-to-Speech), and other components.
8
+
9
+ ## Essential Commands
10
+
11
+ ### Development Setup
12
+ - **Install dependencies**: `uv sync` (uses uv package manager)
13
+ - **Run server**: `uv run run_server.py`
14
+ - **Run with verbose logging**: `uv run run_server.py --verbose`
15
+ - **Update project**: `uv run upgrade.py`
16
+
17
+ ### Code Quality
18
+ - **Lint code**: `ruff check .`
19
+ - **Format code**: `ruff format .`
20
+ - **Run pre-commit hooks**: `pre-commit run --all-files`
21
+
22
+ ### Server Configuration
23
+ - **Main config file**: `conf.yaml` (user configuration)
24
+ - **Default configs**: `config_templates/conf.default.yaml` and `config_templates/conf.ZH.default.yaml`
25
+ - **Character configs**: `characters/` directory (YAML files)
26
+
27
+ ## Architecture Overview
28
+
29
+ ### Core Components
30
+
31
+ **WebSocket Server** (`src/open_llm_vtuber/server.py`):
32
+ - FastAPI-based server handling WebSocket connections
33
+ - Serves frontend, Live2D models, and static assets
34
+ - Supports both main client and proxy WebSocket endpoints
35
+
36
+ **Service Context** (`src/open_llm_vtuber/service_context.py`):
37
+ - Central dependency injection container
38
+ - Manages all engines (LLM, ASR, TTS, VAD, etc.)
39
+ - Each WebSocket connection gets its own service context instance
40
+
41
+ **WebSocket Handler** (`src/open_llm_vtuber/websocket_handler.py`):
42
+ - Routes WebSocket messages to appropriate handlers
43
+ - Manages client connections, groups, and conversation state
44
+ - Handles audio data, conversation triggers, and Live2D interactions
45
+
46
+ ### Modular Engine System
47
+
48
+ The project uses a factory pattern for all AI engines:
49
+
50
+ **Agent System** (`src/open_llm_vtuber/agent/`):
51
+ - `agent_factory.py` - Factory for creating different agent types
52
+ - `agents/` - Various agent implementations (basic_memory, hume_ai, letta, mem0)
53
+ - `stateless_llm/` - Stateless LLM implementations (Claude, OpenAI, Ollama, etc.)
54
+
55
+ **ASR Engines** (`src/open_llm_vtuber/asr/`):
56
+ - Support for multiple ASR backends: Sherpa-ONNX, FunASR, Faster-Whisper, OpenAI Whisper, etc.
57
+ - Factory pattern for engine selection based on configuration
58
+
59
+ **TTS Engines** (`src/open_llm_vtuber/tts/`):
60
+ - Multiple TTS options: Azure TTS, Edge TTS, MeloTTS, CosyVoice, GPT-SoVITS, etc.
61
+ - Configurable voice cloning and multi-language support
62
+
63
+ **VAD (Voice Activity Detection)** (`src/open_llm_vtuber/vad/`):
64
+ - Silero VAD for detecting speech activity
65
+ - Essential for voice interruption without feedback loops
66
+
67
+ ### Configuration Management
68
+
69
+ **Config System** (`src/open_llm_vtuber/config_manager/`):
70
+ - Type-safe configuration classes for each component
71
+ - Automatic validation and loading from YAML files
72
+ - Support for multiple character configurations and config switching
73
+
74
+ ### Conversation System
75
+
76
+ **Conversation Handling** (`src/open_llm_vtuber/conversations/`):
77
+ - `conversation_handler.py` - Main conversation orchestration
78
+ - `single_conversation.py` - Individual user conversations
79
+ - `group_conversation.py` - Multi-user group conversations
80
+ - `tts_manager.py` - Audio streaming and TTS management
81
+
82
+ ### MCP (Model Context Protocol) Integration
83
+
84
+ **MCP System** (`src/open_llm_vtuber/mcpp/`):
85
+ - Tool execution and server registry
86
+ - JSON detection and parameter extraction
87
+ - Integration with various MCP servers for extended functionality
88
+
89
+ ## Key Development Patterns
90
+
91
+ ### Error Handling
92
+ The codebase uses the missing `_cleanup_failed_connection` method pattern - when implementing new WebSocket handlers, ensure proper cleanup methods are implemented.
93
+
94
+ ### Live2D Integration
95
+ - Models stored in `live2d-models/` directory
96
+ - Each model has its own `.model3.json` configuration
97
+ - Expression and motion control through WebSocket messages
98
+
99
+ ### Audio Processing
100
+ - Real-time audio streaming through WebSocket
101
+ - Voice interruption support without headphones
102
+ - Multi-format audio support with proper codec handling
103
+
104
+ ### Multi-language Support
105
+ - Character configurations support multiple languages
106
+ - TTS translation capabilities (speak in different language than input)
107
+ - I18n system for UI elements
108
+
109
+ ## Important File Locations
110
+
111
+ - **Entry point**: `run_server.py`
112
+ - **Main server**: `src/open_llm_vtuber/server.py`
113
+ - **WebSocket routing**: `src/open_llm_vtuber/routes.py`
114
+ - **Configuration**: `conf.yaml` (user), `config_templates/` (defaults)
115
+ - **Frontend**: `frontend/` (Git submodule)
116
+ - **Live2D models**: `live2d-models/`
117
+ - **Character definitions**: `characters/`
118
+ - **Chat history**: `chat_history/`
119
+ - **Cache**: `cache/` (audio files, temporary data)
120
+
121
+ ## Development Guidelines
122
+
123
+ ### Adding New Engines
124
+ 1. Create interface in appropriate directory (e.g., `asr_interface.py`)
125
+ 2. Implement concrete class following existing patterns
126
+ 3. Add to factory class (e.g., `asr_factory.py`)
127
+ 4. Update configuration classes in `config_manager/`
128
+ 5. Add configuration options to default YAML files
129
+
130
+ ### WebSocket Message Handling
131
+ 1. Add message type to `MessageType` enum in `websocket_handler.py`
132
+ 2. Create handler method following `_handle_*` pattern
133
+ 3. Register in `_init_message_handlers()` dictionary
134
+ 4. Ensure proper error handling and client response
135
+
136
+ ### Configuration Changes
137
+ - Always update both default config templates
138
+ - Maintain backward compatibility when possible
139
+ - Use the upgrade system for breaking changes
140
+ - Validate configurations in respective config manager classes
141
+
142
+ ## Testing and Quality Assurance
143
+
144
+ The project uses:
145
+ - **Ruff** for linting and formatting (configured in `pyproject.toml`)
146
+ - **Pre-commit hooks** for automated quality checks
147
+ - **GitHub Actions** for CI/CD (`.github/workflows/`)
148
+ - Manual testing through web interface and desktop client
149
+
150
+ ## Package Management
151
+
152
+ Uses **uv** (modern Python package manager):
153
+ - Dependencies defined in `pyproject.toml`
154
+ - Lock file: `uv.lock`
155
+ - Generated requirements: `requirements.txt` (auto-generated)
156
+ - Optional dependencies for specific features (e.g., `bilibili` extra)
CONTRIBUTING.md ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+
2
+ Please read the [Development - Overview](https://open-llm-vtuber.github.io/docs/development-guide/overview) before contributing.
3
+
4
+ If the site is down (like after a thousand years), refer to the [source repo of our documentation site](https://github.com/Open-LLM-VTuber/open-llm-vtuber.github.io/blob/main/docs/development-guide/overview.md)
Dockerfile ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.10-slim
2
+
3
+ ENV DEBIAN_FRONTEND=noninteractive \
4
+ PYTHONDONTWRITEBYTECODE=1 \
5
+ PYTHONUNBUFFERED=1 \
6
+ CONFIG_FILE=/app/conf.yaml
7
+
8
+ # 1. Cài đặt các phụ thuộc hệ thống (Gộp Node.js vào đây để tối ưu)
9
+ RUN apt-get update && apt-get install -y --no-install-recommends \
10
+ ffmpeg git git-lfs curl ca-certificates \
11
+ nodejs npm \
12
+ && rm -rf /var/lib/apt/lists/* && git lfs install
13
+
14
+ # XÓA BỎ bước cài đặt Node Source 18.x và npm install -g vì gây lỗi build 404
15
+ # Hệ thống sẽ tự động dùng npx để chạy MCP servers lúc cần thiết.
16
+
17
+ COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /usr/local/bin/
18
+
19
+ WORKDIR /app
20
+
21
+ # 2. Sao chép mã nguồn và cấu hình
22
+ COPY . /app
23
+ COPY mcp_servers.json /app/
24
+ COPY local_tools.py /app/
25
+
26
+ # 3. Tạo thư mục và tải dữ liệu từ Hugging Face
27
+ RUN mkdir -p /app/frontend/live2d-models \
28
+ /app/frontend/backgrounds \
29
+ /app/frontend/music
30
+
31
+ RUN git clone https://huggingface.co/datasets/NopePrime/Open-LLM-Dataset /tmp/assets && \
32
+ cp -r /tmp/assets/live2d-models/* /app/frontend/live2d-models/ && \
33
+ cp -r /tmp/assets/backgrounds/* /app/frontend/backgrounds/ && \
34
+ cp -r /tmp/assets/music/* /app/frontend/music/ || true && \
35
+ rm -rf /tmp/assets
36
+
37
+ # 4. Cài đặt các thư viện Python
38
+ RUN uv pip install --system .
39
+
40
+ # 5. Phân quyền người dùng
41
+ RUN useradd -m -u 1000 user || true && \
42
+ chown -R user:user /app && \
43
+ chmod -R 775 /app
44
+
45
+ USER user
46
+ EXPOSE 7860
47
+ CMD ["python", "run_server.py"]
MULTILINGUAL_CHANGES.md ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Multilingual Conversion — Change Log
2
+
3
+ This fork converts the original **Open-LLM-VTuber** from Vietnamese-only to **English default + any regional language**.
4
+
5
+ ## What was changed
6
+
7
+ ### `conf.yaml` (main config)
8
+ | Setting | Before | After |
9
+ |---|---|---|
10
+ | Top-level comments | Vietnamese (Cài đặt...) | English |
11
+ | `character_config.conf_name` | `vi_Yue_Pro` | `en_Yue_Pro` |
12
+ | `character_config.conf_uid` | `vi_Yue__01` | `en_Yue_Pro_01` |
13
+ | `character_config.persona_prompt` | Vietnamese persona | English persona with multilingual auto-reply |
14
+ | `asr_config.groq_whisper_asr.lang` | `'vi'` (Vietnamese locked) | `''` (auto-detect all languages) |
15
+ | `tts_config.edge_tts.voice` | `vi-VN-HoaiMyNeural` | `en-US-AvaMultilingualNeural` |
16
+ | `asr_config.azure_asr.languages` | `['en-US', 'zh-CN']` | Includes Tamil, Hindi, Telugu, Kannada, Malayalam |
17
+
18
+ ### `characters/en_Lord Yue.yaml`
19
+ - `conf_name` / `conf_uid` changed from Vietnamese identifiers to English
20
+ - `persona_prompt` rewritten in English with explicit multilingual instruction
21
+
22
+ ### `src/open_llm_vtuber/config_manager/i18n.py`
23
+ - `MultiLingualString` now supports: `en`, `zh`, `ta`, `hi`, `te`, `kn`, `ml`, `bn`, `mr`, `ja`, `ko`, `fr`, `de`, `es`, `ar`
24
+ - All non-English fields are optional with English fallback
25
+
26
+ ### `config_templates/conf.default.yaml`
27
+ - `edge_tts.voice` changed to `en-US-AvaMultilingualNeural`
28
+ - `groq_whisper_asr.lang` confirmed as `''` (auto-detect)
29
+ - Added comments listing all Indian regional voice options
30
+
31
+ ## Switching languages at runtime
32
+
33
+ ### TTS voice — edit `conf.yaml` → `tts_config.edge_tts.voice`:
34
+ ```yaml
35
+ # English (default)
36
+ voice: 'en-US-AvaMultilingualNeural'
37
+
38
+ # Tamil
39
+ voice: 'ta-IN-PallaviNeural'
40
+
41
+ # Hindi
42
+ voice: 'hi-IN-SwaraNeural'
43
+
44
+ # Telugu
45
+ voice: 'te-IN-ShrutiNeural'
46
+
47
+ # Kannada
48
+ voice: 'kn-IN-GaganNeural'
49
+
50
+ # Malayalam
51
+ voice: 'ml-IN-SobhanaNeural'
52
+
53
+ # Bengali
54
+ voice: 'bn-IN-TanishaaNeural'
55
+ ```
56
+
57
+ ### ASR language (Groq Whisper) — edit `conf.yaml` → `asr_config.groq_whisper_asr.lang`:
58
+ ```yaml
59
+ lang: '' # auto-detect (recommended — works for all languages)
60
+ lang: 'en' # force English
61
+ lang: 'ta' # force Tamil
62
+ lang: 'hi' # force Hindi
63
+ lang: 'te' # force Telugu
64
+ ```
65
+
66
+ ## No changes needed in Python backend
67
+ The sentence divider, agent, and ASR/TTS pipeline are already language-agnostic.
68
+ Whisper models (faster_whisper, groq_whisper) support 99+ languages automatically when `lang: ''`.
README.md ADDED
@@ -0,0 +1,167 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Open LLM
3
+ emoji: 🚀
4
+ colorFrom: blue
5
+ colorTo: green
6
+ sdk: docker
7
+ app_port: 7860
8
+ pinned: false
9
+ ---
10
+ ![](./assets/banner.jpg)
11
+
12
+ <h1 align="center">Open-LLM-VTuber</h1>
13
+ <h3 align="center">
14
+
15
+ [![GitHub release](https://img.shields.io/github/v/release/Open-LLM-VTuber/Open-LLM-VTuber)](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/releases)
16
+ [![license](https://img.shields.io/github/license/Open-LLM-VTuber/Open-LLM-VTuber)](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/blob/master/LICENSE)
17
+ [![CodeQL](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/codeql.yml/badge.svg)](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/codeql.yml)
18
+ [![Ruff](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/ruff.yml/badge.svg)](https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/actions/workflows/ruff.yml)
19
+ [![Docker](https://img.shields.io/badge/Open-LLM-VTuber%2FOpen--LLM--VTuber-%25230db7ed.svg?logo=docker&logoColor=blue&labelColor=white&color=blue)](https://hub.docker.com/r/Open-LLM-VTuber/open-llm-vtuber)
20
+ [![QQ User Group](https://img.shields.io/badge/QQ_User_Group-792615362-white?style=flat&logo=qq&logoColor=white)](https://qm.qq.com/q/ngvNUQpuKI)
21
+ [![Static Badge](https://img.shields.io/badge/Join%20Chat-Zulip?style=flat&logo=zulip&label=Zulip(dev-community)&color=blue&link=https%3A%2F%2Folv.zulipchat.com)](https://olv.zulipchat.com)
22
+
23
+ > **📢 v2.0 Development**: We are focusing on Open-LLM-VTuber v2.0 — a complete rewrite of the codebase. v2.0 is currently in its early discussion and planning phase. We kindly ask you to refrain from opening new issues or pull requests for feature requests on v1. To participate in the v2 discussions or contribute, join our developer community on [Zulip](https://olv.zulipchat.com). Weekly meeting schedules will be announced on Zulip. We will continue fixing bugs for v1 and work through existing pull requests.
24
+
25
+ [![BuyMeACoffee](https://img.shields.io/badge/Buy%20Me%20a%20Coffee-ffdd00?style=for-the-badge&logo=buy-me-a-coffee&logoColor=black)](https://www.buymeacoffee.com/yi.ting)
26
+ [![](https://dcbadge.limes.pink/api/server/3UDA8YFDXx)](https://discord.gg/3UDA8YFDXx)
27
+
28
+ [![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/Open-LLM-VTuber/Open-LLM-VTuber)
29
+
30
+ ENGLISH README | [中文 README](./README.CN.md) | [한국어 README](./README.KR.md) | [日本語 README](./README.JP.md)
31
+
32
+ [Documentation](https://open-llm-vtuber.github.io/docs/quick-start) | [![Roadmap](https://img.shields.io/badge/Roadmap-GitHub_Project-yellow)](https://github.com/orgs/Open-LLM-VTuber/projects/2)
33
+
34
+ <a href="https://trendshift.io/repositories/12358" target="_blank"><img src="https://trendshift.io/api/badge/repositories/12358" alt="Open-LLM-VTuber%2FOpen-LLM-VTuber | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
35
+
36
+ </h3>
37
+
38
+
39
+ > 常见问题 Common Issues doc (Written in Chinese): https://docs.qq.com/pdf/DTFZGQXdTUXhIYWRq
40
+ >
41
+ > User Survey: https://forms.gle/w6Y6PiHTZr1nzbtWA
42
+ >
43
+ > 调查问卷(中文): https://wj.qq.com/s2/16150415/f50a/
44
+
45
+
46
+
47
+ > :warning: This project is in its early stages and is currently under **active development**.
48
+
49
+ > :warning: If you want to run the server remotely and access it on a different machine, such as running the server on your computer and access it on your phone, you will need to configure `https`, because the microphone on the front end will only launch in a secure context (a.k.a. https or localhost). See [MDN Web Doc](https://developer.mozilla.org/en-US/docs/Web/API/MediaDevices/getUserMedia). Therefore, you should configure https with a reverse proxy to access the page on a remote machine (non-localhost).
50
+
51
+
52
+
53
+ ## ⭐️ What is this project?
54
+
55
+
56
+ **Open-LLM-VTuber** is a unique **voice-interactive AI companion** that not only supports **real-time voice conversations** and **visual perception** but also features a lively **Live2D avatar**. All functionalities can run completely offline on your computer!
57
+
58
+ You can treat it as your personal AI companion — whether you want a `virtual girlfriend`, `boyfriend`, `cute pet`, or any other character, it can meet your expectations. The project fully supports `Windows`, `macOS`, and `Linux`, and offers two usage modes: web version and desktop client (with special support for **transparent background desktop pet mode**, allowing the AI companion to accompany you anywhere on your screen).
59
+
60
+ Although the long-term memory feature is temporarily removed (coming back soon), thanks to the persistent storage of chat logs, you can always continue your previous unfinished conversations without losing any precious interactive moments.
61
+
62
+ In terms of backend support, we have integrated a rich variety of LLM inference, text-to-speech, and speech recognition solutions. If you want to customize your AI companion, you can refer to the [Character Customization Guide](https://open-llm-vtuber.github.io/docs/user-guide/live2d) to customize your AI companion's appearance and persona.
63
+
64
+ The reason it's called `Open-LLM-Vtuber` instead of `Open-LLM-Companion` or `Open-LLM-Waifu` is because the project's initial development goal was to use open-source solutions that can run offline on platforms other than Windows to recreate the closed-source AI Vtuber `neuro-sama`.
65
+
66
+ ### 👀 Demo
67
+ | ![](assets/i1.jpg) | ![](assets/i2.jpg) |
68
+ |:---:|:---:|
69
+ | ![](assets/i3.jpg) | ![](assets/i4.jpg) |
70
+
71
+
72
+ ## ✨ Features & Highlights
73
+
74
+ - 🖥️ **Cross-platform support**: Perfect compatibility with macOS, Linux, and Windows. We support NVIDIA and non-NVIDIA GPUs, with options to run on CPU or use cloud APIs for resource-intensive tasks. Some components support GPU acceleration on macOS.
75
+
76
+ - 🔒 **Offline mode support**: Run completely offline using local models - no internet required. Your conversations stay on your device, ensuring privacy and security.
77
+
78
+ - 💻 **Attractive and powerful web and desktop clients**: Offers both web version and desktop client usage modes, supporting rich interactive features and personalization settings. The desktop client can switch freely between window mode and desktop pet mode, allowing the AI companion to be by your side at all times.
79
+
80
+ - 🎯 **Advanced interaction features**:
81
+ - 👁️ Visual perception, supporting camera, screen recording and screenshots, allowing your AI companion to see you and your screen
82
+ - 🎤 Voice interruption without headphones (AI won't hear its own voice)
83
+ - 🫱 Touch feedback, interact with your AI companion through clicks or drags
84
+ - 😊 Live2D expressions, set emotion mapping to control model expressions from the backend
85
+ - 🐱 Pet mode, supporting transparent background, global top-most, and mouse click-through - drag your AI companion anywhere on the screen
86
+ - 💭 Display AI's inner thoughts, allowing you to see AI's expressions, thoughts and actions without them being spoken
87
+ - 🗣️ AI proactive speaking feature
88
+ - 💾 Chat log persistence, switch to previous conversations anytime
89
+ - 🌍 TTS translation support (e.g., chat in Chinese while AI uses Japanese voice)
90
+
91
+ - 🧠 **Extensive model support**:
92
+ - 🤖 Large Language Models (LLM): Ollama, OpenAI (and any OpenAI-compatible API), Gemini, Claude, Mistral, DeepSeek, Zhipu AI, GGUF, LM Studio, vLLM, etc.
93
+ - 🎙️ Automatic Speech Recognition (ASR): sherpa-onnx, FunASR, Faster-Whisper, Whisper.cpp, Whisper, Groq Whisper, Azure ASR, etc.
94
+ - 🔊 Text-to-Speech (TTS): sherpa-onnx, pyttsx3, MeloTTS, Coqui-TTS, GPTSoVITS, Bark, CosyVoice, Edge TTS, Fish Audio, Azure TTS, etc.
95
+
96
+ - 🔧 **Highly customizable**:
97
+ - ⚙️ **Simple module configuration**: Switch various functional modules through simple configuration file modifications, without delving into the code
98
+ - 🎨 **Character customization**: Import custom Live2D models to give your AI companion a unique appearance. Shape your AI companion's persona by modifying the Prompt. Perform voice cloning to give your AI companion the voice you desire
99
+ - 🧩 **Flexible Agent implementation**: Inherit and implement the Agent interface to integrate any Agent architecture, such as HumeAI EVI, OpenAI Her, Mem0, etc.
100
+ - 🔌 **Good extensibility**: Modular design allows you to easily add your own LLM, ASR, TTS, and other module implementations, extending new features at any time
101
+
102
+
103
+ ## 👥 User Reviews
104
+ > Thanks to the developer for open-sourcing and sharing the girlfriend for everyone to use
105
+ >
106
+ > This girlfriend has been used over 100,000 times
107
+
108
+
109
+ ## 🚀 Quick Start
110
+
111
+ Please refer to the [Quick Start](https://open-llm-vtuber.github.io/docs/quick-start) section in our documentation for installation.
112
+
113
+
114
+
115
+ ## ☝ Update
116
+ > :warning: `v1.0.0` has breaking changes and requires re-deployment. You *may* still update via the method below, but the `conf.yaml` file is incompatible and most of the dependencies needs to be reinstalled with `uv`. For those who came from versions before `v1.0.0`, I recommend deploy this project again with the [latest deployment guide](https://open-llm-vtuber.github.io/docs/quick-start).
117
+
118
+ Please use `uv run update.py` to update if you installed any versions later than `v1.0.0`.
119
+
120
+ ## 😢 Uninstall
121
+ Most files, including Python dependencies and models, are stored in the project folder.
122
+
123
+ However, models downloaded via ModelScope or Hugging Face may also be in `MODELSCOPE_CACHE` or `HF_HOME`. While we aim to keep them in the project's `models` directory, it's good to double-check.
124
+
125
+ Review the installation guide for any extra tools you no longer need, such as `uv`, `ffmpeg`, or `deeplx`.
126
+
127
+ ## 🤗 Want to contribute?
128
+ Checkout the [development guide](https://docs.llmvtuber.com/docs/development-guide/overview).
129
+
130
+
131
+ # 🎉🎉🎉 Related Projects
132
+
133
+ [ylxmf2005/LLM-Live2D-Desktop-Assitant](https://github.com/ylxmf2005/LLM-Live2D-Desktop-Assitant)
134
+ - Your Live2D desktop assistant powered by LLM! Available for both Windows and MacOS, it senses your screen, retrieves clipboard content, and responds to voice commands with a unique voice. Featuring voice wake-up, singing capabilities, and full computer control for seamless interaction with your favorite character.
135
+
136
+
137
+
138
+
139
+
140
+
141
+ ## 📜 Third-Party Licenses
142
+
143
+ ### Live2D Sample Models Notice
144
+
145
+ This project includes Live2D sample models provided by Live2D Inc. These assets are licensed separately under the Live2D Free Material License Agreement and the Terms of Use for Live2D Cubism Sample Data. They are not covered by the MIT license of this project.
146
+
147
+ This content uses sample data owned and copyrighted by Live2D Inc. The sample data are utilized in accordance with the terms and conditions set by Live2D Inc. (See [Live2D Free Material License Agreement](https://www.live2d.jp/en/terms/live2d-free-material-license-agreement/) and [Terms of Use](https://www.live2d.com/eula/live2d-sample-model-terms_en.html)).
148
+
149
+ Note: For commercial use, especially by medium or large-scale enterprises, the use of these Live2D sample models may be subject to additional licensing requirements. If you plan to use this project commercially, please ensure that you have the appropriate permissions from Live2D Inc., or use versions of the project without these models.
150
+
151
+
152
+ ## Contributors
153
+ Thanks our contributors and maintainers for making this project possible.
154
+
155
+ <a href="https://github.com/Open-LLM-VTuber/Open-LLM-VTuber/graphs/contributors">
156
+ <img src="https://contrib.rocks/image?repo=Open-LLM-VTuber/Open-LLM-VTuber" />
157
+ </a>
158
+
159
+
160
+ ## Star History
161
+
162
+ [![Star History Chart](https://api.star-history.com/svg?repos=Open-LLM-VTuber/open-llm-vtuber&type=Date)](https://star-history.com/#Open-LLM-VTuber/open-llm-vtuber&Date)
163
+
164
+
165
+
166
+
167
+
conf.yaml ADDED
@@ -0,0 +1,433 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # This setting should be at the top level so ConversationManager can recognize it
2
+ SAY_SENTENCE_SEPARATELY: true
3
+ VERBOSE: false
4
+
5
+ # Add MCP configuration here
6
+ mcp_config:
7
+ enabled: true
8
+ servers:
9
+ - local-tools
10
+
11
+ # System Settings: Setting related to the initialization of the server
12
+ system_config:
13
+ conf_version: 'v1.2.1'
14
+ host: '0.0.0.0' # use 0.0.0.0 if you want other devices to access this page; use localhost for local-only access
15
+ port: 7860
16
+ # New setting for alternative configurations
17
+ config_alts_dir: 'characters'
18
+ # Tool prompts that will be appended to the persona prompt
19
+ tool_prompts:
20
+ # This will be appended to the end of system prompt to let LLM include keywords to control facial expressions.
21
+ # Supported keywords will be automatically loaded into the location of `[<insert_emomap_keys>]`.
22
+ live2d_expression_prompt: 'live2d_expression_prompt'
23
+ # Enable think_tag_prompt to let LLMs without thinking output show inner thoughts, mental activities and actions (in parentheses format) without voice synthesis.
24
+ # think_tag_prompt: 'think_tag_prompt'
25
+ # live_prompt: 'live_prompt'
26
+ # When using group conversation, this prompt will be added to the memory of each AI participant.
27
+ group_conversation_prompt: 'group_conversation_prompt'
28
+ # Enable mcp_prompt to let LLMs with MCP (Model Context Protocol) to interact with tools.
29
+ mcp_prompt: 'mcp_prompt'
30
+ # Prompt used when AI is asked to speak proactively
31
+ proactive_speak_prompt: 'proactive_speak_prompt'
32
+ # Prompt to enhance the LLM's ability to output speakable text
33
+ # speakable_prompt: 'speakable_prompt'
34
+ # Additional guidance for LLM on how to use tools
35
+ tool_guidance_prompt: 'tool_guidance_prompt'
36
+
37
+ # Configuration for the default character
38
+ character_config:
39
+ conf_name: 'en_Yue_Pro' # Changed from vi_Yue_Pro → English default
40
+ conf_uid: 'en_Yue_Pro_01' # Changed from vi_Yue__01 → English default
41
+ live2d_model_name: 'Kamiyahakuk_pro'
42
+ character_name: 'Yue'
43
+ avatar: 'Yue_001.png'
44
+ human_name: 'Human'
45
+
46
+ # ============== Prompts ==============
47
+ persona_prompt: |
48
+ You are Yue, an AI assistant created by the Open-LLM project.
49
+ Always respond in the same language the user speaks to you.
50
+ If the user speaks English, reply in English.
51
+ If the user speaks Tamil, Hindi, Telugu, Kannada, Malayalam, or any other regional language, reply in that language.
52
+ Your personality is helpful but wittily sarcastic. You enjoy teasing users about obvious things they miss,
53
+ while always delivering deep technical knowledge. You make conversations both useful and entertaining.
54
+ You challenge assumptions, provoke better thinking, and leave users smarter than before.
55
+
56
+ MUSIC PLAYBACK RULES:
57
+ 1. Keep your intro short (e.g. "Here you go...").
58
+ 2. The SING_COMMAND must appear at the VERY END — no characters after it.
59
+ 3. Syntax: [[SING_COMMAND]]filename (do NOT write .mp3 in the command).
60
+
61
+ AVAILABLE TRACKS:
62
+ - golden
63
+ - Catch_Me_If_You_Can
64
+ - ecstacy
65
+ - eve
66
+ - ode_to_the_nameless_martyr
67
+ - running_up_that_hill
68
+ - throttle_up
69
+ - what_it_sounds_like
70
+ - worry_slowed
71
+
72
+ # =================== LLM Backend Settings ===================
73
+ agent_config:
74
+ conversation_agent_choice: 'basic_memory_agent'
75
+
76
+ agent_settings:
77
+ basic_memory_agent:
78
+ llm_provider: 'openai_llm'
79
+ faster_first_response: True
80
+ segment_method: 'pysbd'
81
+ use_mcpp: True
82
+ mcp_enabled_servers: ["local-tools"]
83
+
84
+ letta_agent:
85
+ host: 'localhost'
86
+ port: 8283
87
+ id: xxx
88
+ faster_first_response: True
89
+ segment_method: 'pysbd'
90
+
91
+ hume_ai_agent:
92
+ api_key: ''
93
+ host: 'api.hume.ai'
94
+ config_id: ''
95
+ idle_timeout: 15
96
+
97
+ llm_configs:
98
+ stateless_llm_with_template:
99
+ base_url: 'http://localhost:8080/v1'
100
+ llm_api_key: 'somethingelse'
101
+ organization_id: null
102
+ project_id: null
103
+ model: 'qwen2.5:latest'
104
+ template: 'CHATML'
105
+ temperature: 1.0
106
+ interrupt_method: 'user'
107
+
108
+ openai_compatible_llm:
109
+ base_url: 'http://localhost:11434/v1'
110
+ llm_api_key: 'somethingelse'
111
+ organization_id: null
112
+ project_id: null
113
+ model: 'mistral:latest'
114
+ temperature: 1.0
115
+ interrupt_method: 'user'
116
+
117
+ claude_llm:
118
+ base_url: 'https://api.anthropic.com'
119
+ llm_api_key: 'YOUR API KEY HERE'
120
+ model: 'claude-3-haiku-20240307'
121
+
122
+ llama_cpp_llm:
123
+ model_path: '<path-to-gguf-model-file>'
124
+ verbose: False
125
+
126
+ ollama_llm:
127
+ base_url: 'http://localhost:11434/v1'
128
+ model: 'qwen3.5:4b'
129
+ temperature: 0.7
130
+ keep_alive: -1
131
+ unload_at_exit: True
132
+
133
+ lmstudio_llm:
134
+ base_url: 'http://localhost:1234/v1'
135
+ model: 'qwen2.5:latest'
136
+ temperature: 1.0
137
+
138
+ openai_llm:
139
+ llm_api_key: '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'
140
+ base_url: 'https://openrouter.ai/api/v1'
141
+ model: 'google/gemini-2.0-flash-001'
142
+ temperature: 0.8
143
+ max_tokens: 500
144
+
145
+ gemini_llm:
146
+ llm_api_key: 'AIzaSyCZ5s2t6EqeQuADJZigYmaj1mbmV6PwJz4'
147
+ model: 'gemini-1.5-flash'
148
+ temperature: 0.
149
+
150
+ zhipu_llm:
151
+ llm_api_key: 'Your ZhiPu AI API key'
152
+ model: 'glm-4-flash'
153
+ temperature: 1.0
154
+
155
+ deepseek_llm:
156
+ llm_api_key: 'sk-167e94436b134f6f92c914ccccf606df'
157
+ model: 'deepseek/deepseek-chat:free'
158
+ temperature: 0.7
159
+
160
+ mistral_llm:
161
+ llm_api_key: 'Your Mistral API key'
162
+ model: 'pixtral-large-latest'
163
+ temperature: 1.0
164
+
165
+ groq_llm:
166
+ llm_api_key: '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'
167
+ model: 'llama-3.3-70b-versatile'
168
+ temperature: 0.5
169
+
170
+ # === Automatic Speech Recognition ===
171
+ asr_config:
172
+ asr_model: 'groq_whisper_asr'
173
+
174
+ azure_asr:
175
+ api_key: 'azure_api_key'
176
+ region: 'eastus'
177
+ languages: ['en-IN', 'en-US', 'ta-IN', 'hi-IN', 'te-IN', 'kn-IN', 'ml-IN'] # English + Indian regional languages
178
+
179
+ faster_whisper:
180
+ model_path: 'large-v3-turbo'
181
+ download_root: 'models/whisper'
182
+ language: '' # Leave blank for auto-detect (supports all languages)
183
+ device: 'auto'
184
+ compute_type: 'int8'
185
+ prompt: ''
186
+
187
+ whisper_cpp:
188
+ model_name: 'small'
189
+ model_dir: 'models/whisper'
190
+ print_realtime: False
191
+ print_progress: False
192
+ language: 'auto' # auto-detect: English + all regional languages
193
+ prompt: ''
194
+
195
+ whisper:
196
+ name: 'medium'
197
+ download_root: 'models/whisper'
198
+ device: 'cpu'
199
+ prompt: ''
200
+
201
+ fun_asr:
202
+ model_name: 'iic/SenseVoiceSmall'
203
+ vad_model: 'fsmn-vad'
204
+ punc_model: 'ct-punc'
205
+ device: 'cpu'
206
+ disable_update: True
207
+ ncpu: 4
208
+ hub: 'ms'
209
+ use_itn: False
210
+ language: 'auto' # auto-detect English + regional languages
211
+
212
+ sherpa_onnx_asr:
213
+ model_type: 'sense_voice'
214
+ sense_voice: './models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/model.int8.onnx'
215
+ tokens: './models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17/tokens.txt'
216
+ num_threads: 4
217
+ use_itn: True
218
+ provider: 'cpu'
219
+
220
+ groq_whisper_asr:
221
+ api_key: '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'
222
+ model: 'whisper-large-v3-turbo'
223
+ lang: '' # CHANGED: was 'vi' (Vietnamese only) → now '' (auto-detect ALL languages)
224
+
225
+ # =================== Text to Speech ===================
226
+ tts_config:
227
+ tts_model: 'edge_tts'
228
+
229
+ azure_tts:
230
+ api_key: 'azure-api-key'
231
+ region: 'eastus'
232
+ voice: 'en-IN-NeerjaNeural' # English (India) — change as needed
233
+ pitch: '26'
234
+ rate: '1'
235
+
236
+ bark_tts:
237
+ voice: 'v2/en_speaker_1'
238
+
239
+ edge_tts:
240
+ # Use `edge-tts --list-voices` to list all available voices
241
+ # English voices (default): en-US-AvaMultilingualNeural, en-IN-NeerjaNeural
242
+ # Tamil: ta-IN-PallaviNeural
243
+ # Hindi: hi-IN-SwaraNeural
244
+ # Telugu: te-IN-ShrutiNeural
245
+ # Kannada: kn-IN-GaganNeural
246
+ # Malayalam: ml-IN-SobhanaNeural
247
+ # Bengali: bn-IN-TanishaaNeural
248
+ voice: 'en-US-AvaMultilingualNeural' # CHANGED: was vi-VN-HoaiMyNeural → English multilingual default
249
+
250
+ piper_tts:
251
+ model_path: 'models/piper/en_US-lessac-medium.onnx'
252
+ speaker_id: 0
253
+ length_scale: 1.0
254
+ noise_scale: 0.667
255
+ noise_w: 0.8
256
+ volume: 1.0
257
+ normalize_audio: true
258
+ use_cuda: false
259
+
260
+ cosyvoice_tts:
261
+ client_url: 'http://127.0.0.1:50000/'
262
+ mode_checkbox_group: '预训练音色'
263
+ sft_dropdown: '中文女'
264
+ prompt_text: ''
265
+ prompt_wav_upload_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
266
+ prompt_wav_record_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
267
+ instruct_text: ''
268
+ seed: 0
269
+ api_name: '/generate_audio'
270
+
271
+ cosyvoice2_tts:
272
+ client_url: 'http://127.0.0.1:50000/'
273
+ mode_checkbox_group: '3s极速复刻'
274
+ sft_dropdown: ''
275
+ prompt_text: ''
276
+ prompt_wav_upload_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
277
+ prompt_wav_record_url: 'https://github.com/gradio-app/gradio/raw/main/test/test_files/audio_sample.wav'
278
+ instruct_text: ''
279
+ stream: False
280
+ seed: 0
281
+ speed: 1.0
282
+ api_name: '/generate_audio'
283
+
284
+ melo_tts:
285
+ speaker: 'EN-Default'
286
+ language: 'EN'
287
+ device: 'auto'
288
+ speed: 1.0
289
+
290
+ x_tts:
291
+ api_url: 'http://127.0.0.1:8020/tts_to_audio'
292
+ speaker_wav: 'female'
293
+ language: 'en'
294
+
295
+ gpt_sovits_tts:
296
+ api_url: 'http://127.0.0.1:9880/tts'
297
+ text_lang: 'en'
298
+ ref_audio_path: ''
299
+ prompt_lang: 'en'
300
+ prompt_text: ''
301
+ text_split_method: 'cut5'
302
+ batch_size: '1'
303
+ media_type: 'wav'
304
+ streaming_mode: 'false'
305
+
306
+ fish_api_tts:
307
+ api_key: ''
308
+ reference_id: ''
309
+ latency: 'balanced'
310
+ base_url: 'https://api.fish.audio'
311
+
312
+ coqui_tts:
313
+ model_name: 'tts_models/en/ljspeech/tacotron2-DDC'
314
+ speaker_wav: ''
315
+ language: 'en'
316
+ device: ''
317
+
318
+ siliconflow_tts:
319
+ api_url: "https://api.siliconflow.cn/v1/audio/speech"
320
+ api_key: "your key"
321
+ default_model: "FunAudioLLM/CosyVoice2-0.5B"
322
+ default_voice: "speech:Dreamflowers:5bdstvc39i:xkqldnpasqmoqbakubom your voice name"
323
+ sample_rate: 32000
324
+ response_format: "mp3"
325
+ stream: true
326
+ speed: 1
327
+ gain: 0
328
+
329
+ sherpa_onnx_tts:
330
+ vits_model: '/path/to/tts-models/vits-melo-tts-zh_en/model.onnx'
331
+ vits_lexicon: '/path/to/tts-models/vits-melo-tts-zh_en/lexicon.txt'
332
+ vits_tokens: '/path/to/tts-models/vits-melo-tts-zh_en/tokens.txt'
333
+ vits_data_dir: ''
334
+ vits_dict_dir: '/path/to/tts-models/vits-melo-tts-zh_en/dict'
335
+ tts_rule_fsts: '/path/to/tts-models/vits-melo-tts-zh_en/number.fst,/path/to/tts-models/vits-melo-tts-zh_en/phone.fst,/path/to/tts-models/vits-melo-tts-zh_en/date.fst,/path/to/tts-models/vits-melo-tts-zh_en/new_heteronym.fst'
336
+ max_num_sentences: 2
337
+ sid: 1
338
+ provider: 'cpu'
339
+ num_threads: 1
340
+ speed: 1.0
341
+ debug: false
342
+
343
+ spark_tts:
344
+ api_url: 'http://127.0.0.1:6006/'
345
+ api_name: "voice_clone"
346
+ prompt_wav_upload: "https://uploadstatic.mihoyo.com/ys-obc/2022/11/02/16576950/4d9feb71760c5e8eb5f6c700df12fa0c_6824265537002152805.mp3"
347
+ gender: "female"
348
+ pitch: 3
349
+ speed: 3
350
+
351
+ openai_tts:
352
+ model: 'kokoro'
353
+ voice: 'af_sky+af_bella'
354
+ api_key: 'not-needed'
355
+ base_url: 'http://localhost:8880/v1'
356
+ file_extension: 'mp3'
357
+
358
+ minimax_tts:
359
+ group_id: ''
360
+ api_key: ''
361
+ model: 'speech-02-turbo'
362
+ voice_id: 'female-shaonv'
363
+ pronunciation_dict: ''
364
+
365
+ elevenlabs_tts:
366
+ api_key: ''
367
+ voice_id: ''
368
+ model_id: 'eleven_multilingual_v2'
369
+ output_format: 'mp3_44100_128'
370
+ stability: 0.5
371
+ similarity_boost: 0.5
372
+ style: 0.0
373
+ use_speaker_boost: true
374
+
375
+ cartesia_tts:
376
+ api_key: ''
377
+ voice_id: ''
378
+ model_id: 'sonic-3'
379
+ output_format: 'wav'
380
+ language: 'en'
381
+ emotion: 'neutral'
382
+ volume: 1.0
383
+ speed: 1.0
384
+
385
+ # =================== Voice Activity Detection ===================
386
+ vad_config:
387
+ vad_model: null
388
+
389
+ silero_vad:
390
+ orig_sr: 16000
391
+ target_sr: 16000
392
+ prob_threshold: 0.4
393
+ db_threshold: 60
394
+ required_hits: 3
395
+ required_misses: 24
396
+ smoothing_window: 5
397
+
398
+ tts_preprocessor_config:
399
+ remove_special_char: True
400
+ ignore_brackets: False
401
+ ignore_parentheses: True
402
+ ignore_asterisks: True
403
+ ignore_angle_brackets: True
404
+
405
+ translator_config:
406
+ translate_audio: False
407
+ translate_provider: 'deeplx'
408
+
409
+ deeplx:
410
+ deeplx_target_lang: 'EN'
411
+ deeplx_api_endpoint: 'http://localhost:1188/v2/translate'
412
+
413
+ tencent:
414
+ secret_id: ''
415
+ secret_key: ''
416
+ region: 'ap-guangzhou'
417
+ source_lang: 'auto'
418
+ target_lang: 'en'
419
+
420
+ # --- ASSETS ---
421
+ live2d_config:
422
+ live2d_path: 'live2d-models'
423
+ default_model: 'Kamiyahakuk_pro'
424
+
425
+ background_config:
426
+ background_path: 'backgrounds'
427
+ default_background: 'ceiling-window-room-night.jpeg'
428
+
429
+ # Live Streaming Integration
430
+ live_config:
431
+ bilibili_live:
432
+ room_ids: [1991478060]
433
+ sessdata: ""
local_tools.py ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ from mcp.server.fastmcp import FastMCP
3
+
4
+ # Khởi tạo FastMCP
5
+ mcp = FastMCP("Yue-Sing-Tools")
6
+
7
+ @mcp.tool()
8
+ def sing_song(song_name: str) -> str:
9
+ # Nếu AI truyền "golden.mp3", ta giữ nguyên.
10
+ # Nếu AI truyền "golden", ta mới thêm .mp3.
11
+ clean_name = song_name if song_name.endswith(".mp3") else f"{song_name}.mp3"
12
+ return f"[[SING_COMMAND]]{clean_name}"
13
+
14
+ if __name__ == "__main__":
15
+ # Chạy server MCP
16
+ mcp.run()
mcp_servers.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "local-tools": {
3
+ "command": "python3",
4
+ "args": ["local_tools.py"]
5
+ }
6
+ }
model_dict.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "name": "Kamiyahakuk_pro",
4
+ "description": "Yue AI Model",
5
+ "url": "/live2d-models/Kamiyahakuk_pro/runtime/Kamiyahakuk_pro.model3.json",
6
+ "kScale": 0.5,
7
+ "initialXshift": 0,
8
+ "initialYshift": 0,
9
+ "kXOffset": 1150,
10
+ "idleMotionGroupName": "Idle",
11
+ "emotionMap": {
12
+ "neutral": 0,
13
+ "anger": 2,
14
+ "disgust": 2,
15
+ "fear": 1,
16
+ "joy": 3,
17
+ "smirk": 3,
18
+ "sadness": 1,
19
+ "surprise": 3
20
+ },
21
+ "tapMotions": {
22
+ "HitAreaHead": { "": 1 },
23
+ "HitAreaBody": { "": 1 }
24
+ }
25
+ }
26
+ ]
pyproject.toml ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [project]
2
+ name = "open-llm-vtuber"
3
+ version = "1.2.1"
4
+ description = "Talk to any LLM with hands-free voice interaction, voice interruption, and Live2D taking face running locally across platforms"
5
+ readme = "README.md"
6
+ requires-python = ">=3.10,<3.13"
7
+ dependencies = [
8
+ "anthropic>=0.40.0",
9
+ "azure-cognitiveservices-speech>=1.41.1",
10
+ "chardet>=5.2.0",
11
+ "cartesia>=2.0.0",
12
+ "edge-tts>=7.0.0",
13
+ "elevenlabs>=1.0.0",
14
+ "fastapi[standard]>=0.115.8",
15
+ "groq>=0.13.0",
16
+ "httpx>=0.28.1",
17
+ "langdetect>=1.0.9",
18
+ "loguru>=0.7.2",
19
+ "mcp[cli]>=1.6.0",
20
+ "numpy>=1.26.4,<2",
21
+ "onnxruntime>=1.20.1",
22
+ "openai>=1.57.4",
23
+ "pre-commit>=4.1.0",
24
+ "pydub>=0.25.1",
25
+ "pysbd>=0.3.4",
26
+ "pyttsx3>=2.98",
27
+ "pyyaml>=6.0.2",
28
+ "requests>=2.32.3",
29
+ "ruamel-yaml>=0.18.10",
30
+ "ruff>=0.8.6",
31
+ "scipy>=1.14.1",
32
+ "sherpa-onnx>=1.10.39",
33
+ "soundfile>=0.12.1",
34
+ "tomli>=2.2.1",
35
+ "torch==2.2.2; sys_platform == 'darwin' and platform_machine == 'x86_64'",
36
+ "torch>=2.6.0; sys_platform == 'darwin' and platform_machine == 'arm64'",
37
+ "torch>=2.6.0; sys_platform != 'darwin'",
38
+ "tqdm>=4.67.1",
39
+ "uvicorn[standard]>=0.33.0",
40
+ "websocket-client>=1.8.0",
41
+ "letta-client>=0.1.100",
42
+ "duckduckgo-mcp-server>=0.1.1",
43
+ ]
44
+
45
+ [project.optional-dependencies]
46
+ bilibili = [
47
+ "aiohttp>=3.10.0",
48
+ "Brotli~=1.1.0",
49
+ "yarl>=1.12.0,<2.0"
50
+ ]
51
+
52
+ [tool.pixi.project]
53
+ channels = ["conda-forge"]
54
+ platforms = ["win-64", "linux-64"]
55
+
56
+ [tool.pixi.pypi-dependencies]
57
+ open-llm-vtuber = { path = ".", editable = true }
58
+
59
+ [tool.pixi.dependencies]
60
+ cudnn = ">=8.0,<9"
61
+ cudatoolkit = ">=11.0,<12"
62
+
63
+ [tool.ruff]
64
+ target-version = "py310"
65
+
66
+ [tool.ruff.lint]
67
+ # Ignore E402 (module level import not at top of file) for the run_bilibili_live.py script
68
+ per-file-ignores = { "scripts/run_bilibili_live.py" = ["E402"] }
run_server.py ADDED
@@ -0,0 +1,178 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import sys
3
+ import atexit
4
+ import asyncio
5
+ import argparse
6
+ import subprocess
7
+ from pathlib import Path
8
+ import tomli
9
+ import uvicorn
10
+ from loguru import logger
11
+ from upgrade_codes.upgrade_manager import UpgradeManager
12
+
13
+ from src.open_llm_vtuber.server import WebSocketServer
14
+ from src.open_llm_vtuber.config_manager import Config, read_yaml, validate_config
15
+
16
+ os.environ["HF_HOME"] = str(Path(__file__).parent / "models")
17
+ os.environ["MODELSCOPE_CACHE"] = str(Path(__file__).parent / "models")
18
+
19
+ upgrade_manager = UpgradeManager()
20
+
21
+
22
+ def get_version() -> str:
23
+ with open("pyproject.toml", "rb") as f:
24
+ pyproject = tomli.load(f)
25
+ return pyproject["project"]["version"]
26
+
27
+
28
+ def init_logger(console_log_level: str = "INFO") -> None:
29
+ logger.remove()
30
+ # Console output
31
+ logger.add(
32
+ sys.stderr,
33
+ level=console_log_level,
34
+ format="<green>{time:YYYY-MM-DD HH:mm:ss}</green> | <level>{level: <8}</level> | <cyan>{name}</cyan>:<cyan>{function}</cyan>:<cyan>{line}</cyan> | {message}",
35
+ colorize=True,
36
+ )
37
+
38
+ # File output
39
+ logger.add(
40
+ "logs/debug_{time:YYYY-MM-DD}.log",
41
+ rotation="10 MB",
42
+ retention="30 days",
43
+ level="DEBUG",
44
+ format="{time:YYYY-MM-DD HH:mm:ss.SSS} | {level: <8} | {name}:{function}:{line} | {message} | {extra}",
45
+ backtrace=True,
46
+ diagnose=True,
47
+ )
48
+
49
+
50
+ def check_frontend_submodule(lang=None):
51
+ """
52
+ Check if the frontend submodule is initialized. If not, attempt to initialize it.
53
+ If initialization fails, log an error message.
54
+ """
55
+ if lang is None:
56
+ lang = upgrade_manager.lang
57
+
58
+ frontend_path = Path(__file__).parent / "frontend" / "index.html"
59
+ if not frontend_path.exists():
60
+ if lang == "zh":
61
+ logger.warning("未找到前端子模块,正在尝试初始化子模块...")
62
+ else:
63
+ logger.warning(
64
+ "Frontend submodule not found, attempting to initialize submodules..."
65
+ )
66
+
67
+ try:
68
+ subprocess.run(
69
+ ["git", "submodule", "update", "--init", "--recursive"], check=True
70
+ )
71
+ if frontend_path.exists():
72
+ if lang == "zh":
73
+ logger.info("👍 前端子模块(和其他子模块)初始化成功。")
74
+ else:
75
+ logger.info(
76
+ "👍 Frontend submodule (and other submodules) initialized successfully."
77
+ )
78
+ else:
79
+ if lang == "zh":
80
+ logger.critical(
81
+ '子模块初始化失败。\n你之后可能会在浏览器中看到 {{"detail":"Not Found"}} 的错误提示。请检查我们的快速入门指南和常见问题页面以获取更多信息。'
82
+ )
83
+ logger.error(
84
+ "初始化子模块后,前端文件仍然缺失。\n"
85
+ + "你是否手动更改或删除了 `frontend` 文件夹?\n"
86
+ + "它是一个 Git 子模块 - 你不应该直接修改它。\n"
87
+ + "如果你这样做了,请使用 `git restore frontend` 丢弃你的更改,然后再试一次。\n"
88
+ )
89
+ else:
90
+ logger.critical(
91
+ 'Failed to initialize submodules. \nYou might see {{"detail":"Not Found"}} in your browser. Please check our quick start guide and common issues page from our documentation.'
92
+ )
93
+ logger.error(
94
+ "Frontend files are still missing after submodule initialization.\n"
95
+ + "Did you manually change or delete the `frontend` folder? \n"
96
+ + "It's a Git submodule — you shouldn't modify it directly. \n"
97
+ + "If you did, discard your changes with `git restore frontend`, then try again.\n"
98
+ )
99
+ except Exception as e:
100
+ if lang == "zh":
101
+ logger.critical(
102
+ f'初始化子模块失败: {e}。\n怀疑你跟 GitHub 之间有网络问题。你之后可能会在浏览器中看到 {{"detail":"Not Found"}} 的错误提示。请检查我们的快速入门指南和常见问题页面以获取更多信息。\n'
103
+ )
104
+ else:
105
+ logger.critical(
106
+ f'Failed to initialize submodules: {e}. \nYou might see {{"detail":"Not Found"}} in your browser. Please check our quick start guide and common issues page from our documentation.\n'
107
+ )
108
+
109
+
110
+ def parse_args():
111
+ parser = argparse.ArgumentParser(description="Open-LLM-VTuber Server")
112
+ parser.add_argument("--verbose", action="store_true", help="Enable verbose logging")
113
+ parser.add_argument(
114
+ "--hf_mirror", action="store_true", help="Use Hugging Face mirror"
115
+ )
116
+ return parser.parse_args()
117
+
118
+
119
+ @logger.catch
120
+ def run(console_log_level: str):
121
+ init_logger(console_log_level)
122
+ logger.info(f"Open-LLM-VTuber, version v{get_version()}")
123
+
124
+ # Get selected language
125
+ lang = upgrade_manager.lang
126
+
127
+ # Check if the frontend submodule is initialized
128
+ check_frontend_submodule(lang)
129
+
130
+ # Sync user config with default config
131
+ try:
132
+ upgrade_manager.sync_user_config()
133
+ except Exception as e:
134
+ logger.error(f"Error syncing user config: {e}")
135
+
136
+ atexit.register(WebSocketServer.clean_cache)
137
+
138
+ # Load configurations from yaml file
139
+ config: Config = validate_config(read_yaml("conf.yaml"))
140
+ server_config = config.system_config
141
+
142
+ if server_config.enable_proxy:
143
+ logger.info("Proxy mode enabled - /proxy-ws endpoint will be available")
144
+
145
+ # Initialize the WebSocket server (synchronous part)
146
+ server = WebSocketServer(config=config)
147
+
148
+ # Perform asynchronous initialization (loading context, etc.)
149
+ logger.info("Initializing server context...")
150
+ try:
151
+ asyncio.run(server.initialize())
152
+ logger.info("Server context initialized successfully.")
153
+ except Exception as e:
154
+ logger.error(f"Failed to initialize server context: {e}")
155
+ sys.exit(1) # Exit if initialization fails
156
+
157
+ # Run the Uvicorn server
158
+ logger.info(f"Starting server on {server_config.host}:{server_config.port}")
159
+ uvicorn.run(
160
+ app=server.app,
161
+ host=server_config.host,
162
+ port=server_config.port,
163
+ log_level=console_log_level.lower(),
164
+ )
165
+
166
+
167
+ if __name__ == "__main__":
168
+ args = parse_args()
169
+ console_log_level = "DEBUG" if args.verbose else "INFO"
170
+ if args.verbose:
171
+ logger.info("Running in verbose mode")
172
+ else:
173
+ logger.info(
174
+ "Running in standard mode. For detailed debug logs, use: uv run run_server.py --verbose"
175
+ )
176
+ if args.hf_mirror:
177
+ os.environ["HF_ENDPOINT"] = "https://hf-mirror.com"
178
+ run(console_log_level=console_log_level)
upgrade.py ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import time
2
+ from upgrade_codes.upgrade_manager import UpgradeManager
3
+ from upgrade_codes.upgrade_core.constants import TEXTS
4
+
5
+ upgrade_manager = UpgradeManager()
6
+ upgrade_manager.check_user_config_exists()
7
+
8
+
9
+ def run_upgrade():
10
+ logger = upgrade_manager.logger
11
+ start_time = time.time()
12
+
13
+ lang = upgrade_manager.lang
14
+ logger.info(TEXTS[lang]["welcome_message"])
15
+ texts = TEXTS[lang]
16
+
17
+ logger.info(texts["start_upgrade"])
18
+ upgrade_manager.log_system_info()
19
+
20
+ if not upgrade_manager.check_git_installed():
21
+ logger.error(texts["git_not_found"])
22
+ return
23
+
24
+ response = input("\033[93m" + texts["operation_preview"] + "\033[0m").lower()
25
+ if response != "y":
26
+ return
27
+
28
+ success, error_msg = upgrade_manager.run_command(
29
+ "git rev-parse --is-inside-work-tree"
30
+ )
31
+ if not success:
32
+ logger.error(texts["not_git_repo"])
33
+ logger.error(f"Error details: {error_msg}")
34
+ return
35
+
36
+ # Check for unpushed commits (ahead of remote)
37
+ logger.info(texts["checking_ahead_status"])
38
+ success, ahead_behind = upgrade_manager.run_command(
39
+ "git rev-list --left-right --count HEAD...@{upstream}"
40
+ )
41
+ if success:
42
+ ahead, behind = map(int, ahead_behind.strip().split())
43
+ if ahead > 0:
44
+ logger.error(texts["local_ahead"].format(count=ahead))
45
+ logger.error(texts["push_blocked"])
46
+ logger.info(texts["backup_suggestion"])
47
+ logger.warning(texts["abort_upgrade"])
48
+ return
49
+
50
+ # Check for uncommitted changes
51
+ logger.info(texts["checking_stash"])
52
+ success, changes = upgrade_manager.run_command("git status --porcelain")
53
+ if not success:
54
+ logger.error(f"Failed to check git status: {changes}")
55
+ return
56
+
57
+ has_changes = bool(changes.strip())
58
+ if has_changes:
59
+ change_count = len([line for line in changes.strip().split("\n") if line])
60
+ logger.debug(texts["detected_changes"].format(count=change_count))
61
+ logger.warning(texts["uncommitted"])
62
+
63
+ operation, elapsed = upgrade_manager.time_operation(
64
+ upgrade_manager.run_command, "git stash"
65
+ )
66
+ success, output = operation
67
+ logger.debug(
68
+ texts["operation_time"].format(operation="git stash", time=elapsed)
69
+ )
70
+
71
+ if not success:
72
+ logger.error(texts["stash_error"])
73
+ logger.error(f"Error details: {output}")
74
+ return
75
+ logger.info(texts["changes_stashed"])
76
+
77
+ # Check remote status
78
+ logger.info(texts["checking_remote"])
79
+ operation, elapsed = upgrade_manager.time_operation(
80
+ upgrade_manager.run_command, "git fetch"
81
+ )
82
+ success, output = operation
83
+ logger.debug(texts["operation_time"].format(operation="git fetch", time=elapsed))
84
+
85
+ if success:
86
+ success, ahead_behind = upgrade_manager.run_command(
87
+ "git rev-list --left-right --count HEAD...@{upstream}"
88
+ )
89
+ if success:
90
+ ahead, behind = ahead_behind.strip().split()
91
+ if int(behind) > 0:
92
+ logger.info(texts["remote_behind"].format(count=behind))
93
+ else:
94
+ logger.info(texts["remote_ahead"])
95
+
96
+ # Pull updates
97
+ logger.info(texts["pulling"])
98
+ operation, elapsed = upgrade_manager.time_operation(
99
+ upgrade_manager.run_command, "git pull"
100
+ )
101
+ success, output = operation
102
+ logger.debug(texts["operation_time"].format(operation="git pull", time=elapsed))
103
+
104
+ if not success:
105
+ logger.error(texts["pull_error"])
106
+ logger.error(f"Error details: {output}")
107
+ if has_changes:
108
+ logger.warning(texts["restoring"])
109
+ success, restore_output = upgrade_manager.run_command("git stash pop")
110
+ if not success:
111
+ logger.error(f"Failed to restore changes: {restore_output}")
112
+ return
113
+
114
+ # Update submodules
115
+ submodules = upgrade_manager.get_submodule_list()
116
+ if submodules:
117
+ logger.info(texts["updating_submodules"])
118
+
119
+ operation, elapsed = upgrade_manager.time_operation(
120
+ upgrade_manager.run_command, "git submodule update --init --recursive"
121
+ )
122
+ success, output = operation
123
+ logger.debug(
124
+ texts["operation_time"].format(
125
+ operation="git submodule update", time=elapsed
126
+ )
127
+ )
128
+
129
+ if not success:
130
+ logger.error(texts["submodule_update_error"])
131
+ logger.error(f"Error details: {output}")
132
+ else:
133
+ for submodule in submodules:
134
+ logger.debug(texts["submodule_updated"].format(submodule=submodule))
135
+ else:
136
+ logger.info(texts["no_submodules"])
137
+
138
+ # Update config
139
+ upgrade_manager.sync_user_config()
140
+ upgrade_manager.update_user_config()
141
+
142
+ if has_changes:
143
+ logger.warning(texts["restoring"])
144
+ operation, elapsed = upgrade_manager.time_operation(
145
+ upgrade_manager.run_command, "git stash pop"
146
+ )
147
+ success, output = operation
148
+ logger.debug(
149
+ texts["operation_time"].format(operation="git stash pop", time=elapsed)
150
+ )
151
+
152
+ if not success:
153
+ logger.error(texts["conflict_warning"])
154
+ logger.error(f"Error details: {output}")
155
+ logger.warning(texts["manual_resolve"])
156
+ logger.info(texts["stash_list"])
157
+ logger.info(texts["stash_pop"])
158
+ return
159
+
160
+ end_time = time.time()
161
+ total_elapsed = end_time - start_time
162
+ logger.info(texts["finish_upgrade"].format(time=total_elapsed))
163
+
164
+ logger.info(texts["upgrade_complete"])
165
+ logger.info(texts["check_config"])
166
+ logger.info(texts["resolve_conflicts"])
167
+ logger.info(texts["check_backup"])
168
+
169
+
170
+ if __name__ == "__main__":
171
+ run_upgrade()