Commit ·
a55856f
0
Parent(s):
Initial model release
Browse files- .gitattributes +36 -0
- README.md +173 -0
- chat_template.jinja +93 -0
- config.json +162 -0
- configuration_bailing_moe_v3.py +126 -0
- model-00001-of-00028.safetensors +3 -0
- model-00002-of-00028.safetensors +3 -0
- model-00003-of-00028.safetensors +3 -0
- model-00004-of-00028.safetensors +3 -0
- model-00005-of-00028.safetensors +3 -0
- model-00006-of-00028.safetensors +3 -0
- model-00007-of-00028.safetensors +3 -0
- model-00008-of-00028.safetensors +3 -0
- model-00009-of-00028.safetensors +3 -0
- model-00010-of-00028.safetensors +3 -0
- model-00011-of-00028.safetensors +3 -0
- model-00012-of-00028.safetensors +3 -0
- model-00013-of-00028.safetensors +3 -0
- model-00014-of-00028.safetensors +3 -0
- model-00015-of-00028.safetensors +3 -0
- model-00016-of-00028.safetensors +3 -0
- model-00017-of-00028.safetensors +3 -0
- model-00018-of-00028.safetensors +3 -0
- model-00019-of-00028.safetensors +3 -0
- model-00020-of-00028.safetensors +3 -0
- model-00021-of-00028.safetensors +3 -0
- model-00022-of-00028.safetensors +3 -0
- model-00023-of-00028.safetensors +3 -0
- model-00024-of-00028.safetensors +3 -0
- model-00025-of-00028.safetensors +3 -0
- model-00026-of-00028.safetensors +3 -0
- model-00027-of-00028.safetensors +3 -0
- model-00028-of-00028.safetensors +3 -0
- model.safetensors.index.json +0 -0
- modeling_bailing_moe_v3.py +1625 -0
- special_tokens_map.json +30 -0
- tokenizer.json +3 -0
- tokenizer_config.json +2114 -0
.gitattributes
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
README.md
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: mit
|
| 3 |
+
pipeline_tag: text-generation
|
| 4 |
+
---
|
| 5 |
+
<p align="center">
|
| 6 |
+
<img src="https://mdn.alipayobjects.com/huamei_qa8qxu/afts/img/A*4QxcQrBlTiAAAAAAQXAAAAgAemJ7AQ/original" width="100"/>
|
| 7 |
+
</p>
|
| 8 |
+
<p align="center">🤗 <a href="https://huggingface.co/inclusionAI">Hugging Face</a> | 🤖 <a href="https://modelscope.cn/organization/inclusionAI">ModelScope </a> | 🐙 <a href="https://openrouter.ai/inclusionai/ling-3.0-flash:free">OpenRouter </a> </p>
|
| 9 |
+
|
| 10 |
+
## Introduction
|
| 11 |
+
We're introducing Ling-3.0-flash, our next-generation native hybrid reasoning model. Operating with **124B** total and **5.1B** active parameters (~12.4% and ~8.1% of our previous 1T-class flagship Ring-2.6-1T), Ling-3.0-flash matches or outperforms its predecessor across key benchmarks.
|
| 12 |
+
|
| 13 |
+
Key highlights of the model are summarized below:
|
| 14 |
+
|
| 15 |
+
+ **Native Hybrid-Linear Architecture:** Ling-3.0 adopts a native hybrid linear attention architecture from the very start of pretraining (5:1 alternating stacking of Kimi Delta Attention (KDA) and MLA), upgraded with KDA fine-grained diagonal gating and 1/64 sparse MoE. With 124B total parameters and 5.1B activated parameters, it achieves a synergistic leap in long-context efficiency and computational cost.
|
| 16 |
+
+ **Remarkable Efficiency & Performance:** Engineered for speed, compute efficiency, and production deployment, Ling-3.0-flash delivers class-defying performance against both larger SOTA competitors and previous-generation flagships. Activating only 5.1B parameters per token, it provides impressive reasoning, instruction following, and long-context capabilities to empower complex agentic workflows in production environments.
|
| 17 |
+
+ **Comprehensive Agentic Evolution:** Tailored for real-world productivity workflows, the model incorporates 10,000+ interactive training environments to achieve end-to-end closed-loop execution across Coding, General, and Deep Research Agent tasks. It natively integrates the SGLang HiCache + Mooncake hierarchical caching architecture (featuring physical dual-pools and a cluster-shared L3 cache), eliminating redundant recomputation during long-horizon interactions and reducing Time to First Token (TTFT) by 60% to over 80% in long-input scenarios.
|
| 18 |
+
|
| 19 |
+
<!-- Benchmark comparison chart across models -->
|
| 20 |
+

|
| 21 |
+
|
| 22 |
+
## Model Overview
|
| 23 |
+
The model summary information and architecture diagram are as follows:
|
| 24 |
+
|
| 25 |
+
| Architecture | Hybrid-linear MoE |
|
| 26 |
+
| --- | --- |
|
| 27 |
+
| Parameter Scale | Total 124B, Activated 5.1B |
|
| 28 |
+
| Transformer Layers | 35 KDA + 7 Gated MLA (5:1) |
|
| 29 |
+
| Number of Dense Layers | 2 |
|
| 30 |
+
| Number of Routed Experts | 512 |
|
| 31 |
+
| Number of Shared Experts | 1 |
|
| 32 |
+
| Number of Activated Experts | 8 |
|
| 33 |
+
| Attention Heads | 32 |
|
| 34 |
+
| Hidden Size | 2560 |
|
| 35 |
+
| Expert Intermediate Size | 768 |
|
| 36 |
+
| Dense Intermediate Size | 6144 |
|
| 37 |
+
| Vocabulary Size | 157184 |
|
| 38 |
+
| Context Training Schedule | 8K -> 32K -> 256K |
|
| 39 |
+
|
| 40 |
+
<!-- Ling-3.0-flash architecture diagram -->
|
| 41 |
+

|
| 42 |
+
|
| 43 |
+
## Evaluation
|
| 44 |
+
We have conducted a comprehensive evaluation of Ling-3.0-flash across multiple authoritative benchmarks. **Ling-3.0-flash** performs strongly on representative code/agent benchmarks such as **SWE-Bench Pro, SWE-Bench Multilingual, Tau3-banking-AA**, **MCP-Atlas** and **SkillsBench, etc**. In practice, Ling-3.0-flash delivers a strong user experience across frameworks including **Claude Code**,**Kilo Code**,**Qwen Code**,**Hermes Agent**,and **OpenClaw**, etc. Beyond agentic tasks, Ling-3.0-flash also delivers strong performance across **general knowledge**,**mathematical reasoning**,**instruction following**,and **long-context understanding**.
|
| 45 |
+
|
| 46 |
+
<!-- Benchmark evaluation results comparison chart -->
|
| 47 |
+

|
| 48 |
+
|
| 49 |
+
> + Thinking mode is enabled by default. Unless otherwise specified, the default parameters for Ling-3.0-flash are as follows: `temperature=0.6, top_p=0.95, top_k=20`.
|
| 50 |
+
> + SWE-Bench Series: Evaluated using OpenHands as the agent harness with tailored prompts. Decoding uses `temperature=0.6, top_p=0.95, max_new_tokens=32K`, with a 256K context window.
|
| 51 |
+
> + Terminal-Bench 2.1: Evaluated under the Artificial Analysis (AA) protocol using the default Terminus 2 harness, a unified 2-hour timeout, the provided JSON parser in preserve-thinking mode, and 3 runs per task (mean). Decoding uses `temperature=0.6, top_p=1.0, max_new_tokens=32K`, with a 256K context window.
|
| 52 |
+
> + MiniAppBench: A 500-task coding benchmark evaluating whether models can turn a single user request into complete, usable interactive HTML apps in real-world application-generation scenarios. Evaluated with `temperature=1.0, top_p=1.0, max_tokens=128K`.
|
| 53 |
+
> + AntSWEBench: AntSWEBench is an internally used software engineering benchmark that covers mainstream programming languages such as Java, JavaScript, and Python, including various development scenarios like new feature, bug fix, and code refactoring.
|
| 54 |
+
> + Tau3-banking-AA: Aligned with the AA leaderboard, utilizing GPT-5.4-mini (medium reasoning) for both the user simulator and the natural-language assertion judge.
|
| 55 |
+
> + MCP-Atlas: Evaluated on the 500-task public set using the official v1 harness with a 20-turn limit and Gemini-2.5-Pro as the claim-coverage judger.
|
| 56 |
+
> + SkillsBench: Evaluated via kilo-code on 87 tasks (excluding external API-dependent tasks), averaged over 3 runs.
|
| 57 |
+
> + GDPval v2-AA: Evaluated on the public 220-task benchmark using the official Stirrup harness, with a 250-turn limit and a 5-hour timeout.
|
| 58 |
+
> + Search-agent: For all search‑agent tasks, evaluations are performed using an internal harness. The basic ReAct paradigm is adopted for single-agent evaluation, while a multi-agent setup is employed for BrowseComp. The reported metric is the average pass@1.
|
| 59 |
+
> - WideSearch: Evaluated using the official prompt and the official judge model GPT-4.1 on the corrected version of the dataset.
|
| 60 |
+
> - Draco: Scored based on official rubrics per question, with the final score calculated as the average across all questions using Claude Opus 4.6 as the scoring model.
|
| 61 |
+
> - BrowseComp (Single-Agent): Evaluated using a resume strategy for context management: once the context reaches a 64K-token threshold, the trajectory is summarized, the original history is discarded, and execution is resumed from the summary.
|
| 62 |
+
> - BrowseComp (Multi-Agent): Evaluated using an internal multi-agent search harness based on SearchSwarm/Tongyi DeepResearch, configured with `temperature=0.85, top_p=0.95, max_tokens=8K`, and main/sub-agent context windows of 128K and 64K, respectively.
|
| 63 |
+
>
|
| 64 |
+
|
| 65 |
+
## Quickstart
|
| 66 |
+
### SGLang
|
| 67 |
+
|
| 68 |
+
The hardware- and recipe-specific launch matrix (BF16/FP8 × Low-Latency / High-Throughput / HiCache + Mooncake), with a live command generator and verified configurations, lives in the SGLang cookbook:
|
| 69 |
+
|
| 70 |
+
**Cookbook:** https://docs.sglang.io/cookbook/autoregressive/InclusionAI/Ling-3.0-flash
|
| 71 |
+
|
| 72 |
+
#### Install SGLang
|
| 73 |
+
|
| 74 |
+
Use the pre-built image that tracks the Ling-3.0 runtime:
|
| 75 |
+
|
| 76 |
+
```bash
|
| 77 |
+
docker pull lmsysorg/sglang:dev-Ling-3.0-flash
|
| 78 |
+
```
|
| 79 |
+
|
| 80 |
+
#### Run Inference
|
| 81 |
+
|
| 82 |
+
Recommended low-latency recipe (built-in MTP / NEXTN, 256K YaRN context) on 4× 141GB-class GPUs (H20-3e) or 4-GPU Blackwell nodes:
|
| 83 |
+
|
| 84 |
+
```bash
|
| 85 |
+
docker run --rm --gpus all --ipc=host --shm-size 32g \
|
| 86 |
+
-p 30000:30000 \
|
| 87 |
+
-e HF_TOKEN=<your-hf-token> \
|
| 88 |
+
lmsysorg/sglang:dev-Ling-3.0-flash \
|
| 89 |
+
env SGLANG_ALLOW_OVERWRITE_LONGER_CONTEXT_LEN=1 \
|
| 90 |
+
python3 -m sglang.launch_server \
|
| 91 |
+
--model-path inclusionAI/Ling-3.0-flash \
|
| 92 |
+
--tp 4 \
|
| 93 |
+
--context-length 262144 \
|
| 94 |
+
--speculative-algorithm NEXTN \
|
| 95 |
+
--mem-fraction-static 0.8 \
|
| 96 |
+
--host 0.0.0.0 \
|
| 97 |
+
--port 30000
|
| 98 |
+
```
|
| 99 |
+
|
| 100 |
+
On 80GB cards (H100 / H800) use `--tp 8` with the same flags; see the cookbook cell for your hardware.
|
| 101 |
+
|
| 102 |
+
**Client**
|
| 103 |
+
|
| 104 |
+
Thinking is enabled by default by both the chat template and the `ling3` reasoning parser. Disable it per request with `"chat_template_kwargs": {"enable_thinking": false}`. We recommend the sampling parameters `temperature=0.6`, `top_p=0.95`, and `top_k=20`.
|
| 105 |
+
|
| 106 |
+
```bash
|
| 107 |
+
curl -s http://localhost:30000/v1/chat/completions \
|
| 108 |
+
-H "Content-Type: application/json" \
|
| 109 |
+
-d '{"model": "inclusionAI/Ling-3.0-flash",
|
| 110 |
+
"messages": [{"role": "user", "content": "hello!"}],
|
| 111 |
+
"stream": true,
|
| 112 |
+
"temperature": 0.6,
|
| 113 |
+
"top_k": 20,
|
| 114 |
+
"top_p": 0.95
|
| 115 |
+
}'
|
| 116 |
+
```
|
| 117 |
+
|
| 118 |
+
For `--reasoning-parser ling3` / `--tool-call-parser ling3`, the HiCache + Mooncake L3 setup, and GSM8K / `bench_serving` reproduction commands, see the cookbook page linked above.
|
| 119 |
+
|
| 120 |
+
### vLLM
|
| 121 |
+
#### Install our vLLM
|
| 122 |
+
```bash
|
| 123 |
+
pip install uv
|
| 124 |
+
|
| 125 |
+
uv venv ~/my_ling_env
|
| 126 |
+
|
| 127 |
+
source ~/my_ling_env/bin/activate
|
| 128 |
+
|
| 129 |
+
git clone -b ling_3_0 https://github.com/inclusionAI/vllm-ling-v3.git
|
| 130 |
+
|
| 131 |
+
cd vllm-ling-v3
|
| 132 |
+
|
| 133 |
+
VLLM_USE_PRECOMPILED=1 uv pip install --editable . --torch-backend=auto
|
| 134 |
+
```
|
| 135 |
+
|
| 136 |
+
#### Run Inference
|
| 137 |
+
Here is the example to run Ling-3.0-flash with 4 GPUs, where the server port is `${PORT}`:
|
| 138 |
+
|
| 139 |
+
**Server**
|
| 140 |
+
|
| 141 |
+
Since the model is trained with MTP, we recommend enabling MTP during inference (i.e., --speculative-config) for lower latency.
|
| 142 |
+
|
| 143 |
+
```bash
|
| 144 |
+
vllm serve "$MODEL_PATH" \
|
| 145 |
+
--port "$PORT" \
|
| 146 |
+
--trust-remote-code \
|
| 147 |
+
--served-model-name auto \
|
| 148 |
+
--tensor-parallel-size 4 \
|
| 149 |
+
--gpu-memory-utilization 0.85 \
|
| 150 |
+
--enable-prefix-caching \
|
| 151 |
+
--mamba-cache-mode align \
|
| 152 |
+
--enable-auto-tool-choice \
|
| 153 |
+
--tool-call-parser ling3 \
|
| 154 |
+
--reasoning-parser ling3 \
|
| 155 |
+
--speculative-config '{"method":"mtp","num_speculative_tokens":3}'
|
| 156 |
+
```
|
| 157 |
+
|
| 158 |
+
**Client**
|
| 159 |
+
|
| 160 |
+
We recommend using the sampling parameters `temperature=0.6`, `top_p=0.95`, and `top_k=20`, and enabling `enable_thinking` for better performance.
|
| 161 |
+
|
| 162 |
+
```bash
|
| 163 |
+
curl -s http://${MASTER_IP}:${PORT}/v1/chat/completions \
|
| 164 |
+
-H "Content-Type: application/json" \
|
| 165 |
+
-d '{"model": "auto",
|
| 166 |
+
"messages": [{"role": "user", "content": "hello!"}],
|
| 167 |
+
"chat_template_kwargs": {"enable_thinking": true},
|
| 168 |
+
"stream": true,
|
| 169 |
+
"temperature": 0.6,
|
| 170 |
+
"top_k": 20,
|
| 171 |
+
"top_p": 0.95
|
| 172 |
+
}'
|
| 173 |
+
```
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- if tools %}
|
| 2 |
+
{{- '<role>SYSTEM</role>' }}
|
| 3 |
+
{%- if messages[0].role == 'system' %}
|
| 4 |
+
{{- messages[0].content + '\n' }}
|
| 5 |
+
{%- endif %}
|
| 6 |
+
{{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
|
| 7 |
+
{%- for tool in tools %}
|
| 8 |
+
{{- "\n" }}
|
| 9 |
+
{{- tool | tojson }}
|
| 10 |
+
{%- endfor %}
|
| 11 |
+
{{- "\n</tools>\n\nIf none of the functions can be used, point it out. If the given question lacks the parameters required by the function, also point it out.\nIf you need to use a function, for each function call, output the function name and arguments within the following XML format:\n<tool_call>{function-name}\n<arg_key>{arg-key-1}</arg_key>\n<arg_value>{arg-value-1}</arg_value>\n<arg_key>{arg-key-2}</arg_key>\n<arg_value>{arg-value-2}</arg_value>\n...\n</tool_call>\n" }}
|
| 12 |
+
{{- '<|role_end|>' }}
|
| 13 |
+
{%- else %}
|
| 14 |
+
{%- if messages[0].role == 'system' %}
|
| 15 |
+
{{- '<role>SYSTEM</role>' + messages[0].content + '<|role_end|>' }}
|
| 16 |
+
{%- endif %}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 19 |
+
{%- for message in messages[::-1] %}
|
| 20 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 21 |
+
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 22 |
+
{%- set ns.multi_step_tool = false %}
|
| 23 |
+
{%- set ns.last_query_index = index %}
|
| 24 |
+
{%- endif %}
|
| 25 |
+
{%- endfor %}
|
| 26 |
+
{%- for message in messages %}
|
| 27 |
+
{%- if message.content is string %}
|
| 28 |
+
{%- set content = message.content %}
|
| 29 |
+
{%- else %}
|
| 30 |
+
{%- set content = '' %}
|
| 31 |
+
{%- endif %}
|
| 32 |
+
{%- if message.role == "user" %}
|
| 33 |
+
{{- '<role>HUMAN</role>' + message.content + '<|role_end|>' }}
|
| 34 |
+
{%- elif message.role == "system" and not loop.first %}
|
| 35 |
+
{{- '<role>SYSTEM</role>' + message.content + '<|role_end|>' }}
|
| 36 |
+
{%- elif message.role == "assistant" %}
|
| 37 |
+
{%- set reasoning_content = '' %}
|
| 38 |
+
{%- if message.reasoning_content is string %}
|
| 39 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 40 |
+
{%- else %}
|
| 41 |
+
{%- if '</think>' in content %}
|
| 42 |
+
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 43 |
+
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- endif %}
|
| 46 |
+
{%- if loop.index0 > ns.last_query_index %}
|
| 47 |
+
{%- if reasoning_content %}
|
| 48 |
+
{{- '<role>ASSISTANT</role>\n' + reasoning_content.strip('\n') + '\n' + content.lstrip('\n') }}
|
| 49 |
+
{%- else %}
|
| 50 |
+
{{- '<role>ASSISTANT</role>' + content }}
|
| 51 |
+
{%- endif %}
|
| 52 |
+
{%- else %}
|
| 53 |
+
{{- '<role>ASSISTANT</role>' + content }}
|
| 54 |
+
{%- endif %}
|
| 55 |
+
{%- if message.tool_calls %}
|
| 56 |
+
{%- for tool_call in message.tool_calls %}
|
| 57 |
+
{%- if (loop.first and content) or (not loop.first) %}
|
| 58 |
+
{{- '\n' }}
|
| 59 |
+
{%- endif %}
|
| 60 |
+
{%- if tool_call.function %}
|
| 61 |
+
{%- set tc = tool_call.function %}
|
| 62 |
+
{%- endif %}
|
| 63 |
+
{{- '<tool_call>' + tc.name }}
|
| 64 |
+
{% set _args = tc.arguments %}
|
| 65 |
+
{%- for k, v in _args.items() %}
|
| 66 |
+
{{- '<arg_key>' + k + '</arg_key>' }}
|
| 67 |
+
{{- '\n<arg_value>' }}
|
| 68 |
+
{%- if v is string %}
|
| 69 |
+
{{- v }}
|
| 70 |
+
{%- else %}
|
| 71 |
+
{{- v | tojson(ensure_ascii=False) }}
|
| 72 |
+
{%- endif %}
|
| 73 |
+
{{- '</arg_value>' }}
|
| 74 |
+
{%- endfor %}
|
| 75 |
+
{{- '\n</tool_call>' }}
|
| 76 |
+
{%- endfor %}
|
| 77 |
+
{%- endif %}
|
| 78 |
+
{{- '<|role_end|>' }}
|
| 79 |
+
{%- elif message.role == "tool" %}
|
| 80 |
+
{%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
|
| 81 |
+
{{- '<role>OBSERVATION</role>' }}
|
| 82 |
+
{%- endif %}
|
| 83 |
+
{{- '\n<tool_response>\n' }}
|
| 84 |
+
{{- content }}
|
| 85 |
+
{{- '\n</tool_response>' }}
|
| 86 |
+
{%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
|
| 87 |
+
{{- '<|role_end|>' }}
|
| 88 |
+
{%- endif %}
|
| 89 |
+
{%- endif %}
|
| 90 |
+
{%- endfor %}
|
| 91 |
+
{%- if add_generation_prompt %}
|
| 92 |
+
{{- '<role>ASSISTANT</role>' }}
|
| 93 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BailingMoeV3ForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_dropout": 0.0,
|
| 6 |
+
"auto_map": {
|
| 7 |
+
"AutoConfig": "configuration_bailing_moe_v3.BailingMoeV3Config",
|
| 8 |
+
"AutoModel": "modeling_bailing_moe_v3.BailingMoeV3Model",
|
| 9 |
+
"AutoModelForCausalLM": "modeling_bailing_moe_v3.BailingMoeV3ForCausalLM"
|
| 10 |
+
},
|
| 11 |
+
"embedding_dropout": 0.0,
|
| 12 |
+
"eos_token_id": 156892,
|
| 13 |
+
"expert_swiglu_limit_list": [
|
| 14 |
+
0,
|
| 15 |
+
0,
|
| 16 |
+
0,
|
| 17 |
+
0,
|
| 18 |
+
0,
|
| 19 |
+
0,
|
| 20 |
+
0,
|
| 21 |
+
0,
|
| 22 |
+
0,
|
| 23 |
+
0,
|
| 24 |
+
0,
|
| 25 |
+
0,
|
| 26 |
+
0,
|
| 27 |
+
0,
|
| 28 |
+
0,
|
| 29 |
+
0,
|
| 30 |
+
0,
|
| 31 |
+
0,
|
| 32 |
+
0,
|
| 33 |
+
0,
|
| 34 |
+
0,
|
| 35 |
+
0,
|
| 36 |
+
0,
|
| 37 |
+
0,
|
| 38 |
+
0,
|
| 39 |
+
0,
|
| 40 |
+
0,
|
| 41 |
+
0,
|
| 42 |
+
0,
|
| 43 |
+
0,
|
| 44 |
+
0,
|
| 45 |
+
0,
|
| 46 |
+
0,
|
| 47 |
+
0,
|
| 48 |
+
0,
|
| 49 |
+
4,
|
| 50 |
+
4,
|
| 51 |
+
4,
|
| 52 |
+
4,
|
| 53 |
+
4,
|
| 54 |
+
4,
|
| 55 |
+
4
|
| 56 |
+
],
|
| 57 |
+
"first_k_dense_replace": 2,
|
| 58 |
+
"gated_attention_proj_granularity_type": "head_wise",
|
| 59 |
+
"group_norm_size": 1,
|
| 60 |
+
"head_dim": 128,
|
| 61 |
+
"hidden_act": "silu",
|
| 62 |
+
"hidden_size": 2560,
|
| 63 |
+
"initializer_range": 0.02,
|
| 64 |
+
"intermediate_size": 6144,
|
| 65 |
+
"kda_lower_bound": -5.0,
|
| 66 |
+
"kda_safe_gate": true,
|
| 67 |
+
"kv_lora_rank": 512,
|
| 68 |
+
"layer_group_size": 6,
|
| 69 |
+
"max_position_embeddings": 262144,
|
| 70 |
+
"moe_intermediate_size": 768,
|
| 71 |
+
"moe_router_enable_expert_bias": true,
|
| 72 |
+
"moe_shared_expert_intermediate_size": 768,
|
| 73 |
+
"mtp_loss_scaling_factor": 0,
|
| 74 |
+
"n_group": 8,
|
| 75 |
+
"no_kda_lora": true,
|
| 76 |
+
"norm_topk_prob": true,
|
| 77 |
+
"num_attention_heads": 32,
|
| 78 |
+
"num_experts": 512,
|
| 79 |
+
"num_experts_per_tok": 8,
|
| 80 |
+
"num_hidden_layers": 42,
|
| 81 |
+
"num_key_value_heads": 32,
|
| 82 |
+
"num_nextn_predict_layers": 1,
|
| 83 |
+
"num_shared_experts": 1,
|
| 84 |
+
"output_dropout": 0.0,
|
| 85 |
+
"output_router_logits": false,
|
| 86 |
+
"pad_token_id": 156892,
|
| 87 |
+
"partial_rotary_factor": 0.5,
|
| 88 |
+
"q_lora_rank": null,
|
| 89 |
+
"qk_head_dim": 192,
|
| 90 |
+
"qk_nope_head_dim": 128,
|
| 91 |
+
"qk_rope_head_dim": 64,
|
| 92 |
+
"rms_norm_eps": 1e-06,
|
| 93 |
+
"rope_interleave": true,
|
| 94 |
+
"rope_scaling": null,
|
| 95 |
+
"rope_theta": 6000000,
|
| 96 |
+
"rotary_dim": 64,
|
| 97 |
+
"routed_scaling_factor": 2.5,
|
| 98 |
+
"router_dtype": "fp32",
|
| 99 |
+
"scale_router_input": false,
|
| 100 |
+
"score_function": "sigmoid",
|
| 101 |
+
"scoring_func": "sigmoid",
|
| 102 |
+
"seq_aux": true,
|
| 103 |
+
"share_expert_swiglu_limit_list": [
|
| 104 |
+
0,
|
| 105 |
+
0,
|
| 106 |
+
0,
|
| 107 |
+
0,
|
| 108 |
+
0,
|
| 109 |
+
0,
|
| 110 |
+
0,
|
| 111 |
+
0,
|
| 112 |
+
0,
|
| 113 |
+
0,
|
| 114 |
+
0,
|
| 115 |
+
0,
|
| 116 |
+
0,
|
| 117 |
+
0,
|
| 118 |
+
0,
|
| 119 |
+
0,
|
| 120 |
+
0,
|
| 121 |
+
0,
|
| 122 |
+
0,
|
| 123 |
+
0,
|
| 124 |
+
0,
|
| 125 |
+
0,
|
| 126 |
+
0,
|
| 127 |
+
0,
|
| 128 |
+
0,
|
| 129 |
+
0,
|
| 130 |
+
0,
|
| 131 |
+
0,
|
| 132 |
+
0,
|
| 133 |
+
0,
|
| 134 |
+
0,
|
| 135 |
+
0,
|
| 136 |
+
0,
|
| 137 |
+
0,
|
| 138 |
+
5,
|
| 139 |
+
5,
|
| 140 |
+
5,
|
| 141 |
+
5,
|
| 142 |
+
5,
|
| 143 |
+
5,
|
| 144 |
+
7,
|
| 145 |
+
7
|
| 146 |
+
],
|
| 147 |
+
"short_conv_kernel_size": 4,
|
| 148 |
+
"tie_word_embeddings": false,
|
| 149 |
+
"topk_group": 4,
|
| 150 |
+
"topk_method": "noaux_tc",
|
| 151 |
+
"transformers_version": "4.56.2",
|
| 152 |
+
"up_proj_norm": false,
|
| 153 |
+
"use_bias": false,
|
| 154 |
+
"use_cache": true,
|
| 155 |
+
"use_mla_nope": false,
|
| 156 |
+
"use_qk_norm": true,
|
| 157 |
+
"use_qkv_bias": false,
|
| 158 |
+
"v_head_dim": 128,
|
| 159 |
+
"vocab_size": 157184,
|
| 160 |
+
"model_type": "bailing_hybrid",
|
| 161 |
+
"torch_dtype": "bfloat16"
|
| 162 |
+
}
|
configuration_bailing_moe_v3.py
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Bailing MoE V2 model configuration"""
|
| 2 |
+
|
| 3 |
+
from transformers.configuration_utils import PretrainedConfig
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class BailingMoeV3Config(PretrainedConfig):
|
| 7 |
+
|
| 8 |
+
def __init__(
|
| 9 |
+
self,
|
| 10 |
+
vocab_size=157184,
|
| 11 |
+
hidden_size=2048,
|
| 12 |
+
intermediate_size=5120,
|
| 13 |
+
num_hidden_layers=20,
|
| 14 |
+
num_attention_heads=16,
|
| 15 |
+
num_key_value_heads=4,
|
| 16 |
+
hidden_act="silu",
|
| 17 |
+
use_qkv_bias=False, # bailing only
|
| 18 |
+
use_bias=False, # bailing only
|
| 19 |
+
rms_norm_eps=1e-06,
|
| 20 |
+
tie_word_embeddings=False, # PretrainedConfig key, here change default value.
|
| 21 |
+
embedding_dropout=0.0,
|
| 22 |
+
attention_dropout=0.0,
|
| 23 |
+
output_dropout=0.0,
|
| 24 |
+
initializer_range=0.02,
|
| 25 |
+
max_position_embeddings=32768,
|
| 26 |
+
rope_theta=600000.0,
|
| 27 |
+
use_cache=True,
|
| 28 |
+
max_window_layers=20,
|
| 29 |
+
rope_scaling=None,
|
| 30 |
+
pad_token_id=156892,
|
| 31 |
+
eos_token_id=156892,
|
| 32 |
+
num_experts=256,
|
| 33 |
+
num_shared_experts=1,
|
| 34 |
+
num_experts_per_tok=8,
|
| 35 |
+
n_group=8,
|
| 36 |
+
topk_group=4,
|
| 37 |
+
moe_intermediate_size=512,
|
| 38 |
+
moe_shared_expert_intermediate_size=512,
|
| 39 |
+
first_k_dense_replace=1,
|
| 40 |
+
head_dim=128,
|
| 41 |
+
output_router_logits=False,
|
| 42 |
+
use_qk_norm=True,
|
| 43 |
+
num_nextn_predict_layers=0,
|
| 44 |
+
mtp_loss_scaling_factor=0,
|
| 45 |
+
moe_router_enable_expert_bias=True,
|
| 46 |
+
routed_scaling_factor=1.0,
|
| 47 |
+
layer_group_size=5,
|
| 48 |
+
kv_lora_rank=512,
|
| 49 |
+
q_lora_rank=None,
|
| 50 |
+
qk_rope_head_dim=64,
|
| 51 |
+
v_head_dim=128,
|
| 52 |
+
qk_nope_head_dim=128,
|
| 53 |
+
rope_interleave=True,
|
| 54 |
+
score_function="sigmoid",
|
| 55 |
+
scoring_func="sigmoid",
|
| 56 |
+
seq_aux=True,
|
| 57 |
+
topk_method="noaux_tc",
|
| 58 |
+
router_dtype="fp32",
|
| 59 |
+
gated_attention_proj_granularity_type=None,
|
| 60 |
+
no_kda_lora=False,
|
| 61 |
+
kda_safe_gate=False,
|
| 62 |
+
kda_lower_bound=None,
|
| 63 |
+
short_conv_kernel_size=4,
|
| 64 |
+
**kwargs,
|
| 65 |
+
):
|
| 66 |
+
self.num_hidden_layers = num_hidden_layers
|
| 67 |
+
self.vocab_size = vocab_size
|
| 68 |
+
self.hidden_size = hidden_size
|
| 69 |
+
self.intermediate_size = intermediate_size
|
| 70 |
+
self.num_attention_heads = num_attention_heads
|
| 71 |
+
self.num_key_value_heads = num_key_value_heads
|
| 72 |
+
self.hidden_act = hidden_act
|
| 73 |
+
self.use_qkv_bias = use_qkv_bias
|
| 74 |
+
self.use_bias = use_bias
|
| 75 |
+
self.rms_norm_eps = rms_norm_eps
|
| 76 |
+
self.embedding_dropout = embedding_dropout
|
| 77 |
+
self.attention_dropout = attention_dropout
|
| 78 |
+
self.output_dropout = output_dropout
|
| 79 |
+
self.num_nextn_predict_layers = num_nextn_predict_layers
|
| 80 |
+
self.mtp_loss_scaling_factor = mtp_loss_scaling_factor
|
| 81 |
+
self.initializer_range = initializer_range
|
| 82 |
+
self.max_position_embeddings = max_position_embeddings
|
| 83 |
+
self.rope_theta = rope_theta
|
| 84 |
+
self.use_cache = use_cache
|
| 85 |
+
self.max_window_layers = max_window_layers
|
| 86 |
+
self.head_dim = head_dim or self.hidden_size // self.num_attention_heads
|
| 87 |
+
self.rope_scaling = rope_scaling
|
| 88 |
+
self.use_qk_norm = use_qk_norm
|
| 89 |
+
self.moe_router_enable_expert_bias = moe_router_enable_expert_bias
|
| 90 |
+
self.routed_scaling_factor = routed_scaling_factor
|
| 91 |
+
|
| 92 |
+
# MoE configs
|
| 93 |
+
self.num_experts = num_experts
|
| 94 |
+
self.num_shared_experts = num_shared_experts
|
| 95 |
+
self.num_experts_per_tok = num_experts_per_tok
|
| 96 |
+
self.n_group = n_group
|
| 97 |
+
self.topk_group = topk_group
|
| 98 |
+
self.moe_intermediate_size = moe_intermediate_size
|
| 99 |
+
self.moe_shared_expert_intermediate_size = moe_shared_expert_intermediate_size
|
| 100 |
+
self.first_k_dense_replace = first_k_dense_replace
|
| 101 |
+
self.output_router_logits = output_router_logits
|
| 102 |
+
|
| 103 |
+
# Linear configs
|
| 104 |
+
self.layer_group_size = layer_group_size
|
| 105 |
+
# mla
|
| 106 |
+
self.kv_lora_rank = kv_lora_rank
|
| 107 |
+
self.q_lora_rank = q_lora_rank
|
| 108 |
+
self.qk_rope_head_dim = qk_rope_head_dim
|
| 109 |
+
|
| 110 |
+
self.score_function = score_function
|
| 111 |
+
self.scoring_func = scoring_func
|
| 112 |
+
self.seq_aux = seq_aux
|
| 113 |
+
self.topk_method = topk_method
|
| 114 |
+
self.v_head_dim = v_head_dim
|
| 115 |
+
self.qk_nope_head_dim = qk_nope_head_dim
|
| 116 |
+
self.qk_head_dim = qk_nope_head_dim + qk_rope_head_dim
|
| 117 |
+
self.rope_interleave = rope_interleave
|
| 118 |
+
self.router_dtype = router_dtype
|
| 119 |
+
self.gated_attention_proj_granularity_type = gated_attention_proj_granularity_type
|
| 120 |
+
self.no_kda_lora = no_kda_lora
|
| 121 |
+
self.kda_safe_gate = kda_safe_gate
|
| 122 |
+
self.kda_lower_bound = kda_lower_bound
|
| 123 |
+
self.short_conv_kernel_size = short_conv_kernel_size
|
| 124 |
+
super().__init__(
|
| 125 |
+
pad_token_id=pad_token_id, eos_token_id=eos_token_id, tie_word_embeddings=tie_word_embeddings, **kwargs
|
| 126 |
+
)
|
model-00001-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b6ba5336ed89af5f92c58d0f6de15e84bd1c651592a4a0d8d20b270f8b64757b
|
| 3 |
+
size 10087618176
|
model-00002-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:17531123f6d1a77173300569cdd3b7a198c5e0152d10925ff4e9fd8b920d7189
|
| 3 |
+
size 8053317688
|
model-00003-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ff717576e64b07e364e0c49ed485b15c81195e3e0675e398a276c36d8d907f76
|
| 3 |
+
size 10066647528
|
model-00004-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:12c270c1e20d7ea2b544947fb8b1f5133a412ebbfefdaefd90a6aee9e2cea776
|
| 3 |
+
size 8053319736
|
model-00005-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f9fd093c7a0ae3de224adfd35e80d200a4b13f5d80157cedf1af39825b1932e5
|
| 3 |
+
size 10066649064
|
model-00006-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2e078687102cdd92cf4b9109d7aa36cdaf1b9ede45a57429fd68857f3d70536a
|
| 3 |
+
size 8053319736
|
model-00007-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b65f7f4718466e4c2df029ff22d16927a6af4107bd58cacfa2873fc29ec39556
|
| 3 |
+
size 8545835224
|
model-00008-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8ce5db9b40a2c8681eb9458e37c113f4fa8be67399a64eb458739ec7834dbdf6
|
| 3 |
+
size 10066646504
|
model-00009-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:62edb57cb51a4975e193fbdd697499c12b160920e1e03beaf928fe4d1468df9d
|
| 3 |
+
size 8053317688
|
model-00010-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25da24ccf4ce03e65d5494189e4e07df1dc5c20671b4f81a567ad8aa74463e5f
|
| 3 |
+
size 10066647528
|
model-00011-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cda465271175d8b188f216cdb96009328ff1ed49e61c90fd85c7d40eec1d2eb9
|
| 3 |
+
size 8074291376
|
model-00012-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f53ca2f713f17f5f79ad1b5fa0a26be9a520da67c5a0bf51dea34c26409c9e28
|
| 3 |
+
size 10066649064
|
model-00013-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6eb9454dc421d107230ddeee074656a452abfb6646af0b2784f8f586a80b091b
|
| 3 |
+
size 8053319736
|
model-00014-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d6a74fd6edb54a96a0441f48c1779629ef9d0fedefa79040877ad8a093694c37
|
| 3 |
+
size 10066649064
|
model-00015-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:17f7bc8221aeccb9bb337435ba57a98a84488de1484967bc5512925eb93654e5
|
| 3 |
+
size 9594480064
|
model-00016-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:98cdc7d0a28b94190de76fda2a0153232b1053959eebb2c1e216fd4a1db13e83
|
| 3 |
+
size 10087620720
|
model-00017-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b58e1797889b11cd2eef23cfba1318cbfb746493eefb17988dd5ddc85452fdbb
|
| 3 |
+
size 8053319736
|
model-00018-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9608b4b3828a45f34d543adf164cd6f11ef82f0ebf87162b1530e1ff3fe3f5a6
|
| 3 |
+
size 10066649064
|
model-00019-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c7d81dc089c50ee7099a5fd8b1f5bf60fabce0c4fd6a0a980937f4f996eebb17
|
| 3 |
+
size 8053319736
|
model-00020-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0d13c624bdddfab9511382c0edc3851bf210b1abc2b2e0fbdfbb2044e2c9318f
|
| 3 |
+
size 10066649064
|
model-00021-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:79f6a9cfb5031e131df6462b83481f103104b5ad66dc6b8e7ab7505d669ff345
|
| 3 |
+
size 8053319736
|
model-00022-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:427d2229a2f32fa75873cb0e4762c5e6fcd4faf8d808d46affb714220bdeda72
|
| 3 |
+
size 10066649064
|
model-00023-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5de5a93560fe8fe2b2f98ac26997f5c5616d8ad88e0f4117a8ff19b0ae31a033
|
| 3 |
+
size 9594485384
|
model-00024-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b02cee085b8ae560f4d023453d6812e4e2f8286d444053bc0bf16f423ccc6d41
|
| 3 |
+
size 10087620720
|
model-00025-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:810b8dd5bd502c65afb73727ee13a76c18fc0e6c12dc3326dd82e4d12d7289e9
|
| 3 |
+
size 8053319736
|
model-00026-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d1b1a0e8ad14efe0760aef24ac72663aaece2ad2f1097e3490bbed3b2ba7f01b
|
| 3 |
+
size 10066649064
|
model-00027-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c54a9c563ed986bfc5b35971de63c23d27292722c20fa43244da839f1d46126a
|
| 3 |
+
size 8053319736
|
model-00028-of-00028.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8b9075b3be89a986f3f3ac3ec76bf4cac3156f2c840fb2438437439a297b94af
|
| 3 |
+
size 7709459672
|
model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
modeling_bailing_moe_v3.py
ADDED
|
@@ -0,0 +1,1625 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# coding=utf-8
|
| 2 |
+
# Copyright 2025 Antgroup and The HuggingFace Inc. team. All rights reserved.
|
| 3 |
+
#
|
| 4 |
+
# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
|
| 5 |
+
# and OPT implementations in this library. It has been modified from its
|
| 6 |
+
# original forms to accommodate minor architectural differences compared
|
| 7 |
+
# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
|
| 8 |
+
#
|
| 9 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 10 |
+
# you may not use this file except in compliance with the License.
|
| 11 |
+
# You may obtain a copy of the License at
|
| 12 |
+
#
|
| 13 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 14 |
+
#
|
| 15 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 16 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 17 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 18 |
+
# See the License for the specific language governing permissions and
|
| 19 |
+
# limitations under the License.
|
| 20 |
+
"""PyTorch BailingMoE model."""
|
| 21 |
+
|
| 22 |
+
import math
|
| 23 |
+
import warnings
|
| 24 |
+
from typing import List, Optional, Tuple, Union, Callable
|
| 25 |
+
from copy import deepcopy
|
| 26 |
+
|
| 27 |
+
import torch
|
| 28 |
+
import torch.nn.functional as F
|
| 29 |
+
from torch import nn
|
| 30 |
+
|
| 31 |
+
from transformers.activations import ACT2FN
|
| 32 |
+
from transformers.cache_utils import Cache, DynamicCache
|
| 33 |
+
from transformers.modeling_attn_mask_utils import (
|
| 34 |
+
AttentionMaskConverter,
|
| 35 |
+
_prepare_4d_attention_mask,
|
| 36 |
+
_prepare_4d_causal_attention_mask,
|
| 37 |
+
_prepare_4d_causal_attention_mask_for_sdpa,
|
| 38 |
+
)
|
| 39 |
+
from transformers.modeling_outputs import MoeModelOutputWithPast
|
| 40 |
+
from transformers.modeling_rope_utils import ROPE_INIT_FUNCTIONS, dynamic_rope_update
|
| 41 |
+
from transformers.modeling_utils import PreTrainedModel
|
| 42 |
+
from transformers.pytorch_utils import ALL_LAYERNORM_LAYERS, is_torch_greater_or_equal_than_1_13
|
| 43 |
+
from transformers.utils import (
|
| 44 |
+
add_start_docstrings,
|
| 45 |
+
add_start_docstrings_to_model_forward,
|
| 46 |
+
logging,
|
| 47 |
+
replace_return_docstrings,
|
| 48 |
+
)
|
| 49 |
+
from transformers.utils.import_utils import is_torch_fx_available
|
| 50 |
+
from .configuration_bailing_moe_v3 import BailingMoeV3Config
|
| 51 |
+
from transformers.generation.utils import GenerationMixin
|
| 52 |
+
from dataclasses import dataclass
|
| 53 |
+
from transformers.utils import ModelOutput
|
| 54 |
+
from transformers import DynamicLayer
|
| 55 |
+
from transformers.processing_utils import Unpack
|
| 56 |
+
from transformers.utils import TransformersKwargs
|
| 57 |
+
from transformers.utils.deprecation import deprecate_kwarg
|
| 58 |
+
from transformers.modeling_flash_attention_utils import FlashAttentionKwargs
|
| 59 |
+
|
| 60 |
+
from fla.ops.simple_gla.fused_recurrent import fused_recurrent_simple_gla
|
| 61 |
+
from fla.ops.simple_gla.chunk import chunk_simple_gla
|
| 62 |
+
from einops import rearrange, repeat
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
try:
|
| 66 |
+
from fla.modules import FusedRMSNormGated, ShortConvolution
|
| 67 |
+
from fla.ops.kda import chunk_kda, fused_recurrent_kda
|
| 68 |
+
|
| 69 |
+
from fla.ops.utils.index import prepare_cu_seqlens_from_mask, prepare_lens_from_mask
|
| 70 |
+
from fla.utils import tensor_cache
|
| 71 |
+
except ImportError:
|
| 72 |
+
raise ImportError("Plese run `pip install -U fla-core`")
|
| 73 |
+
|
| 74 |
+
# This makes `_prepare_4d_causal_attention_mask` a leaf function in the FX graph.
|
| 75 |
+
# It means that the function will not be traced through and simply appear as a node in the graph.
|
| 76 |
+
if is_torch_fx_available():
|
| 77 |
+
if not is_torch_greater_or_equal_than_1_13:
|
| 78 |
+
import torch.fx
|
| 79 |
+
|
| 80 |
+
_prepare_4d_causal_attention_mask = torch.fx.wrap(_prepare_4d_causal_attention_mask)
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
logger = logging.get_logger(__name__)
|
| 84 |
+
|
| 85 |
+
_CONFIG_FOR_DOC = "BailingMoeV3Config"
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
def roll_tensor(tensor, shifts=-1, dims=-1, fill_value=0):
|
| 89 |
+
"""Roll the tensor input along the given dimension(s).
|
| 90 |
+
Inserted elements are set to be 0.0.
|
| 91 |
+
"""
|
| 92 |
+
rolled_tensor = torch.roll(tensor, shifts=shifts, dims=dims)
|
| 93 |
+
rolled_tensor.select(dims, shifts).fill_(fill_value)
|
| 94 |
+
return rolled_tensor, rolled_tensor.sum()
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
@dataclass
|
| 98 |
+
class MoEV3CausalLMOutputWithPast(ModelOutput):
|
| 99 |
+
"""
|
| 100 |
+
Base class for causal language model (or autoregressive) outputs as well as Mixture of Expert's router hidden
|
| 101 |
+
states terms, to train a MoE model.
|
| 102 |
+
Args:
|
| 103 |
+
loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
|
| 104 |
+
Language modeling loss (for next-token prediction).
|
| 105 |
+
logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
|
| 106 |
+
Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
|
| 107 |
+
past_key_values (`Cache`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
|
| 108 |
+
It is a [`~cache_utils.Cache`] instance. For more details, see our [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache).
|
| 109 |
+
Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
|
| 110 |
+
`past_key_values` input) to speed up sequential decoding.
|
| 111 |
+
hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
|
| 112 |
+
Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
|
| 113 |
+
one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
|
| 114 |
+
Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
|
| 115 |
+
attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
|
| 116 |
+
Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
|
| 117 |
+
sequence_length)`.
|
| 118 |
+
Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
|
| 119 |
+
heads.
|
| 120 |
+
z_loss (`torch.FloatTensor`, *optional*, returned when `labels` is provided):
|
| 121 |
+
z_loss for the sparse modules.
|
| 122 |
+
aux_loss (`torch.FloatTensor`, *optional*, returned when `labels` is provided):
|
| 123 |
+
aux_loss for the sparse modules.
|
| 124 |
+
router_logits (`tuple(torch.FloatTensor)`, *optional*, returned when `output_router_logits=True` is passed or when `config.add_router_probs=True`):
|
| 125 |
+
Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, sequence_length, num_experts)`.
|
| 126 |
+
Router logits of the encoder model, useful to compute the auxiliary loss and the z_loss for the sparse
|
| 127 |
+
modules.
|
| 128 |
+
"""
|
| 129 |
+
|
| 130 |
+
loss: Optional[torch.FloatTensor] = None
|
| 131 |
+
logits: Optional[torch.FloatTensor] = None
|
| 132 |
+
past_key_values: Optional[Cache] = None
|
| 133 |
+
hidden_states: Optional[tuple[torch.FloatTensor, ...]] = None
|
| 134 |
+
attentions: Optional[tuple[torch.FloatTensor, ...]] = None
|
| 135 |
+
z_loss: Optional[torch.FloatTensor] = None
|
| 136 |
+
aux_loss: Optional[torch.FloatTensor] = None
|
| 137 |
+
router_logits: Optional[tuple[torch.FloatTensor]] = None
|
| 138 |
+
mtp_loss: Optional[torch.FloatTensor] = None
|
| 139 |
+
mtp_logits: Optional[tuple[torch.FloatTensor, ...]] = None
|
| 140 |
+
|
| 141 |
+
|
| 142 |
+
class MoeV3ModelOutputWithPast(MoeModelOutputWithPast):
|
| 143 |
+
|
| 144 |
+
def __init__(self, mtp_hidden_states=None, **kwargs):
|
| 145 |
+
super().__init__(**kwargs)
|
| 146 |
+
self.mtp_hidden_states = mtp_hidden_states
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
def index_first_axis(x, indices):
|
| 150 |
+
other_shape = x.shape[1:]
|
| 151 |
+
second_dim = other_shape.numel()
|
| 152 |
+
return torch.gather(
|
| 153 |
+
rearrange(x, "b ... -> b (...)"),
|
| 154 |
+
0,
|
| 155 |
+
repeat(indices, "z -> z d", d=second_dim),
|
| 156 |
+
).reshape(-1, *other_shape)
|
| 157 |
+
|
| 158 |
+
|
| 159 |
+
def index_put_first_axis(x, indices, first_axis_dim):
|
| 160 |
+
y = torch.zeros(first_axis_dim, *x.shape[1:], device=x.device, dtype=x.dtype)
|
| 161 |
+
y[indices] = x
|
| 162 |
+
# y.scatter_(0, repeat(indices, 'z -> z d', d=x.shape[1]), x)
|
| 163 |
+
return y
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
def pad_input(
|
| 167 |
+
hidden_states: torch.Tensor,
|
| 168 |
+
indices: torch.LongTensor,
|
| 169 |
+
batch_size: int,
|
| 170 |
+
seq_len: int,
|
| 171 |
+
) -> torch.Tensor:
|
| 172 |
+
output = index_put_first_axis(hidden_states, indices, batch_size * seq_len)
|
| 173 |
+
return rearrange(output, "(b s) ... -> b s ...", b=batch_size)
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
@tensor_cache
|
| 177 |
+
def _get_unpad_data(attention_mask):
|
| 178 |
+
seqlens_in_batch = attention_mask.sum(dim=-1, dtype=torch.int32)
|
| 179 |
+
indices = torch.nonzero(attention_mask.flatten(), as_tuple=False).flatten()
|
| 180 |
+
max_seqlen_in_batch = seqlens_in_batch.max().item()
|
| 181 |
+
cu_seqlens = F.pad(torch.cumsum(seqlens_in_batch, dim=0, dtype=torch.torch.int32), (1, 0))
|
| 182 |
+
return (
|
| 183 |
+
indices,
|
| 184 |
+
cu_seqlens,
|
| 185 |
+
max_seqlen_in_batch,
|
| 186 |
+
)
|
| 187 |
+
|
| 188 |
+
|
| 189 |
+
def _expand_mask(mask: torch.Tensor, dtype: torch.dtype, tgt_len: Optional[int] = None):
|
| 190 |
+
warnings.warn(
|
| 191 |
+
"Calling `transformers.models.BailingMoeV3.modeling_BailingMoeV3._prepare_4d_attention_mask` is deprecated and will be removed in v4.37. Use `transformers.modeling_attn_mask_utils._prepare_4d_attention_mask"
|
| 192 |
+
)
|
| 193 |
+
return _prepare_4d_attention_mask(mask=mask, dtype=dtype, tgt_len=tgt_len)
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
def _make_causal_mask(
|
| 197 |
+
input_ids_shape: torch.Size, dtype: torch.dtype, device: torch.device, past_key_values_length: int = 0
|
| 198 |
+
):
|
| 199 |
+
warnings.warn(
|
| 200 |
+
"Calling `transformers.models.BailingMoeV3.modeling_BailingMoeV3._make_causal_mask` is deprecated and will be removed in v4.37. Use `transformers.models.BailingMoeV3.modeling_BailingMoeV3.AttentionMaskConverter._make_causal_mask"
|
| 201 |
+
)
|
| 202 |
+
return AttentionMaskConverter._make_causal_mask(
|
| 203 |
+
input_ids_shape=input_ids_shape, dtype=dtype, device=device, past_key_values_length=past_key_values_length
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
|
| 207 |
+
class BailingMoeV3RMSNorm(nn.Module):
|
| 208 |
+
def __init__(self, hidden_size, eps=1e-6):
|
| 209 |
+
"""
|
| 210 |
+
BailingMoeV3RMSNorm is equivalent to T5LayerNorm
|
| 211 |
+
"""
|
| 212 |
+
super().__init__()
|
| 213 |
+
self.weight = nn.Parameter(torch.ones(hidden_size))
|
| 214 |
+
self.variance_epsilon = eps
|
| 215 |
+
|
| 216 |
+
def forward(self, hidden_states):
|
| 217 |
+
input_dtype = hidden_states.dtype
|
| 218 |
+
hidden_states = hidden_states.to(torch.float32)
|
| 219 |
+
variance = hidden_states.pow(2).mean(-1, keepdim=True)
|
| 220 |
+
hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
|
| 221 |
+
return self.weight * hidden_states.to(input_dtype)
|
| 222 |
+
|
| 223 |
+
|
| 224 |
+
class BailingMoeV3GroupRMSNorm(nn.Module):
|
| 225 |
+
def __init__(self, hidden_size, group_norm_size, eps=1e-6):
|
| 226 |
+
"""
|
| 227 |
+
BailingMoeV3RMSNorm is equivalent to T5LayerNorm
|
| 228 |
+
"""
|
| 229 |
+
super().__init__()
|
| 230 |
+
self.weight = nn.Parameter(torch.ones(hidden_size))
|
| 231 |
+
self.group_norm_size = group_norm_size
|
| 232 |
+
assert hidden_size % group_norm_size == 0, "hidden_size must be divisible by group_norm_size"
|
| 233 |
+
self.variance_epsilon = eps
|
| 234 |
+
|
| 235 |
+
def forward(self, hidden_states):
|
| 236 |
+
input_dtype = hidden_states.dtype
|
| 237 |
+
input_shape = hidden_states.size()
|
| 238 |
+
group_input_shape = input_shape[:-1] + (self.group_norm_size, input_shape[-1] // self.group_norm_size)
|
| 239 |
+
hidden_states = hidden_states.view(group_input_shape)
|
| 240 |
+
hidden_states = hidden_states.to(torch.float32)
|
| 241 |
+
variance = hidden_states.pow(2).mean(-1, keepdim=True)
|
| 242 |
+
hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
|
| 243 |
+
return self.weight * hidden_states.to(input_dtype).view(input_shape)
|
| 244 |
+
|
| 245 |
+
|
| 246 |
+
ALL_LAYERNORM_LAYERS.append(BailingMoeV3RMSNorm)
|
| 247 |
+
|
| 248 |
+
|
| 249 |
+
class BailingMoeV3RotaryEmbedding(nn.Module):
|
| 250 |
+
def __init__(self, config: BailingMoeV3Config, device=None):
|
| 251 |
+
super().__init__()
|
| 252 |
+
# BC: "rope_type" was originally "type"
|
| 253 |
+
if hasattr(config, "rope_scaling") and config.rope_scaling is not None:
|
| 254 |
+
self.rope_type = config.rope_scaling.get("rope_type", config.rope_scaling.get("type"))
|
| 255 |
+
else:
|
| 256 |
+
self.rope_type = "default"
|
| 257 |
+
self.max_seq_len_cached = config.max_position_embeddings
|
| 258 |
+
self.original_max_seq_len = config.max_position_embeddings
|
| 259 |
+
|
| 260 |
+
self.config = deepcopy(config)
|
| 261 |
+
self.config.head_dim = config.qk_rope_head_dim
|
| 262 |
+
self.config.partial_rotary_factor = 1.0
|
| 263 |
+
self.rope_init_fn = ROPE_INIT_FUNCTIONS[self.rope_type]
|
| 264 |
+
|
| 265 |
+
inv_freq, self.attention_scaling = self.rope_init_fn(self.config, device)
|
| 266 |
+
self.register_buffer("inv_freq", inv_freq, persistent=False)
|
| 267 |
+
self.original_inv_freq = self.inv_freq
|
| 268 |
+
|
| 269 |
+
@torch.no_grad()
|
| 270 |
+
@dynamic_rope_update # power user: used with advanced RoPE types (e.g. dynamic rope)
|
| 271 |
+
def forward(self, x, position_ids):
|
| 272 |
+
inv_freq_expanded = self.inv_freq[None, :, None].float().expand(position_ids.shape[0], -1, 1).to(x.device)
|
| 273 |
+
position_ids_expanded = position_ids[:, None, :].float()
|
| 274 |
+
|
| 275 |
+
device_type = x.device.type if isinstance(x.device.type, str) and x.device.type != "mps" else "cpu"
|
| 276 |
+
with torch.autocast(device_type=device_type, enabled=False): # Force float32
|
| 277 |
+
freqs = (inv_freq_expanded.float() @ position_ids_expanded.float()).transpose(1, 2)
|
| 278 |
+
emb = torch.cat((freqs, freqs), dim=-1)
|
| 279 |
+
cos = emb.cos() * self.attention_scaling
|
| 280 |
+
sin = emb.sin() * self.attention_scaling
|
| 281 |
+
|
| 282 |
+
return cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
# Copied from transformers.models.llama.modeling_llama.rotate_half
|
| 286 |
+
def rotate_half(x):
|
| 287 |
+
"""Rotates half the hidden dims of the input."""
|
| 288 |
+
x1 = x[..., : x.shape[-1] // 2]
|
| 289 |
+
x2 = x[..., x.shape[-1] // 2 :]
|
| 290 |
+
return torch.cat((-x2, x1), dim=-1)
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
# Copied from transformers.models.llama.modeling_llama.apply_rotary_pos_emb
|
| 294 |
+
def apply_rotary_pos_emb(q, k, cos, sin, unsqueeze_dim=1):
|
| 295 |
+
"""Applies Rotary Position Embedding to the query and key tensors.
|
| 296 |
+
Args:
|
| 297 |
+
q (`torch.Tensor`): The query tensor.
|
| 298 |
+
k (`torch.Tensor`): The key tensor.
|
| 299 |
+
cos (`torch.Tensor`): The cosine part of the rotary embedding.
|
| 300 |
+
sin (`torch.Tensor`): The sine part of the rotary embedding.
|
| 301 |
+
unsqueeze_dim (`int`, *optional*, defaults to 1):
|
| 302 |
+
The 'unsqueeze_dim' argument specifies the dimension along which to unsqueeze cos[position_ids] and
|
| 303 |
+
sin[position_ids] so that they can be properly broadcasted to the dimensions of q and k. For example, note
|
| 304 |
+
that cos[position_ids] and sin[position_ids] have the shape [batch_size, seq_len, head_dim]. Then, if q and
|
| 305 |
+
k have the shape [batch_size, heads, seq_len, head_dim], then setting unsqueeze_dim=1 makes
|
| 306 |
+
cos[position_ids] and sin[position_ids] broadcastable to the shapes of q and k. Similarly, if q and k have
|
| 307 |
+
the shape [batch_size, seq_len, heads, head_dim], then set unsqueeze_dim=2.
|
| 308 |
+
Returns:
|
| 309 |
+
`tuple(torch.Tensor)` comprising the query and key tensors rotated using the Rotary Position Embedding.
|
| 310 |
+
"""
|
| 311 |
+
cos = cos.unsqueeze(unsqueeze_dim)
|
| 312 |
+
sin = sin.unsqueeze(unsqueeze_dim)
|
| 313 |
+
|
| 314 |
+
# Keep half or full tensor for later concatenation
|
| 315 |
+
rotary_dim = cos.shape[-1]
|
| 316 |
+
q_rot, q_pass = q[..., :rotary_dim], q[..., rotary_dim:]
|
| 317 |
+
k_rot, k_pass = k[..., :rotary_dim], k[..., rotary_dim:]
|
| 318 |
+
|
| 319 |
+
# Apply rotary embeddings on the first half or full tensor
|
| 320 |
+
q_embed = (q_rot * cos) + (rotate_half(q_rot) * sin)
|
| 321 |
+
k_embed = (k_rot * cos) + (rotate_half(k_rot) * sin)
|
| 322 |
+
|
| 323 |
+
# Concatenate back to full shape
|
| 324 |
+
q_embed = torch.cat([q_embed, q_pass], dim=-1)
|
| 325 |
+
k_embed = torch.cat([k_embed, k_pass], dim=-1)
|
| 326 |
+
return q_embed, k_embed
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
class BailingMoeV3MLP(nn.Module):
|
| 330 |
+
def __init__(self, config: BailingMoeV3Config, intermediate_size: int):
|
| 331 |
+
super().__init__()
|
| 332 |
+
self.config = config
|
| 333 |
+
self.hidden_size = config.hidden_size
|
| 334 |
+
self.intermediate_size = intermediate_size
|
| 335 |
+
|
| 336 |
+
self.gate_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
|
| 337 |
+
self.up_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
|
| 338 |
+
self.down_proj = nn.Linear(self.intermediate_size, self.hidden_size, bias=False)
|
| 339 |
+
self.act_fn = ACT2FN[config.hidden_act]
|
| 340 |
+
|
| 341 |
+
def forward(self, x):
|
| 342 |
+
return self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 343 |
+
|
| 344 |
+
|
| 345 |
+
class BailingMoeV3Gate(nn.Module):
|
| 346 |
+
def __init__(self, config):
|
| 347 |
+
super().__init__()
|
| 348 |
+
self.config = config
|
| 349 |
+
self.top_k = config.num_experts_per_tok
|
| 350 |
+
self.num_experts = config.num_experts
|
| 351 |
+
|
| 352 |
+
self.n_group = config.n_group
|
| 353 |
+
self.topk_group = config.topk_group
|
| 354 |
+
|
| 355 |
+
# topk selection algorithm
|
| 356 |
+
self.gating_dim = config.hidden_size
|
| 357 |
+
self.weight = nn.Parameter(torch.empty((self.num_experts, self.gating_dim)))
|
| 358 |
+
self.routed_scaling_factor = config.routed_scaling_factor
|
| 359 |
+
|
| 360 |
+
self.register_buffer("expert_bias", torch.zeros((self.num_experts)))
|
| 361 |
+
self.reset_parameters()
|
| 362 |
+
|
| 363 |
+
def reset_parameters(self) -> None:
|
| 364 |
+
import torch.nn.init as init
|
| 365 |
+
|
| 366 |
+
init.kaiming_uniform_(self.weight, a=math.sqrt(5))
|
| 367 |
+
|
| 368 |
+
def group_limited_topk(
|
| 369 |
+
self,
|
| 370 |
+
scores: torch.Tensor,
|
| 371 |
+
):
|
| 372 |
+
num_tokens, _ = scores.size()
|
| 373 |
+
# Organize the experts into groups
|
| 374 |
+
group_scores = scores.view(num_tokens, self.n_group, -1).topk(2, dim=-1)[0].sum(dim=-1)
|
| 375 |
+
group_idx = torch.topk(group_scores, k=self.topk_group, dim=-1, sorted=False)[1]
|
| 376 |
+
group_mask = torch.zeros_like(group_scores)
|
| 377 |
+
group_mask.scatter_(1, group_idx, 1)
|
| 378 |
+
|
| 379 |
+
# Mask the experts based on selection groups
|
| 380 |
+
score_mask = (
|
| 381 |
+
group_mask.unsqueeze(-1)
|
| 382 |
+
.expand(num_tokens, self.n_group, self.num_experts // self.n_group)
|
| 383 |
+
.reshape(num_tokens, -1)
|
| 384 |
+
)
|
| 385 |
+
|
| 386 |
+
masked_scores = scores.masked_fill(~score_mask.bool(), float('-inf'))
|
| 387 |
+
probs, top_indices = torch.topk(masked_scores, k=self.top_k, dim=-1)
|
| 388 |
+
|
| 389 |
+
return probs, top_indices
|
| 390 |
+
|
| 391 |
+
def forward(self, hidden_states):
|
| 392 |
+
# compute gating score
|
| 393 |
+
hidden_states = hidden_states.view(-1, hidden_states.shape[-1])
|
| 394 |
+
logits = F.linear(hidden_states.type(torch.float32), self.weight.type(torch.float32))
|
| 395 |
+
|
| 396 |
+
scores = torch.sigmoid(logits.float()).type_as(logits)
|
| 397 |
+
|
| 398 |
+
scores_for_routing = scores + self.expert_bias
|
| 399 |
+
_, topk_idx = self.group_limited_topk(scores_for_routing)
|
| 400 |
+
|
| 401 |
+
scores = torch.gather(scores, dim=1, index=topk_idx).type_as(logits)
|
| 402 |
+
|
| 403 |
+
topk_weight = scores / (scores.sum(dim=-1, keepdim=True) + 1e-20) if self.top_k > 1 else scores
|
| 404 |
+
topk_weight = topk_weight * self.routed_scaling_factor
|
| 405 |
+
|
| 406 |
+
return topk_idx, topk_weight, logits
|
| 407 |
+
|
| 408 |
+
|
| 409 |
+
class BailingMoeV3SparseMoeBlock(nn.Module):
|
| 410 |
+
"""
|
| 411 |
+
A mixed expert module containing shared experts.
|
| 412 |
+
"""
|
| 413 |
+
|
| 414 |
+
def __init__(self, config: BailingMoeV3Config):
|
| 415 |
+
super().__init__()
|
| 416 |
+
self.config = config
|
| 417 |
+
self.num_experts_per_tok = config.num_experts_per_tok
|
| 418 |
+
self._setup_experts()
|
| 419 |
+
self.gate = BailingMoeV3Gate(config)
|
| 420 |
+
if config.num_shared_experts is not None:
|
| 421 |
+
self.shared_experts = BailingMoeV3MLP(
|
| 422 |
+
config=config, intermediate_size=config.moe_shared_expert_intermediate_size * config.num_shared_experts
|
| 423 |
+
)
|
| 424 |
+
|
| 425 |
+
def _setup_experts(self):
|
| 426 |
+
self.experts = nn.ModuleList(
|
| 427 |
+
[
|
| 428 |
+
BailingMoeV3MLP(config=self.config, intermediate_size=self.config.moe_intermediate_size)
|
| 429 |
+
for _ in range(self.config.num_experts)
|
| 430 |
+
]
|
| 431 |
+
)
|
| 432 |
+
|
| 433 |
+
def forward(self, hidden_states):
|
| 434 |
+
identity = hidden_states
|
| 435 |
+
bsz, seq_len, h = hidden_states.shape
|
| 436 |
+
topk_idx, topk_weight, router_logits = self.gate(hidden_states)
|
| 437 |
+
hidden_states = hidden_states.view(-1, hidden_states.shape[-1])
|
| 438 |
+
flat_topk_idx = topk_idx.view(-1)
|
| 439 |
+
if self.training:
|
| 440 |
+
hidden_states = hidden_states.repeat_interleave(self.num_experts_per_tok, dim=0)
|
| 441 |
+
y = torch.empty_like(hidden_states)
|
| 442 |
+
for i, expert in enumerate(self.experts):
|
| 443 |
+
y[flat_topk_idx == i] = expert(hidden_states[flat_topk_idx == i])
|
| 444 |
+
y = (y.view(*topk_weight.shape, -1) * topk_weight.unsqueeze(-1)).sum(dim=1)
|
| 445 |
+
y = y.to(hidden_states.dtype).view(bsz, seq_len, h)
|
| 446 |
+
else:
|
| 447 |
+
y = self.moe_infer(hidden_states, topk_idx, topk_weight).view(bsz, seq_len, h)
|
| 448 |
+
if self.config.num_shared_experts is not None:
|
| 449 |
+
y = y + self.shared_experts(identity)
|
| 450 |
+
return y, (router_logits.view(bsz, seq_len, -1), topk_idx.view(bsz, seq_len, -1))
|
| 451 |
+
|
| 452 |
+
@torch.no_grad()
|
| 453 |
+
def moe_infer(self, x, topk_ids, topk_weight):
|
| 454 |
+
cnts = topk_ids.new_zeros((topk_ids.shape[0], len(self.experts)))
|
| 455 |
+
cnts.scatter_(1, topk_ids, 1)
|
| 456 |
+
tokens_per_expert = cnts.sum(dim=0)
|
| 457 |
+
idxs = topk_ids.view(-1).argsort()
|
| 458 |
+
sorted_tokens = x[idxs // topk_ids.shape[1]]
|
| 459 |
+
tokens_per_expert = tokens_per_expert.cpu().numpy()
|
| 460 |
+
outputs = []
|
| 461 |
+
start_idx = 0
|
| 462 |
+
for i, num_tokens in enumerate(tokens_per_expert):
|
| 463 |
+
end_idx = start_idx + num_tokens
|
| 464 |
+
if num_tokens == 0:
|
| 465 |
+
continue
|
| 466 |
+
expert = self.experts[i]
|
| 467 |
+
tokens_for_this_expert = sorted_tokens[start_idx:end_idx]
|
| 468 |
+
expert_out = expert(tokens_for_this_expert)
|
| 469 |
+
outputs.append(expert_out.to(x.device))
|
| 470 |
+
start_idx = end_idx
|
| 471 |
+
|
| 472 |
+
outs = torch.cat(outputs, dim=0) if len(outputs) else sorted_tokens.new_empty(0)
|
| 473 |
+
new_x = torch.empty_like(outs)
|
| 474 |
+
new_x[idxs] = outs
|
| 475 |
+
final_out = (
|
| 476 |
+
new_x.view(*topk_ids.shape, -1)
|
| 477 |
+
.type(topk_weight.dtype)
|
| 478 |
+
.mul_(topk_weight.unsqueeze(dim=-1))
|
| 479 |
+
.sum(dim=1)
|
| 480 |
+
.type(new_x.dtype)
|
| 481 |
+
)
|
| 482 |
+
return final_out
|
| 483 |
+
|
| 484 |
+
|
| 485 |
+
# Copied from transformers.models.llama.modeling_llama.repeat_kv
|
| 486 |
+
def repeat_kv(hidden_states: torch.Tensor, n_rep: int, head_first: bool = True) -> torch.Tensor:
|
| 487 |
+
"""
|
| 488 |
+
This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). If head_first is True, the hidden states go from (batch,
|
| 489 |
+
num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
|
| 490 |
+
"""
|
| 491 |
+
if n_rep == 1:
|
| 492 |
+
return hidden_states
|
| 493 |
+
if head_first:
|
| 494 |
+
batch, num_key_value_heads, slen, head_dim = hidden_states.shape
|
| 495 |
+
hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
|
| 496 |
+
return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
|
| 497 |
+
else:
|
| 498 |
+
batch, slen, num_key_value_heads, head_dim = hidden_states.shape
|
| 499 |
+
hidden_states = hidden_states[:, :, :, None, :].expand(batch, slen, num_key_value_heads, n_rep, head_dim)
|
| 500 |
+
return hidden_states.reshape(batch, slen, num_key_value_heads * n_rep, head_dim)
|
| 501 |
+
|
| 502 |
+
|
| 503 |
+
def repeat_kv2(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
|
| 504 |
+
"""
|
| 505 |
+
This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
|
| 506 |
+
num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
|
| 507 |
+
"""
|
| 508 |
+
batch, num_key_value_heads, slen, head_dim = hidden_states.shape
|
| 509 |
+
if n_rep == 1:
|
| 510 |
+
return hidden_states
|
| 511 |
+
hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
|
| 512 |
+
return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
|
| 513 |
+
|
| 514 |
+
|
| 515 |
+
def eager_attention_forward(
|
| 516 |
+
module: nn.Module,
|
| 517 |
+
query: torch.Tensor,
|
| 518 |
+
key: torch.Tensor,
|
| 519 |
+
value: torch.Tensor,
|
| 520 |
+
attention_mask: Optional[torch.Tensor],
|
| 521 |
+
scaling: float,
|
| 522 |
+
dropout: float = 0.0,
|
| 523 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 524 |
+
):
|
| 525 |
+
key_states = repeat_kv2(key, module.num_key_value_groups)
|
| 526 |
+
value_states = repeat_kv2(value, module.num_key_value_groups)
|
| 527 |
+
|
| 528 |
+
attn_weights = torch.matmul(query, key_states.transpose(2, 3)) * scaling
|
| 529 |
+
if attention_mask is not None:
|
| 530 |
+
causal_mask = attention_mask[:, :, :, : key_states.shape[-2]]
|
| 531 |
+
attn_weights = attn_weights + causal_mask
|
| 532 |
+
|
| 533 |
+
attn_weights = nn.functional.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query.dtype)
|
| 534 |
+
attn_weights = nn.functional.dropout(attn_weights, p=dropout, training=module.training)
|
| 535 |
+
attn_output = torch.matmul(attn_weights, value_states)
|
| 536 |
+
attn_output = attn_output.transpose(1, 2).contiguous()
|
| 537 |
+
|
| 538 |
+
return attn_output, attn_weights
|
| 539 |
+
|
| 540 |
+
|
| 541 |
+
def apply_rotary_pos_emb_interleave(q, k, cos, sin, position_ids=None, unsqueeze_dim=1):
|
| 542 |
+
r"""
|
| 543 |
+
TODO let's just use the original freqcis computation to not have the view
|
| 544 |
+
transpose + reshape! This is not optimized!
|
| 545 |
+
Applies Rotary Position Embedding to the query and key tensors.
|
| 546 |
+
|
| 547 |
+
Args:
|
| 548 |
+
q (`torch.Tensor`): The query tensor.
|
| 549 |
+
k (`torch.Tensor`): The key tensor.
|
| 550 |
+
cos (`torch.Tensor`): The cosine part of the rotary embedding.
|
| 551 |
+
sin (`torch.Tensor`): The sine part of the rotary embedding.
|
| 552 |
+
position_ids (`torch.Tensor`):
|
| 553 |
+
The position indices of the tokens corresponding to the query and key tensors. For example, this can be
|
| 554 |
+
used to pass offsetted position ids when working with a KV-cache.
|
| 555 |
+
unsqueeze_dim (`int`, *optional*, defaults to 1):
|
| 556 |
+
The 'unsqueeze_dim' argument specifies the dimension along which to unsqueeze cos[position_ids] and
|
| 557 |
+
sin[position_ids] so that they can be properly broadcasted to the dimensions of q and k. For example, note
|
| 558 |
+
that cos[position_ids] and sin[position_ids] have the shape [batch_size, seq_len, head_dim]. Then, if q and
|
| 559 |
+
k have the shape [batch_size, heads, seq_len, head_dim], then setting unsqueeze_dim=1 makes
|
| 560 |
+
cos[position_ids] and sin[position_ids] broadcastable to the shapes of q and k. Similarly, if q and k have
|
| 561 |
+
the shape [batch_size, seq_len, heads, head_dim], then set unsqueeze_dim=2.
|
| 562 |
+
Returns:
|
| 563 |
+
`tuple(torch.Tensor)` comprising of the query and key tensors rotated using the Rotary Position Embedding.
|
| 564 |
+
"""
|
| 565 |
+
cos = cos.unsqueeze(unsqueeze_dim)
|
| 566 |
+
sin = sin.unsqueeze(unsqueeze_dim)
|
| 567 |
+
|
| 568 |
+
b, h, s, d = q.shape
|
| 569 |
+
q = q.view(b, h, s, d // 2, 2).transpose(4, 3).reshape(b, h, s, d)
|
| 570 |
+
|
| 571 |
+
b, h, s, d = k.shape
|
| 572 |
+
k = k.view(b, h, s, d // 2, 2).transpose(4, 3).reshape(b, h, s, d)
|
| 573 |
+
|
| 574 |
+
q_embed = (q * cos) + (rotate_half(q) * sin)
|
| 575 |
+
k_embed = (k * cos) + (rotate_half(k) * sin)
|
| 576 |
+
return q_embed, k_embed
|
| 577 |
+
|
| 578 |
+
|
| 579 |
+
def yarn_get_mscale(scale=1, mscale=1):
|
| 580 |
+
if scale <= 1:
|
| 581 |
+
return 1.0
|
| 582 |
+
return 0.1 * mscale * math.log(scale) + 1.0
|
| 583 |
+
|
| 584 |
+
|
| 585 |
+
class BailingMoeV3MultiLatentAttention(nn.Module):
|
| 586 |
+
"""Multi-headed attention from 'Attention Is All You Need' paper"""
|
| 587 |
+
|
| 588 |
+
def __init__(self, config: BailingMoeV3Config, layer_idx: int):
|
| 589 |
+
super().__init__()
|
| 590 |
+
self.config = config
|
| 591 |
+
self.layer_idx = layer_idx
|
| 592 |
+
self.num_key_value_groups = config.num_attention_heads // config.num_key_value_heads
|
| 593 |
+
self.attention_dropout = config.attention_dropout
|
| 594 |
+
self.num_heads = config.num_attention_heads
|
| 595 |
+
self.rope_theta = config.rope_theta
|
| 596 |
+
self.q_lora_rank = config.q_lora_rank
|
| 597 |
+
self.qk_rope_head_dim = config.qk_rope_head_dim
|
| 598 |
+
self.kv_lora_rank = config.kv_lora_rank
|
| 599 |
+
self.v_head_dim = config.v_head_dim
|
| 600 |
+
self.qk_nope_head_dim = config.qk_nope_head_dim
|
| 601 |
+
self.qk_head_dim = config.qk_head_dim
|
| 602 |
+
self.gated_attention_proj_granularity_type = config.gated_attention_proj_granularity_type
|
| 603 |
+
|
| 604 |
+
self.is_causal = True
|
| 605 |
+
if self.q_lora_rank is None:
|
| 606 |
+
self.q_proj = nn.Linear(config.hidden_size, self.num_heads * self.qk_head_dim, bias=False)
|
| 607 |
+
else:
|
| 608 |
+
self.q_a_proj = nn.Linear(config.hidden_size, config.q_lora_rank, bias=config.use_qkv_bias)
|
| 609 |
+
self.q_a_layernorm = BailingMoeV3RMSNorm(config.q_lora_rank)
|
| 610 |
+
self.q_b_proj = nn.Linear(config.q_lora_rank, self.num_heads * self.qk_head_dim, bias=False)
|
| 611 |
+
|
| 612 |
+
self.kv_a_proj_with_mqa = nn.Linear(
|
| 613 |
+
config.hidden_size,
|
| 614 |
+
self.kv_lora_rank + self.qk_rope_head_dim,
|
| 615 |
+
bias=config.use_qkv_bias,
|
| 616 |
+
)
|
| 617 |
+
self.kv_a_layernorm = BailingMoeV3RMSNorm(self.kv_lora_rank)
|
| 618 |
+
self.kv_b_proj = nn.Linear(
|
| 619 |
+
self.kv_lora_rank,
|
| 620 |
+
self.num_heads * (self.qk_nope_head_dim + self.v_head_dim),
|
| 621 |
+
bias=False,
|
| 622 |
+
)
|
| 623 |
+
if self.gated_attention_proj_granularity_type is None:
|
| 624 |
+
self.g_proj = None
|
| 625 |
+
elif self.gated_attention_proj_granularity_type == "head_wise":
|
| 626 |
+
self.g_proj = nn.Linear(config.hidden_size, self.num_heads, bias=False)
|
| 627 |
+
elif self.gated_attention_proj_granularity_type == "element_wise":
|
| 628 |
+
self.g_proj = nn.Linear(config.hidden_size, self.num_heads * self.v_head_dim, bias=False)
|
| 629 |
+
|
| 630 |
+
self.dense = nn.Linear(
|
| 631 |
+
self.num_heads * self.v_head_dim,
|
| 632 |
+
config.hidden_size,
|
| 633 |
+
bias=config.use_qkv_bias,
|
| 634 |
+
)
|
| 635 |
+
|
| 636 |
+
self.scaling = self.qk_head_dim ** (-0.5)
|
| 637 |
+
if self.config.rope_scaling is not None:
|
| 638 |
+
mscale_all_dim = self.config.rope_scaling.get("mscale_all_dim", 0)
|
| 639 |
+
scaling_factor = self.config.rope_scaling["factor"]
|
| 640 |
+
if mscale_all_dim:
|
| 641 |
+
mscale = yarn_get_mscale(scaling_factor, mscale_all_dim)
|
| 642 |
+
self.scaling = self.scaling * mscale * mscale
|
| 643 |
+
|
| 644 |
+
@deprecate_kwarg("past_key_value", new_name="past_key_values", version="4.58")
|
| 645 |
+
def forward(
|
| 646 |
+
self,
|
| 647 |
+
hidden_states: torch.Tensor,
|
| 648 |
+
position_embeddings: tuple[torch.Tensor, torch.Tensor],
|
| 649 |
+
attention_mask: Optional[torch.Tensor],
|
| 650 |
+
past_key_values: Optional[Cache] = None,
|
| 651 |
+
cache_position: Optional[torch.LongTensor] = None,
|
| 652 |
+
**kwargs: Unpack[FlashAttentionKwargs],
|
| 653 |
+
) -> tuple[torch.Tensor, Optional[torch.Tensor], Optional[tuple[torch.Tensor]]]:
|
| 654 |
+
|
| 655 |
+
batch_size, seq_length = hidden_states.shape[:-1]
|
| 656 |
+
query_shape = (batch_size, seq_length, -1, self.qk_head_dim)
|
| 657 |
+
key_shape = (batch_size, seq_length, -1, self.qk_nope_head_dim + self.v_head_dim)
|
| 658 |
+
|
| 659 |
+
if self.q_lora_rank is None:
|
| 660 |
+
q_states = self.q_proj(hidden_states)
|
| 661 |
+
else:
|
| 662 |
+
q_states = self.q_b_proj(self.q_a_layernorm(self.q_a_proj(hidden_states)))
|
| 663 |
+
q_states = q_states.view(query_shape).transpose(1, 2)
|
| 664 |
+
q_pass, q_rot = torch.split(q_states, [self.qk_nope_head_dim, self.qk_rope_head_dim], dim=-1)
|
| 665 |
+
|
| 666 |
+
compressed_kv = self.kv_a_proj_with_mqa(hidden_states)
|
| 667 |
+
k_pass, k_rot = torch.split(compressed_kv, [self.kv_lora_rank, self.qk_rope_head_dim], dim=-1)
|
| 668 |
+
|
| 669 |
+
k_pass = self.kv_b_proj(self.kv_a_layernorm(k_pass)).view(key_shape).transpose(1, 2)
|
| 670 |
+
k_pass, value_states = torch.split(k_pass, [self.qk_nope_head_dim, self.v_head_dim], dim=-1)
|
| 671 |
+
|
| 672 |
+
k_rot = k_rot.view(batch_size, 1, seq_length, self.qk_rope_head_dim)
|
| 673 |
+
|
| 674 |
+
cos, sin = position_embeddings # tptest
|
| 675 |
+
if self.config.rope_interleave: # support using interleaved weights for efficiency
|
| 676 |
+
q_rot, k_rot = apply_rotary_pos_emb_interleave(q_rot, k_rot, cos, sin)
|
| 677 |
+
else:
|
| 678 |
+
x = 1 / 0
|
| 679 |
+
q_rot, k_rot = apply_rotary_pos_emb(q_rot, k_rot, cos, sin)
|
| 680 |
+
k_rot = k_rot.expand(*k_pass.shape[:-1], -1)
|
| 681 |
+
|
| 682 |
+
query_states = torch.cat((q_pass, q_rot), dim=-1)
|
| 683 |
+
key_states = torch.cat((k_pass, k_rot), dim=-1)
|
| 684 |
+
|
| 685 |
+
if past_key_values is not None:
|
| 686 |
+
# sin and cos are specific to RoPE models; cache_position needed for the static cache
|
| 687 |
+
cache_kwargs = {"sin": sin, "cos": cos, "cache_position": cache_position}
|
| 688 |
+
key_states, value_states = past_key_values.update(key_states, value_states, self.layer_idx, cache_kwargs)
|
| 689 |
+
|
| 690 |
+
if self.config._attn_implementation == "flash_attention_2" and self.qk_head_dim != self.v_head_dim:
|
| 691 |
+
value_states = F.pad(value_states, [0, self.qk_head_dim - self.v_head_dim])
|
| 692 |
+
|
| 693 |
+
attention_interface: Callable = eager_attention_forward
|
| 694 |
+
|
| 695 |
+
attn_output, attn_weights = attention_interface(
|
| 696 |
+
self,
|
| 697 |
+
query_states,
|
| 698 |
+
key_states,
|
| 699 |
+
value_states,
|
| 700 |
+
attention_mask,
|
| 701 |
+
dropout=0.0 if not self.training else self.attention_dropout,
|
| 702 |
+
scaling=self.scaling,
|
| 703 |
+
**kwargs,
|
| 704 |
+
)
|
| 705 |
+
|
| 706 |
+
if self.config._attn_implementation == "flash_attention_2" and self.qk_head_dim != self.v_head_dim:
|
| 707 |
+
attn_output = attn_output[:, :, :, : self.v_head_dim]
|
| 708 |
+
|
| 709 |
+
if self.g_proj is not None:
|
| 710 |
+
gate = self.g_proj(hidden_states)
|
| 711 |
+
gate = F.sigmoid(gate.float()).type_as(hidden_states)
|
| 712 |
+
if self.gated_attention_proj_granularity_type == "head_wise":
|
| 713 |
+
attn_output = attn_output * gate[:, :, :, None]
|
| 714 |
+
else:
|
| 715 |
+
attn_output = attn_output * gate.view(batch_size, seq_length, self.num_heads, self.v_head_dim)
|
| 716 |
+
|
| 717 |
+
attn_output = attn_output.reshape(batch_size, seq_length, -1).contiguous()
|
| 718 |
+
attn_output = self.dense(attn_output)
|
| 719 |
+
return attn_output, attn_weights, past_key_values
|
| 720 |
+
|
| 721 |
+
|
| 722 |
+
class BailingMoeV3KimiDeltaAttention(nn.Module):
|
| 723 |
+
def __init__(self, config: BailingMoeV3Config, layer_idx: int):
|
| 724 |
+
super().__init__()
|
| 725 |
+
self.config = config
|
| 726 |
+
self.mode = "chunk"
|
| 727 |
+
|
| 728 |
+
self.hidden_size = config.hidden_size
|
| 729 |
+
self.conv_size = config.short_conv_kernel_size
|
| 730 |
+
self.head_dim = config.head_dim
|
| 731 |
+
self.num_heads = config.num_attention_heads
|
| 732 |
+
self.head_k_dim = self.head_dim
|
| 733 |
+
self.num_k_heads = self.num_heads
|
| 734 |
+
self.no_kda_lora = config.no_kda_lora
|
| 735 |
+
self.safe_gate = config.kda_safe_gate
|
| 736 |
+
self.lower_bound = config.kda_lower_bound
|
| 737 |
+
|
| 738 |
+
self.layer_idx = layer_idx
|
| 739 |
+
|
| 740 |
+
assert self.mode in ['chunk', 'fused_recurrent'], f"Not suppoerted mode `{self.mode}`."
|
| 741 |
+
|
| 742 |
+
projection_k_size = self.head_k_dim * self.num_k_heads
|
| 743 |
+
projection_size = self.head_dim * self.num_heads
|
| 744 |
+
|
| 745 |
+
self.q_proj = nn.Linear(self.hidden_size, projection_k_size, bias=False)
|
| 746 |
+
self.k_proj = nn.Linear(self.hidden_size, projection_k_size, bias=False)
|
| 747 |
+
self.v_proj = nn.Linear(self.hidden_size, projection_size, bias=False)
|
| 748 |
+
|
| 749 |
+
self.q_conv1d = ShortConvolution(
|
| 750 |
+
hidden_size=projection_k_size,
|
| 751 |
+
kernel_size=self.conv_size,
|
| 752 |
+
activation='silu',
|
| 753 |
+
)
|
| 754 |
+
self.k_conv1d = ShortConvolution(
|
| 755 |
+
hidden_size=projection_k_size,
|
| 756 |
+
kernel_size=self.conv_size,
|
| 757 |
+
activation='silu',
|
| 758 |
+
)
|
| 759 |
+
self.v_conv1d = ShortConvolution(
|
| 760 |
+
hidden_size=projection_size,
|
| 761 |
+
kernel_size=self.conv_size,
|
| 762 |
+
activation='silu',
|
| 763 |
+
)
|
| 764 |
+
|
| 765 |
+
self.A_log = torch.nn.Parameter(torch.log(torch.empty(self.num_heads, dtype=torch.float32).uniform_(1, 16)))
|
| 766 |
+
|
| 767 |
+
if self.no_kda_lora:
|
| 768 |
+
self.f_proj = nn.Linear(self.hidden_size, projection_size, bias=False)
|
| 769 |
+
else:
|
| 770 |
+
self.f_a_proj = nn.Linear(self.hidden_size, self.head_dim, bias=False)
|
| 771 |
+
self.f_b_proj = nn.Linear(self.head_dim, projection_size, bias=False)
|
| 772 |
+
|
| 773 |
+
self.dt_bias = nn.Parameter(torch.empty(projection_size, dtype=torch.float32))
|
| 774 |
+
|
| 775 |
+
self.b_proj = nn.Linear(self.hidden_size, self.num_heads, bias=False)
|
| 776 |
+
|
| 777 |
+
if self.no_kda_lora:
|
| 778 |
+
self.g_proj = nn.Linear(self.hidden_size, projection_size, bias=False)
|
| 779 |
+
else:
|
| 780 |
+
self.g_a_proj = nn.Linear(self.hidden_size, self.head_dim, bias=False)
|
| 781 |
+
self.g_b_proj = nn.Linear(self.head_dim, projection_size, bias=False)
|
| 782 |
+
|
| 783 |
+
self.o_norm = FusedRMSNormGated(self.head_dim, eps=config.rms_norm_eps, activation='sigmoid')
|
| 784 |
+
self.o_proj = nn.Linear(projection_size, self.hidden_size, bias=False)
|
| 785 |
+
|
| 786 |
+
def forward(
|
| 787 |
+
self,
|
| 788 |
+
hidden_states: torch.Tensor,
|
| 789 |
+
attention_mask: torch.Tensor | None = None,
|
| 790 |
+
past_key_value=None,
|
| 791 |
+
**kwargs: Unpack[dict],
|
| 792 |
+
) -> tuple[torch.Tensor, torch.Tensor | None, Cache | None]:
|
| 793 |
+
attention_mask = None
|
| 794 |
+
if attention_mask is not None:
|
| 795 |
+
if attention_mask.dim() != 2:
|
| 796 |
+
attention_mask = kwargs.get("padding_mask")
|
| 797 |
+
|
| 798 |
+
if attention_mask is not None and attention_mask.dim() != 2:
|
| 799 |
+
raise ValueError(
|
| 800 |
+
"attention_mask must be a 0-1 matrix of shape [batch_size, seq_len] "
|
| 801 |
+
"(0 = padding). 3D masks are not supported here.",
|
| 802 |
+
)
|
| 803 |
+
use_cache = past_key_value is not None
|
| 804 |
+
batch_size, q_len, _ = hidden_states.shape
|
| 805 |
+
mode = 'fused_recurrent' if q_len <= 64 else self.mode
|
| 806 |
+
if self.training:
|
| 807 |
+
assert mode == 'chunk', "Only chunk mode is supported in training."
|
| 808 |
+
|
| 809 |
+
cu_seqlens = kwargs.get('cu_seqlens')
|
| 810 |
+
indices = None
|
| 811 |
+
if attention_mask is not None:
|
| 812 |
+
indices, cu_seqlens, _ = _get_unpad_data(attention_mask[:, -q_len:])
|
| 813 |
+
hidden_states = index_first_axis(rearrange(hidden_states, "b s ... -> (b s) ..."), indices).unsqueeze(0)
|
| 814 |
+
|
| 815 |
+
conv_state_q, conv_state_k, conv_state_v = None, None, None
|
| 816 |
+
recurrent_state = None
|
| 817 |
+
|
| 818 |
+
if past_key_value is not None and isinstance(past_key_value, Cache):
|
| 819 |
+
# ensure the cache list is long enough
|
| 820 |
+
while len(past_key_value.layers) <= self.layer_idx:
|
| 821 |
+
past_key_value.layers.append(DynamicLayer())
|
| 822 |
+
|
| 823 |
+
if past_key_value.layers[self.layer_idx].keys is not None:
|
| 824 |
+
recurrent_state = past_key_value.layers[self.layer_idx].keys
|
| 825 |
+
# ensure recurrent_state is on the same device as hidden_states
|
| 826 |
+
if recurrent_state.device != hidden_states.device:
|
| 827 |
+
recurrent_state = recurrent_state.to(hidden_states.device).contiguous()
|
| 828 |
+
|
| 829 |
+
if past_key_value.layers[self.layer_idx].values is not None:
|
| 830 |
+
conv_state_q, conv_state_k, conv_state_v = past_key_value.layers[self.layer_idx].values
|
| 831 |
+
|
| 832 |
+
q, conv_state_q = self.q_conv1d(
|
| 833 |
+
x=self.q_proj(hidden_states),
|
| 834 |
+
cache=conv_state_q,
|
| 835 |
+
output_final_state=use_cache,
|
| 836 |
+
cu_seqlens=cu_seqlens,
|
| 837 |
+
)
|
| 838 |
+
k, conv_state_k = self.k_conv1d(
|
| 839 |
+
x=self.k_proj(hidden_states),
|
| 840 |
+
cache=conv_state_k,
|
| 841 |
+
output_final_state=use_cache,
|
| 842 |
+
cu_seqlens=cu_seqlens,
|
| 843 |
+
)
|
| 844 |
+
v, conv_state_v = self.v_conv1d(
|
| 845 |
+
x=self.v_proj(hidden_states),
|
| 846 |
+
cache=conv_state_v,
|
| 847 |
+
output_final_state=use_cache,
|
| 848 |
+
cu_seqlens=cu_seqlens,
|
| 849 |
+
)
|
| 850 |
+
|
| 851 |
+
if self.no_kda_lora:
|
| 852 |
+
g = self.f_proj(hidden_states)
|
| 853 |
+
else:
|
| 854 |
+
g = self.f_b_proj(self.f_a_proj(hidden_states))
|
| 855 |
+
|
| 856 |
+
beta = self.b_proj(hidden_states).float().sigmoid()
|
| 857 |
+
|
| 858 |
+
q, k = map(lambda x: rearrange(x, '... (h d) -> ... h d', d=self.head_k_dim), (q, k))
|
| 859 |
+
v = rearrange(v, '... (h d) -> ... h d', d=self.head_dim)
|
| 860 |
+
g = rearrange(g, '... (h d) -> ... h d', d=self.head_dim)
|
| 861 |
+
|
| 862 |
+
if mode == 'chunk':
|
| 863 |
+
o, recurrent_state = chunk_kda(
|
| 864 |
+
q=q,
|
| 865 |
+
k=k,
|
| 866 |
+
v=v,
|
| 867 |
+
g=g,
|
| 868 |
+
beta=beta,
|
| 869 |
+
A_log=self.A_log,
|
| 870 |
+
dt_bias=self.dt_bias,
|
| 871 |
+
initial_state=recurrent_state,
|
| 872 |
+
output_final_state=True,
|
| 873 |
+
use_qk_l2norm_in_kernel=True,
|
| 874 |
+
use_gate_in_kernel=True,
|
| 875 |
+
safe_gate=self.safe_gate,
|
| 876 |
+
lower_bound=self.lower_bound,
|
| 877 |
+
cu_seqlens=cu_seqlens,
|
| 878 |
+
)
|
| 879 |
+
else:
|
| 880 |
+
o, recurrent_state = fused_recurrent_kda(
|
| 881 |
+
q=q,
|
| 882 |
+
k=k,
|
| 883 |
+
v=v,
|
| 884 |
+
g=g,
|
| 885 |
+
beta=beta,
|
| 886 |
+
A_log=self.A_log,
|
| 887 |
+
dt_bias=self.dt_bias,
|
| 888 |
+
initial_state=recurrent_state,
|
| 889 |
+
output_final_state=True,
|
| 890 |
+
use_qk_l2norm_in_kernel=True,
|
| 891 |
+
use_gate_in_kernel=True,
|
| 892 |
+
lower_bound=self.lower_bound,
|
| 893 |
+
cu_seqlens=cu_seqlens,
|
| 894 |
+
)
|
| 895 |
+
|
| 896 |
+
if use_cache and past_key_value is not None and isinstance(past_key_value, Cache):
|
| 897 |
+
target_device = None
|
| 898 |
+
for cache in past_key_value.layers:
|
| 899 |
+
if cache.keys is not None:
|
| 900 |
+
target_device = cache.keys.device
|
| 901 |
+
break
|
| 902 |
+
if target_device is None:
|
| 903 |
+
target_device = recurrent_state.device
|
| 904 |
+
|
| 905 |
+
# move to target device
|
| 906 |
+
if recurrent_state.device != target_device:
|
| 907 |
+
recurrent_state = recurrent_state.to(target_device)
|
| 908 |
+
|
| 909 |
+
past_key_value.layers[self.layer_idx].keys = recurrent_state
|
| 910 |
+
past_key_value.layers[self.layer_idx].values = (conv_state_q, conv_state_k, conv_state_v)
|
| 911 |
+
|
| 912 |
+
if self.no_kda_lora:
|
| 913 |
+
g = self.g_proj(hidden_states)
|
| 914 |
+
else:
|
| 915 |
+
g = self.g_b_proj(self.g_a_proj(hidden_states))
|
| 916 |
+
g = rearrange(g, '... (h d) -> ... h d', d=self.head_dim)
|
| 917 |
+
o = self.o_norm(o, g)
|
| 918 |
+
|
| 919 |
+
o = rearrange(o, 'b t h d -> b t (h d)')
|
| 920 |
+
o = self.o_proj(o)
|
| 921 |
+
if attention_mask is not None:
|
| 922 |
+
o = pad_input(o.squeeze(0), indices, batch_size, q_len)
|
| 923 |
+
|
| 924 |
+
return o, None, past_key_value
|
| 925 |
+
|
| 926 |
+
|
| 927 |
+
class BailingMoeV3MTPLayer(nn.Module):
|
| 928 |
+
def __init__(self, config: BailingMoeV3Config, layer_idx: int):
|
| 929 |
+
super().__init__()
|
| 930 |
+
self.layer_idx = layer_idx
|
| 931 |
+
self.input_layernorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 932 |
+
self.enorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 933 |
+
|
| 934 |
+
self.eh_proj = nn.Linear(config.hidden_size * 2, config.hidden_size, bias=False)
|
| 935 |
+
self.post_attention_layernorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 936 |
+
self.attention = BailingMoeV3MultiLatentAttention(config=config, layer_idx=layer_idx)
|
| 937 |
+
self.mlp = BailingMoeV3SparseMoeBlock(config)
|
| 938 |
+
|
| 939 |
+
self.hnorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 940 |
+
self.final_layernorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 941 |
+
|
| 942 |
+
def forward(
|
| 943 |
+
self,
|
| 944 |
+
input_embeds,
|
| 945 |
+
hidden_states: torch.Tensor,
|
| 946 |
+
attention_mask: Optional[torch.Tensor] = None,
|
| 947 |
+
position_ids: Optional[torch.LongTensor] = None,
|
| 948 |
+
past_key_value: Optional[Tuple[torch.Tensor]] = None,
|
| 949 |
+
output_attentions: Optional[bool] = False,
|
| 950 |
+
output_router_logits: Optional[bool] = False,
|
| 951 |
+
use_cache: Optional[bool] = False,
|
| 952 |
+
position_embeddings: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, # necessary, but kept here for BC
|
| 953 |
+
**kwargs,
|
| 954 |
+
) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
|
| 955 |
+
input_embeds = self.enorm(input_embeds)
|
| 956 |
+
hidden_states = self.hnorm(hidden_states)
|
| 957 |
+
hidden_states = self.eh_proj(torch.cat([input_embeds, hidden_states], dim=-1))
|
| 958 |
+
residual = hidden_states
|
| 959 |
+
|
| 960 |
+
hidden_states = self.input_layernorm(hidden_states)
|
| 961 |
+
|
| 962 |
+
# Self Attention
|
| 963 |
+
hidden_states, self_attn_weights, present_key_value = self.attention(
|
| 964 |
+
hidden_states=hidden_states,
|
| 965 |
+
attention_mask=attention_mask,
|
| 966 |
+
position_ids=position_ids,
|
| 967 |
+
past_key_value=past_key_value,
|
| 968 |
+
output_attentions=output_attentions,
|
| 969 |
+
position_embeddings=position_embeddings,
|
| 970 |
+
use_cache=use_cache,
|
| 971 |
+
)
|
| 972 |
+
hidden_states = residual + hidden_states
|
| 973 |
+
|
| 974 |
+
# Fully Connected
|
| 975 |
+
residual = hidden_states
|
| 976 |
+
hidden_states = self.post_attention_layernorm(hidden_states)
|
| 977 |
+
hidden_states = self.mlp(hidden_states)
|
| 978 |
+
if isinstance(hidden_states, tuple):
|
| 979 |
+
hidden_states, router_logits = hidden_states
|
| 980 |
+
else:
|
| 981 |
+
router_logits = None
|
| 982 |
+
hidden_states = residual + hidden_states.to(residual.device)
|
| 983 |
+
hidden_states = self.final_layernorm(hidden_states)
|
| 984 |
+
|
| 985 |
+
outputs = (hidden_states,)
|
| 986 |
+
|
| 987 |
+
if output_attentions:
|
| 988 |
+
outputs += (self_attn_weights,)
|
| 989 |
+
|
| 990 |
+
if use_cache:
|
| 991 |
+
outputs += (present_key_value,)
|
| 992 |
+
|
| 993 |
+
if output_router_logits:
|
| 994 |
+
outputs += (router_logits,)
|
| 995 |
+
|
| 996 |
+
return outputs
|
| 997 |
+
|
| 998 |
+
|
| 999 |
+
class BailingMoeV3DecoderLayer(nn.Module):
|
| 1000 |
+
def __init__(self, config: BailingMoeV3Config, layer_idx: int):
|
| 1001 |
+
super().__init__()
|
| 1002 |
+
self.hidden_size = config.hidden_size
|
| 1003 |
+
self.layer_idx = layer_idx
|
| 1004 |
+
self.attention_layer_type = (
|
| 1005 |
+
"attention"
|
| 1006 |
+
if (layer_idx + 1) % config.layer_group_size == 0
|
| 1007 |
+
or layer_idx >= config.num_hidden_layers // config.layer_group_size * config.layer_group_size
|
| 1008 |
+
else "linear_attention"
|
| 1009 |
+
)
|
| 1010 |
+
|
| 1011 |
+
if self.attention_layer_type == "attention":
|
| 1012 |
+
self.attention = BailingMoeV3MultiLatentAttention(config=config, layer_idx=layer_idx)
|
| 1013 |
+
else:
|
| 1014 |
+
self.attention = BailingMoeV3KimiDeltaAttention(config=config, layer_idx=layer_idx)
|
| 1015 |
+
|
| 1016 |
+
self.mlp = (
|
| 1017 |
+
BailingMoeV3SparseMoeBlock(config)
|
| 1018 |
+
if (config.num_experts is not None and layer_idx >= config.first_k_dense_replace)
|
| 1019 |
+
else BailingMoeV3MLP(config=config, intermediate_size=config.intermediate_size)
|
| 1020 |
+
)
|
| 1021 |
+
self.input_layernorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 1022 |
+
self.post_attention_layernorm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 1023 |
+
|
| 1024 |
+
def forward(
|
| 1025 |
+
self,
|
| 1026 |
+
hidden_states: torch.Tensor,
|
| 1027 |
+
attention_mask: Optional[torch.Tensor] = None,
|
| 1028 |
+
position_ids: Optional[torch.LongTensor] = None,
|
| 1029 |
+
past_key_value: Optional[Tuple[torch.Tensor]] = None,
|
| 1030 |
+
cache_position: Optional[torch.LongTensor] = None,
|
| 1031 |
+
output_attentions: Optional[bool] = False,
|
| 1032 |
+
output_router_logits: Optional[bool] = False,
|
| 1033 |
+
use_cache: Optional[bool] = False,
|
| 1034 |
+
position_embeddings: Optional[Tuple[torch.Tensor, torch.Tensor]] = None, # necessary, but kept here for BC
|
| 1035 |
+
**kwargs,
|
| 1036 |
+
) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
|
| 1037 |
+
"""
|
| 1038 |
+
Args:
|
| 1039 |
+
hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
|
| 1040 |
+
attention_mask (`torch.FloatTensor`, *optional*):
|
| 1041 |
+
attention mask of size `(batch_size, sequence_length)` if flash attention is used or `(batch_size, 1,
|
| 1042 |
+
query_sequence_length, key_sequence_length)` if default attention is used.
|
| 1043 |
+
position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1044 |
+
Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
|
| 1045 |
+
config.n_positions - 1]`.
|
| 1046 |
+
past_key_value (`Tuple(torch.FloatTensor)`, *optional*):
|
| 1047 |
+
cached past key and value projection states
|
| 1048 |
+
output_attentions (`bool`, *optional*):
|
| 1049 |
+
Whether to return the attentions tensors of all attention layers. See `attentions` under
|
| 1050 |
+
returned tensors for more detail.
|
| 1051 |
+
output_router_logits (`bool`, *optional*):
|
| 1052 |
+
Whether or not to return the logits of all the routers. They are useful for computing the router loss,
|
| 1053 |
+
and should not be returned during inference.
|
| 1054 |
+
use_cache (`bool`, *optional*):
|
| 1055 |
+
If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
|
| 1056 |
+
(see `past_key_values`).
|
| 1057 |
+
"""
|
| 1058 |
+
residual = hidden_states
|
| 1059 |
+
|
| 1060 |
+
hidden_states = self.input_layernorm(hidden_states)
|
| 1061 |
+
|
| 1062 |
+
# Self Attention
|
| 1063 |
+
if self.attention_layer_type == "attention":
|
| 1064 |
+
hidden_states, self_attn_weights, present_key_value = self.attention(
|
| 1065 |
+
hidden_states=hidden_states,
|
| 1066 |
+
attention_mask=attention_mask,
|
| 1067 |
+
position_ids=position_ids,
|
| 1068 |
+
past_key_values=past_key_value,
|
| 1069 |
+
use_cache=use_cache,
|
| 1070 |
+
cache_position=cache_position, #
|
| 1071 |
+
position_embeddings=position_embeddings, #
|
| 1072 |
+
**kwargs,
|
| 1073 |
+
)
|
| 1074 |
+
else:
|
| 1075 |
+
batch_size, seq_len = hidden_states.shape[0], hidden_states.shape[1]
|
| 1076 |
+
device = hidden_states.device
|
| 1077 |
+
|
| 1078 |
+
if attention_mask is None:
|
| 1079 |
+
# if attention_mask is None, create a full mask
|
| 1080 |
+
attention_mask = torch.ones((batch_size, seq_len), dtype=torch.int32, device=device)
|
| 1081 |
+
elif attention_mask.dim() == 4 and attention_mask.shape[1] == 1:
|
| 1082 |
+
attention_mask = attention_mask[:, 0, -1, :].to(torch.int32)
|
| 1083 |
+
attention_mask = (attention_mask > -1e4).to(torch.int32)
|
| 1084 |
+
elif attention_mask.dim() == 2:
|
| 1085 |
+
attention_mask = attention_mask.to(torch.int32)
|
| 1086 |
+
else:
|
| 1087 |
+
raise ValueError(f"Unsupported mask dimension: {attention_mask.shape}")
|
| 1088 |
+
|
| 1089 |
+
hidden_states, self_attn_weights, present_key_value = self.attention(
|
| 1090 |
+
hidden_states=hidden_states,
|
| 1091 |
+
attention_mask=attention_mask,
|
| 1092 |
+
past_key_value=past_key_value,
|
| 1093 |
+
position_ids=position_ids,
|
| 1094 |
+
use_cache=use_cache,
|
| 1095 |
+
output_attentions=output_attentions,
|
| 1096 |
+
)
|
| 1097 |
+
|
| 1098 |
+
hidden_states = residual + hidden_states
|
| 1099 |
+
|
| 1100 |
+
# Fully Connected
|
| 1101 |
+
residual = hidden_states
|
| 1102 |
+
hidden_states = self.post_attention_layernorm(hidden_states)
|
| 1103 |
+
hidden_states = self.mlp(hidden_states)
|
| 1104 |
+
if isinstance(hidden_states, tuple):
|
| 1105 |
+
hidden_states, router_logits = hidden_states
|
| 1106 |
+
else:
|
| 1107 |
+
router_logits = None
|
| 1108 |
+
hidden_states = residual + hidden_states.to(residual.device)
|
| 1109 |
+
|
| 1110 |
+
outputs = (hidden_states,)
|
| 1111 |
+
|
| 1112 |
+
if output_attentions:
|
| 1113 |
+
outputs += (self_attn_weights,)
|
| 1114 |
+
|
| 1115 |
+
if use_cache:
|
| 1116 |
+
outputs += (present_key_value,)
|
| 1117 |
+
|
| 1118 |
+
if output_router_logits:
|
| 1119 |
+
outputs += (router_logits,)
|
| 1120 |
+
|
| 1121 |
+
return outputs
|
| 1122 |
+
|
| 1123 |
+
|
| 1124 |
+
BAILINGMOEV3_START_DOCSTRING = r"""
|
| 1125 |
+
This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
|
| 1126 |
+
library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
|
| 1127 |
+
etc.)
|
| 1128 |
+
This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
|
| 1129 |
+
Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
|
| 1130 |
+
and behavior.
|
| 1131 |
+
Parameters:
|
| 1132 |
+
config ([`BailingMoeV3Config`]):
|
| 1133 |
+
Model configuration class with all the parameters of the model. Initializing with a config file does not
|
| 1134 |
+
load the weights associated with the model, only the configuration. Check out the
|
| 1135 |
+
[`~PreTrainedModel.from_pretrained`] method to load the model weights.
|
| 1136 |
+
"""
|
| 1137 |
+
|
| 1138 |
+
|
| 1139 |
+
@add_start_docstrings(
|
| 1140 |
+
"The bare BailingMoeV3 Model outputting raw hidden-states without any specific head on top.",
|
| 1141 |
+
BAILINGMOEV3_START_DOCSTRING,
|
| 1142 |
+
)
|
| 1143 |
+
class BailingMoeV3PreTrainedModel(PreTrainedModel):
|
| 1144 |
+
config_class = BailingMoeV3Config
|
| 1145 |
+
base_model_prefix = "model"
|
| 1146 |
+
supports_gradient_checkpointing = True
|
| 1147 |
+
_no_split_modules = ["BailingMoeV3DecoderLayer"]
|
| 1148 |
+
_skip_keys_device_placement = "past_key_values"
|
| 1149 |
+
_supports_flash_attn_2 = True
|
| 1150 |
+
_supports_sdpa = True
|
| 1151 |
+
_supports_cache_class = True
|
| 1152 |
+
|
| 1153 |
+
def _init_weights(self, module):
|
| 1154 |
+
std = self.config.initializer_range
|
| 1155 |
+
if isinstance(module, nn.Linear):
|
| 1156 |
+
module.weight.data.normal_(mean=0.0, std=std)
|
| 1157 |
+
if module.bias is not None:
|
| 1158 |
+
module.bias.data.zero_()
|
| 1159 |
+
elif isinstance(module, nn.Embedding):
|
| 1160 |
+
module.weight.data.normal_(mean=0.0, std=std)
|
| 1161 |
+
if module.padding_idx is not None:
|
| 1162 |
+
module.weight.data[module.padding_idx].zero_()
|
| 1163 |
+
|
| 1164 |
+
|
| 1165 |
+
BAILINGMOEV3_INPUTS_DOCSTRING = r"""
|
| 1166 |
+
Args:
|
| 1167 |
+
input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
|
| 1168 |
+
Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
|
| 1169 |
+
it.
|
| 1170 |
+
Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
|
| 1171 |
+
[`PreTrainedTokenizer.__call__`] for details.
|
| 1172 |
+
[What are input IDs?](../glossary#input-ids)
|
| 1173 |
+
attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1174 |
+
Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
|
| 1175 |
+
- 1 for tokens that are **not masked**,
|
| 1176 |
+
- 0 for tokens that are **masked**.
|
| 1177 |
+
[What are attention masks?](../glossary#attention-mask)
|
| 1178 |
+
Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
|
| 1179 |
+
[`PreTrainedTokenizer.__call__`] for details.
|
| 1180 |
+
If `past_key_values` is used, optionally only the last `input_ids` have to be input (see
|
| 1181 |
+
`past_key_values`).
|
| 1182 |
+
If you want to change padding behavior, you should read [`modeling_opt._prepare_decoder_attention_mask`]
|
| 1183 |
+
and modify to your needs. See diagram 1 in [the paper](https://arxiv.org/abs/1910.13461) for more
|
| 1184 |
+
information on the default strategy.
|
| 1185 |
+
- 1 indicates the head is **not masked**,
|
| 1186 |
+
- 0 indicates the head is **masked**.
|
| 1187 |
+
position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1188 |
+
Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
|
| 1189 |
+
config.n_positions - 1]`.
|
| 1190 |
+
[What are position IDs?](../glossary#position-ids)
|
| 1191 |
+
past_key_values (`Cache` or `tuple(tuple(torch.FloatTensor))`, *optional*):
|
| 1192 |
+
Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
|
| 1193 |
+
blocks) that can be used to speed up sequential decoding. This typically consists in the `past_key_values`
|
| 1194 |
+
returned by the model at a previous stage of decoding, when `use_cache=True` or `config.use_cache=True`.
|
| 1195 |
+
Two formats are allowed:
|
| 1196 |
+
- a [`~cache_utils.Cache`] instance;
|
| 1197 |
+
- Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of
|
| 1198 |
+
shape `(batch_size, num_heads, sequence_length, embed_size_per_head)`). This is also known as the legacy
|
| 1199 |
+
cache format.
|
| 1200 |
+
The model will output the same cache format that is fed as input. If no `past_key_values` are passed, the
|
| 1201 |
+
legacy cache format will be returned.
|
| 1202 |
+
If `past_key_values` are used, the user can optionally input only the last `input_ids` (those that don't
|
| 1203 |
+
have their past key value states given to this model) of shape `(batch_size, 1)` instead of all `input_ids`
|
| 1204 |
+
of shape `(batch_size, sequence_length)`.
|
| 1205 |
+
inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
|
| 1206 |
+
Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
|
| 1207 |
+
is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
|
| 1208 |
+
model's internal embedding lookup matrix.
|
| 1209 |
+
use_cache (`bool`, *optional*):
|
| 1210 |
+
If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
|
| 1211 |
+
`past_key_values`).
|
| 1212 |
+
output_attentions (`bool`, *optional*):
|
| 1213 |
+
Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
|
| 1214 |
+
tensors for more detail.
|
| 1215 |
+
output_hidden_states (`bool`, *optional*):
|
| 1216 |
+
Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
|
| 1217 |
+
more detail.
|
| 1218 |
+
return_dict (`bool`, *optional*):
|
| 1219 |
+
Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
|
| 1220 |
+
"""
|
| 1221 |
+
|
| 1222 |
+
|
| 1223 |
+
@add_start_docstrings(
|
| 1224 |
+
"The bare BailingMoeV3 Model outputting raw hidden-states without any specific head on top.",
|
| 1225 |
+
BAILINGMOEV3_START_DOCSTRING,
|
| 1226 |
+
)
|
| 1227 |
+
class BailingMoeV3Model(BailingMoeV3PreTrainedModel):
|
| 1228 |
+
"""
|
| 1229 |
+
Transformer decoder consisting of *config.num_hidden_layers* layers. Each layer is a [`BailingMoeV3DecoderLayer`]
|
| 1230 |
+
Args:
|
| 1231 |
+
config: BailingMoeV3Config
|
| 1232 |
+
"""
|
| 1233 |
+
|
| 1234 |
+
def __init__(self, config: BailingMoeV3Config):
|
| 1235 |
+
super().__init__(config)
|
| 1236 |
+
self.padding_idx = config.pad_token_id
|
| 1237 |
+
self.vocab_size = config.vocab_size
|
| 1238 |
+
self.num_nextn_predict_layers = config.num_nextn_predict_layers
|
| 1239 |
+
|
| 1240 |
+
self.word_embeddings = nn.Embedding(config.vocab_size, config.hidden_size, self.padding_idx)
|
| 1241 |
+
self.layers = []
|
| 1242 |
+
for layer_idx in range(config.num_hidden_layers + config.num_nextn_predict_layers):
|
| 1243 |
+
layer_cls = BailingMoeV3DecoderLayer if layer_idx < config.num_hidden_layers else BailingMoeV3MTPLayer
|
| 1244 |
+
self.layers.append(layer_cls(config, layer_idx))
|
| 1245 |
+
|
| 1246 |
+
self.layers = nn.ModuleList(self.layers)
|
| 1247 |
+
|
| 1248 |
+
self._use_sdpa = config._attn_implementation == "sdpa"
|
| 1249 |
+
self._use_flash_attention_2 = config._attn_implementation == "flash_attention_2"
|
| 1250 |
+
self.norm = BailingMoeV3RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 1251 |
+
self.rotary_emb = BailingMoeV3RotaryEmbedding(config=config)
|
| 1252 |
+
self.gradient_checkpointing = False
|
| 1253 |
+
# Initialize weights and apply final processing
|
| 1254 |
+
self.post_init()
|
| 1255 |
+
|
| 1256 |
+
def get_input_embeddings(self):
|
| 1257 |
+
return self.word_embeddings
|
| 1258 |
+
|
| 1259 |
+
def set_input_embeddings(self, value):
|
| 1260 |
+
self.word_embeddings = value
|
| 1261 |
+
|
| 1262 |
+
@add_start_docstrings_to_model_forward(BAILINGMOEV3_INPUTS_DOCSTRING)
|
| 1263 |
+
def forward(
|
| 1264 |
+
self,
|
| 1265 |
+
input_ids: torch.LongTensor = None,
|
| 1266 |
+
attention_mask: Optional[torch.Tensor] = None,
|
| 1267 |
+
position_ids: Optional[torch.LongTensor] = None,
|
| 1268 |
+
past_key_values: Optional[List[torch.FloatTensor]] = None,
|
| 1269 |
+
inputs_embeds: Optional[torch.FloatTensor] = None,
|
| 1270 |
+
cache_position: Optional[torch.LongTensor] = None,
|
| 1271 |
+
use_cache: Optional[bool] = None,
|
| 1272 |
+
output_attentions: Optional[bool] = None,
|
| 1273 |
+
output_hidden_states: Optional[bool] = None,
|
| 1274 |
+
output_router_logits: Optional[bool] = None,
|
| 1275 |
+
return_dict: Optional[bool] = None,
|
| 1276 |
+
**kwargs,
|
| 1277 |
+
) -> Union[Tuple, MoeV3ModelOutputWithPast]:
|
| 1278 |
+
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
| 1279 |
+
output_hidden_states = (
|
| 1280 |
+
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
| 1281 |
+
)
|
| 1282 |
+
output_router_logits = (
|
| 1283 |
+
output_router_logits if output_router_logits is not None else self.config.output_router_logits
|
| 1284 |
+
)
|
| 1285 |
+
use_cache = use_cache if use_cache is not None else self.config.use_cache
|
| 1286 |
+
|
| 1287 |
+
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
| 1288 |
+
|
| 1289 |
+
# retrieve input_ids and inputs_embeds
|
| 1290 |
+
if input_ids is not None and inputs_embeds is not None:
|
| 1291 |
+
raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
|
| 1292 |
+
elif input_ids is not None:
|
| 1293 |
+
batch_size, seq_length = input_ids.shape[:2]
|
| 1294 |
+
elif inputs_embeds is not None:
|
| 1295 |
+
batch_size, seq_length = inputs_embeds.shape[:2]
|
| 1296 |
+
else:
|
| 1297 |
+
raise ValueError("You have to specify either input_ids or inputs_embeds")
|
| 1298 |
+
|
| 1299 |
+
if self.gradient_checkpointing and self.training:
|
| 1300 |
+
if use_cache:
|
| 1301 |
+
logger.warning_once(
|
| 1302 |
+
"`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`transformers."
|
| 1303 |
+
)
|
| 1304 |
+
use_cache = False
|
| 1305 |
+
|
| 1306 |
+
if use_cache and past_key_values is None:
|
| 1307 |
+
past_key_values = DynamicCache()
|
| 1308 |
+
|
| 1309 |
+
if inputs_embeds is None:
|
| 1310 |
+
inputs_embeds = self.word_embeddings(input_ids)
|
| 1311 |
+
|
| 1312 |
+
if cache_position is None:
|
| 1313 |
+
past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
|
| 1314 |
+
cache_position: torch.Tensor = torch.arange(
|
| 1315 |
+
past_seen_tokens, past_seen_tokens + inputs_embeds.shape[1], device=inputs_embeds.device
|
| 1316 |
+
)
|
| 1317 |
+
|
| 1318 |
+
if position_ids is None:
|
| 1319 |
+
position_ids = cache_position.unsqueeze(0)
|
| 1320 |
+
|
| 1321 |
+
softmax_attention_layer_id = self.config.layer_group_size - 1
|
| 1322 |
+
past_seen_tokens = (
|
| 1323 |
+
past_key_values.get_seq_length(layer_idx=softmax_attention_layer_id) if past_key_values is not None else 0
|
| 1324 |
+
)
|
| 1325 |
+
|
| 1326 |
+
if position_ids is None:
|
| 1327 |
+
position_ids = torch.arange(
|
| 1328 |
+
past_seen_tokens, past_seen_tokens + inputs_embeds.shape[1], device=inputs_embeds.device
|
| 1329 |
+
)
|
| 1330 |
+
position_ids = position_ids.unsqueeze(0)
|
| 1331 |
+
|
| 1332 |
+
if self._use_flash_attention_2:
|
| 1333 |
+
# 2d mask is passed through the layers
|
| 1334 |
+
attention_mask = attention_mask if (attention_mask is not None and 0 in attention_mask) else None
|
| 1335 |
+
elif self._use_sdpa and not output_attentions:
|
| 1336 |
+
# output_attentions=True can not be supported when using SDPA, and we fall back on
|
| 1337 |
+
# the manual implementation that requires a 4D causal mask in all cases.
|
| 1338 |
+
attention_mask = _prepare_4d_causal_attention_mask_for_sdpa(
|
| 1339 |
+
attention_mask,
|
| 1340 |
+
(batch_size, seq_length),
|
| 1341 |
+
inputs_embeds,
|
| 1342 |
+
past_seen_tokens,
|
| 1343 |
+
)
|
| 1344 |
+
else:
|
| 1345 |
+
# 4d mask is passed through the layers
|
| 1346 |
+
attention_mask = _prepare_4d_causal_attention_mask(
|
| 1347 |
+
attention_mask, (batch_size, seq_length), inputs_embeds, past_seen_tokens
|
| 1348 |
+
)
|
| 1349 |
+
|
| 1350 |
+
# embed positions
|
| 1351 |
+
hidden_states = inputs_embeds
|
| 1352 |
+
|
| 1353 |
+
# create position embeddings to be shared across the decoder layers
|
| 1354 |
+
position_embeddings = self.rotary_emb(hidden_states, position_ids)
|
| 1355 |
+
|
| 1356 |
+
# decoder layers
|
| 1357 |
+
all_hidden_states = () if output_hidden_states else None
|
| 1358 |
+
all_self_attns = () if output_attentions else None
|
| 1359 |
+
all_router_logits = () if output_router_logits else None
|
| 1360 |
+
next_decoder_cache = None
|
| 1361 |
+
layers = self.layers[: -self.num_nextn_predict_layers] if self.num_nextn_predict_layers > 0 else self.layers
|
| 1362 |
+
mtp_layers = self.layers[-self.num_nextn_predict_layers :] if self.num_nextn_predict_layers > 0 else None
|
| 1363 |
+
|
| 1364 |
+
# tptest miss causal_mask = create_causal_mask(
|
| 1365 |
+
|
| 1366 |
+
for decoder_layer in layers:
|
| 1367 |
+
if output_hidden_states:
|
| 1368 |
+
all_hidden_states += (hidden_states,)
|
| 1369 |
+
|
| 1370 |
+
if self.gradient_checkpointing and self.training:
|
| 1371 |
+
layer_outputs = self._gradient_checkpointing_func(
|
| 1372 |
+
decoder_layer.__call__,
|
| 1373 |
+
hidden_states,
|
| 1374 |
+
attention_mask,
|
| 1375 |
+
position_ids,
|
| 1376 |
+
past_key_values,
|
| 1377 |
+
cache_position,
|
| 1378 |
+
output_attentions,
|
| 1379 |
+
output_router_logits,
|
| 1380 |
+
use_cache,
|
| 1381 |
+
position_embeddings,
|
| 1382 |
+
)
|
| 1383 |
+
else:
|
| 1384 |
+
layer_outputs = decoder_layer(
|
| 1385 |
+
hidden_states,
|
| 1386 |
+
attention_mask=attention_mask,
|
| 1387 |
+
position_ids=position_ids,
|
| 1388 |
+
past_key_value=past_key_values,
|
| 1389 |
+
cache_position=cache_position,
|
| 1390 |
+
output_attentions=output_attentions,
|
| 1391 |
+
output_router_logits=output_router_logits,
|
| 1392 |
+
use_cache=use_cache,
|
| 1393 |
+
position_embeddings=position_embeddings,
|
| 1394 |
+
)
|
| 1395 |
+
hidden_states = layer_outputs[0]
|
| 1396 |
+
|
| 1397 |
+
if use_cache:
|
| 1398 |
+
next_decoder_cache = layer_outputs[2 if output_attentions else 1]
|
| 1399 |
+
|
| 1400 |
+
if output_attentions:
|
| 1401 |
+
all_self_attns += (layer_outputs[1],)
|
| 1402 |
+
|
| 1403 |
+
if output_router_logits and layer_outputs[-1] is not None:
|
| 1404 |
+
all_router_logits += (layer_outputs[-1],)
|
| 1405 |
+
|
| 1406 |
+
hidden_states = self.norm(hidden_states)
|
| 1407 |
+
main_hidden_states = hidden_states
|
| 1408 |
+
|
| 1409 |
+
# add hidden states from the last decoder layer
|
| 1410 |
+
if output_hidden_states:
|
| 1411 |
+
all_hidden_states += (main_hidden_states,)
|
| 1412 |
+
|
| 1413 |
+
mtp_hidden_states = None
|
| 1414 |
+
|
| 1415 |
+
if mtp_layers:
|
| 1416 |
+
for decoder_layer in mtp_layers:
|
| 1417 |
+
input_ids, _ = roll_tensor(input_ids, shifts=-1, dims=-1)
|
| 1418 |
+
inputs_embeds = self.word_embeddings(input_ids)
|
| 1419 |
+
|
| 1420 |
+
if self.gradient_checkpointing and self.training:
|
| 1421 |
+
layer_outputs = self._gradient_checkpointing_func(
|
| 1422 |
+
decoder_layer.__call__,
|
| 1423 |
+
inputs_embeds,
|
| 1424 |
+
hidden_states,
|
| 1425 |
+
attention_mask,
|
| 1426 |
+
position_ids,
|
| 1427 |
+
past_key_values,
|
| 1428 |
+
output_attentions,
|
| 1429 |
+
output_router_logits,
|
| 1430 |
+
use_cache,
|
| 1431 |
+
position_embeddings,
|
| 1432 |
+
)
|
| 1433 |
+
else:
|
| 1434 |
+
layer_outputs = decoder_layer(
|
| 1435 |
+
inputs_embeds,
|
| 1436 |
+
hidden_states,
|
| 1437 |
+
attention_mask=attention_mask,
|
| 1438 |
+
position_ids=position_ids,
|
| 1439 |
+
past_key_value=past_key_values,
|
| 1440 |
+
output_attentions=output_attentions,
|
| 1441 |
+
output_router_logits=output_router_logits,
|
| 1442 |
+
use_cache=use_cache,
|
| 1443 |
+
position_embeddings=position_embeddings,
|
| 1444 |
+
)
|
| 1445 |
+
if mtp_hidden_states is None:
|
| 1446 |
+
mtp_hidden_states = []
|
| 1447 |
+
hidden_states = layer_outputs[0]
|
| 1448 |
+
mtp_hidden_states.append(hidden_states)
|
| 1449 |
+
|
| 1450 |
+
if output_hidden_states:
|
| 1451 |
+
all_hidden_states += (hidden_states,)
|
| 1452 |
+
|
| 1453 |
+
if use_cache:
|
| 1454 |
+
next_decoder_cache = layer_outputs[2 if output_attentions else 1]
|
| 1455 |
+
|
| 1456 |
+
if output_attentions:
|
| 1457 |
+
all_self_attns += (layer_outputs[1],)
|
| 1458 |
+
|
| 1459 |
+
if output_router_logits and layer_outputs[-1] is not None:
|
| 1460 |
+
all_router_logits += (layer_outputs[-1],)
|
| 1461 |
+
|
| 1462 |
+
next_cache = None
|
| 1463 |
+
if use_cache:
|
| 1464 |
+
next_cache = next_decoder_cache
|
| 1465 |
+
if not return_dict:
|
| 1466 |
+
return tuple(
|
| 1467 |
+
v
|
| 1468 |
+
for v in [main_hidden_states, next_cache, all_hidden_states, all_self_attns, all_router_logits]
|
| 1469 |
+
if v is not None
|
| 1470 |
+
)
|
| 1471 |
+
return MoeV3ModelOutputWithPast(
|
| 1472 |
+
last_hidden_state=main_hidden_states,
|
| 1473 |
+
past_key_values=next_cache,
|
| 1474 |
+
hidden_states=all_hidden_states,
|
| 1475 |
+
mtp_hidden_states=mtp_hidden_states,
|
| 1476 |
+
attentions=all_self_attns,
|
| 1477 |
+
router_logits=all_router_logits,
|
| 1478 |
+
)
|
| 1479 |
+
|
| 1480 |
+
|
| 1481 |
+
class BailingMoeV3ForCausalLM(BailingMoeV3PreTrainedModel, GenerationMixin):
|
| 1482 |
+
_tied_weights_keys = ["lm_head.weight"]
|
| 1483 |
+
|
| 1484 |
+
def __init__(self, config: BailingMoeV3Config):
|
| 1485 |
+
super().__init__(config)
|
| 1486 |
+
self.model = BailingMoeV3Model(config)
|
| 1487 |
+
self.vocab_size = config.vocab_size
|
| 1488 |
+
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
| 1489 |
+
self.num_nextn_predict_layers = config.num_nextn_predict_layers
|
| 1490 |
+
self.mtp_loss_scaling_factor = config.mtp_loss_scaling_factor
|
| 1491 |
+
|
| 1492 |
+
# Initialize weights and apply final processing
|
| 1493 |
+
self.post_init()
|
| 1494 |
+
|
| 1495 |
+
def get_input_embeddings(self):
|
| 1496 |
+
return self.model.word_embeddings
|
| 1497 |
+
|
| 1498 |
+
def set_input_embeddings(self, value):
|
| 1499 |
+
self.model.word_embeddings = value
|
| 1500 |
+
|
| 1501 |
+
def get_output_embeddings(self):
|
| 1502 |
+
return self.lm_head
|
| 1503 |
+
|
| 1504 |
+
def set_output_embeddings(self, new_embeddings):
|
| 1505 |
+
self.lm_head = new_embeddings
|
| 1506 |
+
|
| 1507 |
+
def set_decoder(self, decoder):
|
| 1508 |
+
self.model = decoder
|
| 1509 |
+
|
| 1510 |
+
def get_decoder(self):
|
| 1511 |
+
return self.model
|
| 1512 |
+
|
| 1513 |
+
@add_start_docstrings_to_model_forward(BAILINGMOEV3_INPUTS_DOCSTRING)
|
| 1514 |
+
@replace_return_docstrings(output_type=MoEV3CausalLMOutputWithPast, config_class=_CONFIG_FOR_DOC)
|
| 1515 |
+
def forward(
|
| 1516 |
+
self,
|
| 1517 |
+
input_ids: torch.LongTensor = None,
|
| 1518 |
+
attention_mask: Optional[torch.Tensor] = None,
|
| 1519 |
+
position_ids: Optional[torch.LongTensor] = None,
|
| 1520 |
+
past_key_values: Optional[List[torch.FloatTensor]] = None,
|
| 1521 |
+
inputs_embeds: Optional[torch.FloatTensor] = None,
|
| 1522 |
+
labels: Optional[torch.LongTensor] = None,
|
| 1523 |
+
use_cache: Optional[bool] = None,
|
| 1524 |
+
output_attentions: Optional[bool] = None,
|
| 1525 |
+
output_hidden_states: Optional[bool] = None,
|
| 1526 |
+
output_router_logits: Optional[bool] = None,
|
| 1527 |
+
return_dict: Optional[bool] = None,
|
| 1528 |
+
**kwargs,
|
| 1529 |
+
) -> Union[Tuple, MoEV3CausalLMOutputWithPast]:
|
| 1530 |
+
r"""
|
| 1531 |
+
Args:
|
| 1532 |
+
labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1533 |
+
Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
|
| 1534 |
+
config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
|
| 1535 |
+
(masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
|
| 1536 |
+
Returns:
|
| 1537 |
+
Example:
|
| 1538 |
+
```python
|
| 1539 |
+
>>> from transformers import AutoTokenizer
|
| 1540 |
+
>>> model = BailingMoeV3ForCausalLM.from_pretrained(PATH_TO_CONVERTED_WEIGHTS)
|
| 1541 |
+
>>> tokenizer = AutoTokenizer.from_pretrained(PATH_TO_CONVERTED_TOKENIZER)
|
| 1542 |
+
>>> prompt = "Hey, are you conscious? Can you talk to me?"
|
| 1543 |
+
>>> inputs = tokenizer(prompt, return_tensors="pt")
|
| 1544 |
+
>>> # Generate
|
| 1545 |
+
>>> generate_ids = model.generate(inputs.input_ids, max_length=30)
|
| 1546 |
+
>>> tokenizer.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
|
| 1547 |
+
"Hey, are you conscious? Can you talk to me?\nI'm not conscious, but I can talk to you."
|
| 1548 |
+
```"""
|
| 1549 |
+
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
|
| 1550 |
+
output_hidden_states = (
|
| 1551 |
+
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
|
| 1552 |
+
)
|
| 1553 |
+
output_router_logits = (
|
| 1554 |
+
output_router_logits if output_router_logits is not None else self.config.output_router_logits
|
| 1555 |
+
)
|
| 1556 |
+
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
| 1557 |
+
# decoder outputs consists of (dec_features, layer_state, dec_hidden, dec_attn)
|
| 1558 |
+
outputs = self.model(
|
| 1559 |
+
input_ids=input_ids,
|
| 1560 |
+
attention_mask=attention_mask,
|
| 1561 |
+
position_ids=position_ids,
|
| 1562 |
+
past_key_values=past_key_values,
|
| 1563 |
+
inputs_embeds=inputs_embeds,
|
| 1564 |
+
use_cache=use_cache,
|
| 1565 |
+
output_attentions=output_attentions,
|
| 1566 |
+
output_hidden_states=output_hidden_states,
|
| 1567 |
+
output_router_logits=output_router_logits,
|
| 1568 |
+
return_dict=return_dict,
|
| 1569 |
+
**kwargs,
|
| 1570 |
+
)
|
| 1571 |
+
|
| 1572 |
+
loss = None
|
| 1573 |
+
all_mtp_loss = None
|
| 1574 |
+
aux_loss = None
|
| 1575 |
+
hidden_states = outputs[0]
|
| 1576 |
+
logits = self.lm_head(hidden_states)
|
| 1577 |
+
logits = logits.float()
|
| 1578 |
+
|
| 1579 |
+
if labels is not None:
|
| 1580 |
+
loss = self.loss_function(logits, labels, self.config.vocab_size, **kwargs)
|
| 1581 |
+
|
| 1582 |
+
all_mtp_logits = None
|
| 1583 |
+
if self.num_nextn_predict_layers > 0:
|
| 1584 |
+
mtp_hidden_states = outputs.mtp_hidden_states
|
| 1585 |
+
shift_labels_mtp = None
|
| 1586 |
+
for i in range(self.num_nextn_predict_layers):
|
| 1587 |
+
mtp_hidden_states = mtp_hidden_states[i]
|
| 1588 |
+
mtp_logits = self.lm_head(mtp_hidden_states).float()
|
| 1589 |
+
if all_mtp_logits is None:
|
| 1590 |
+
all_mtp_logits = []
|
| 1591 |
+
all_mtp_logits.append(mtp_logits)
|
| 1592 |
+
if labels is not None:
|
| 1593 |
+
if shift_labels_mtp is None:
|
| 1594 |
+
shift_labels_mtp = labels.clone()
|
| 1595 |
+
shift_labels_mtp, _ = roll_tensor(shift_labels_mtp, shifts=-1, dims=-1, fill_value=-100)
|
| 1596 |
+
mtp_logits_ = mtp_logits.view(-1, self.config.vocab_size)
|
| 1597 |
+
mtp_loss = self.loss_function(
|
| 1598 |
+
mtp_logits_, shift_labels_mtp.to(mtp_logits_.device).view(-1), self.config.vocab_size, **kwargs
|
| 1599 |
+
)
|
| 1600 |
+
if loss is not None:
|
| 1601 |
+
loss += self.mtp_loss_scaling_factor * mtp_loss
|
| 1602 |
+
else:
|
| 1603 |
+
loss = self.mtp_loss_scaling_factor * mtp_loss
|
| 1604 |
+
|
| 1605 |
+
if all_mtp_loss is None:
|
| 1606 |
+
all_mtp_loss = []
|
| 1607 |
+
all_mtp_loss.append(mtp_loss)
|
| 1608 |
+
|
| 1609 |
+
if not return_dict:
|
| 1610 |
+
output = (logits,) + outputs[1:]
|
| 1611 |
+
if output_router_logits:
|
| 1612 |
+
output = (aux_loss,) + output
|
| 1613 |
+
return (loss,) + output if loss is not None else output
|
| 1614 |
+
|
| 1615 |
+
return MoEV3CausalLMOutputWithPast(
|
| 1616 |
+
loss=loss,
|
| 1617 |
+
mtp_loss=all_mtp_loss,
|
| 1618 |
+
aux_loss=aux_loss,
|
| 1619 |
+
logits=logits,
|
| 1620 |
+
mtp_logits=all_mtp_logits,
|
| 1621 |
+
past_key_values=outputs.past_key_values,
|
| 1622 |
+
hidden_states=outputs.hidden_states,
|
| 1623 |
+
attentions=outputs.attentions,
|
| 1624 |
+
router_logits=outputs.router_logits,
|
| 1625 |
+
)
|
special_tokens_map.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": {
|
| 3 |
+
"content": "<|startoftext|>",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"cls_token": {
|
| 10 |
+
"content": "[CLS]",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"eos_token": {
|
| 17 |
+
"content": "<|endoftext|>",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
},
|
| 23 |
+
"pad_token": {
|
| 24 |
+
"content": "<|endoftext|>",
|
| 25 |
+
"lstrip": false,
|
| 26 |
+
"normalized": false,
|
| 27 |
+
"rstrip": false,
|
| 28 |
+
"single_word": false
|
| 29 |
+
}
|
| 30 |
+
}
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:69b270ecab6ec613830ac2ca66e020af842ea1a81a97e4e2bf47142c911c68d8
|
| 3 |
+
size 12205755
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,2114 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_bos_token": false,
|
| 3 |
+
"add_eos_token": false,
|
| 4 |
+
"added_tokens_decoder": {
|
| 5 |
+
"156891": {
|
| 6 |
+
"content": "<|startoftext|>",
|
| 7 |
+
"lstrip": false,
|
| 8 |
+
"normalized": false,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false,
|
| 11 |
+
"special": true
|
| 12 |
+
},
|
| 13 |
+
"156892": {
|
| 14 |
+
"content": "<|endoftext|>",
|
| 15 |
+
"lstrip": false,
|
| 16 |
+
"normalized": false,
|
| 17 |
+
"rstrip": false,
|
| 18 |
+
"single_word": false,
|
| 19 |
+
"special": true
|
| 20 |
+
},
|
| 21 |
+
"156893": {
|
| 22 |
+
"content": "[CLS]",
|
| 23 |
+
"lstrip": false,
|
| 24 |
+
"normalized": false,
|
| 25 |
+
"rstrip": false,
|
| 26 |
+
"single_word": false,
|
| 27 |
+
"special": true
|
| 28 |
+
},
|
| 29 |
+
"156894": {
|
| 30 |
+
"content": "[gMASK]",
|
| 31 |
+
"lstrip": false,
|
| 32 |
+
"normalized": false,
|
| 33 |
+
"rstrip": false,
|
| 34 |
+
"single_word": false,
|
| 35 |
+
"special": true
|
| 36 |
+
},
|
| 37 |
+
"156895": {
|
| 38 |
+
"content": "<|role_end|>",
|
| 39 |
+
"lstrip": false,
|
| 40 |
+
"normalized": false,
|
| 41 |
+
"rstrip": false,
|
| 42 |
+
"single_word": false,
|
| 43 |
+
"special": true
|
| 44 |
+
},
|
| 45 |
+
"156896": {
|
| 46 |
+
"content": "<tool_call>",
|
| 47 |
+
"lstrip": false,
|
| 48 |
+
"normalized": false,
|
| 49 |
+
"rstrip": false,
|
| 50 |
+
"single_word": false,
|
| 51 |
+
"special": false
|
| 52 |
+
},
|
| 53 |
+
"156897": {
|
| 54 |
+
"content": "</tool_call>",
|
| 55 |
+
"lstrip": false,
|
| 56 |
+
"normalized": false,
|
| 57 |
+
"rstrip": false,
|
| 58 |
+
"single_word": false,
|
| 59 |
+
"special": false
|
| 60 |
+
},
|
| 61 |
+
"156898": {
|
| 62 |
+
"content": "<tool_response>",
|
| 63 |
+
"lstrip": false,
|
| 64 |
+
"normalized": false,
|
| 65 |
+
"rstrip": false,
|
| 66 |
+
"single_word": false,
|
| 67 |
+
"special": false
|
| 68 |
+
},
|
| 69 |
+
"156899": {
|
| 70 |
+
"content": "</tool_response>",
|
| 71 |
+
"lstrip": false,
|
| 72 |
+
"normalized": false,
|
| 73 |
+
"rstrip": false,
|
| 74 |
+
"single_word": false,
|
| 75 |
+
"special": false
|
| 76 |
+
},
|
| 77 |
+
"156900": {
|
| 78 |
+
"content": "<|fim_start|>",
|
| 79 |
+
"lstrip": false,
|
| 80 |
+
"normalized": false,
|
| 81 |
+
"rstrip": false,
|
| 82 |
+
"single_word": false,
|
| 83 |
+
"special": true
|
| 84 |
+
},
|
| 85 |
+
"156901": {
|
| 86 |
+
"content": "<|fim_hole|>",
|
| 87 |
+
"lstrip": false,
|
| 88 |
+
"normalized": false,
|
| 89 |
+
"rstrip": false,
|
| 90 |
+
"single_word": false,
|
| 91 |
+
"special": true
|
| 92 |
+
},
|
| 93 |
+
"156902": {
|
| 94 |
+
"content": "<|fim_end|>",
|
| 95 |
+
"lstrip": false,
|
| 96 |
+
"normalized": false,
|
| 97 |
+
"rstrip": false,
|
| 98 |
+
"single_word": false,
|
| 99 |
+
"special": true
|
| 100 |
+
},
|
| 101 |
+
"156903": {
|
| 102 |
+
"content": "<|reserved_token_8|>",
|
| 103 |
+
"lstrip": false,
|
| 104 |
+
"normalized": false,
|
| 105 |
+
"rstrip": false,
|
| 106 |
+
"single_word": false,
|
| 107 |
+
"special": true
|
| 108 |
+
},
|
| 109 |
+
"156904": {
|
| 110 |
+
"content": "<|reserved_token_9|>",
|
| 111 |
+
"lstrip": false,
|
| 112 |
+
"normalized": false,
|
| 113 |
+
"rstrip": false,
|
| 114 |
+
"single_word": false,
|
| 115 |
+
"special": true
|
| 116 |
+
},
|
| 117 |
+
"156905": {
|
| 118 |
+
"content": "<arg_key>",
|
| 119 |
+
"lstrip": false,
|
| 120 |
+
"normalized": false,
|
| 121 |
+
"rstrip": false,
|
| 122 |
+
"single_word": false,
|
| 123 |
+
"special": false
|
| 124 |
+
},
|
| 125 |
+
"156906": {
|
| 126 |
+
"content": "</arg_key>",
|
| 127 |
+
"lstrip": false,
|
| 128 |
+
"normalized": false,
|
| 129 |
+
"rstrip": false,
|
| 130 |
+
"single_word": false,
|
| 131 |
+
"special": false
|
| 132 |
+
},
|
| 133 |
+
"156907": {
|
| 134 |
+
"content": "<arg_value>",
|
| 135 |
+
"lstrip": false,
|
| 136 |
+
"normalized": false,
|
| 137 |
+
"rstrip": false,
|
| 138 |
+
"single_word": false,
|
| 139 |
+
"special": false
|
| 140 |
+
},
|
| 141 |
+
"156908": {
|
| 142 |
+
"content": "</arg_value>",
|
| 143 |
+
"lstrip": false,
|
| 144 |
+
"normalized": false,
|
| 145 |
+
"rstrip": false,
|
| 146 |
+
"single_word": false,
|
| 147 |
+
"special": false
|
| 148 |
+
},
|
| 149 |
+
"156909": {
|
| 150 |
+
"content": "<|reserved_token_14|>",
|
| 151 |
+
"lstrip": false,
|
| 152 |
+
"normalized": false,
|
| 153 |
+
"rstrip": false,
|
| 154 |
+
"single_word": false,
|
| 155 |
+
"special": true
|
| 156 |
+
},
|
| 157 |
+
"156910": {
|
| 158 |
+
"content": "<|reserved_token_15|>",
|
| 159 |
+
"lstrip": false,
|
| 160 |
+
"normalized": false,
|
| 161 |
+
"rstrip": false,
|
| 162 |
+
"single_word": false,
|
| 163 |
+
"special": true
|
| 164 |
+
},
|
| 165 |
+
"156911": {
|
| 166 |
+
"content": "<|reserved_token_16|>",
|
| 167 |
+
"lstrip": false,
|
| 168 |
+
"normalized": false,
|
| 169 |
+
"rstrip": false,
|
| 170 |
+
"single_word": false,
|
| 171 |
+
"special": true
|
| 172 |
+
},
|
| 173 |
+
"156912": {
|
| 174 |
+
"content": "<|reserved_token_17|>",
|
| 175 |
+
"lstrip": false,
|
| 176 |
+
"normalized": false,
|
| 177 |
+
"rstrip": false,
|
| 178 |
+
"single_word": false,
|
| 179 |
+
"special": true
|
| 180 |
+
},
|
| 181 |
+
"156913": {
|
| 182 |
+
"content": "<|reserved_token_18|>",
|
| 183 |
+
"lstrip": false,
|
| 184 |
+
"normalized": false,
|
| 185 |
+
"rstrip": false,
|
| 186 |
+
"single_word": false,
|
| 187 |
+
"special": true
|
| 188 |
+
},
|
| 189 |
+
"156914": {
|
| 190 |
+
"content": "<|reserved_token_19|>",
|
| 191 |
+
"lstrip": false,
|
| 192 |
+
"normalized": false,
|
| 193 |
+
"rstrip": false,
|
| 194 |
+
"single_word": false,
|
| 195 |
+
"special": true
|
| 196 |
+
},
|
| 197 |
+
"156915": {
|
| 198 |
+
"content": "<|reserved_token_20|>",
|
| 199 |
+
"lstrip": false,
|
| 200 |
+
"normalized": false,
|
| 201 |
+
"rstrip": false,
|
| 202 |
+
"single_word": false,
|
| 203 |
+
"special": true
|
| 204 |
+
},
|
| 205 |
+
"156916": {
|
| 206 |
+
"content": "<|reserved_token_21|>",
|
| 207 |
+
"lstrip": false,
|
| 208 |
+
"normalized": false,
|
| 209 |
+
"rstrip": false,
|
| 210 |
+
"single_word": false,
|
| 211 |
+
"special": true
|
| 212 |
+
},
|
| 213 |
+
"156917": {
|
| 214 |
+
"content": "<|reserved_token_22|>",
|
| 215 |
+
"lstrip": false,
|
| 216 |
+
"normalized": false,
|
| 217 |
+
"rstrip": false,
|
| 218 |
+
"single_word": false,
|
| 219 |
+
"special": true
|
| 220 |
+
},
|
| 221 |
+
"156918": {
|
| 222 |
+
"content": "<|reserved_token_23|>",
|
| 223 |
+
"lstrip": false,
|
| 224 |
+
"normalized": false,
|
| 225 |
+
"rstrip": false,
|
| 226 |
+
"single_word": false,
|
| 227 |
+
"special": true
|
| 228 |
+
},
|
| 229 |
+
"156919": {
|
| 230 |
+
"content": "<|reserved_token_24|>",
|
| 231 |
+
"lstrip": false,
|
| 232 |
+
"normalized": false,
|
| 233 |
+
"rstrip": false,
|
| 234 |
+
"single_word": false,
|
| 235 |
+
"special": true
|
| 236 |
+
},
|
| 237 |
+
"156920": {
|
| 238 |
+
"content": "<|reserved_token_25|>",
|
| 239 |
+
"lstrip": false,
|
| 240 |
+
"normalized": false,
|
| 241 |
+
"rstrip": false,
|
| 242 |
+
"single_word": false,
|
| 243 |
+
"special": true
|
| 244 |
+
},
|
| 245 |
+
"156921": {
|
| 246 |
+
"content": "<|reserved_token_26|>",
|
| 247 |
+
"lstrip": false,
|
| 248 |
+
"normalized": false,
|
| 249 |
+
"rstrip": false,
|
| 250 |
+
"single_word": false,
|
| 251 |
+
"special": true
|
| 252 |
+
},
|
| 253 |
+
"156922": {
|
| 254 |
+
"content": "<|reserved_token_27|>",
|
| 255 |
+
"lstrip": false,
|
| 256 |
+
"normalized": false,
|
| 257 |
+
"rstrip": false,
|
| 258 |
+
"single_word": false,
|
| 259 |
+
"special": true
|
| 260 |
+
},
|
| 261 |
+
"156923": {
|
| 262 |
+
"content": "<|reserved_token_28|>",
|
| 263 |
+
"lstrip": false,
|
| 264 |
+
"normalized": false,
|
| 265 |
+
"rstrip": false,
|
| 266 |
+
"single_word": false,
|
| 267 |
+
"special": true
|
| 268 |
+
},
|
| 269 |
+
"156924": {
|
| 270 |
+
"content": "<|reserved_token_29|>",
|
| 271 |
+
"lstrip": false,
|
| 272 |
+
"normalized": false,
|
| 273 |
+
"rstrip": false,
|
| 274 |
+
"single_word": false,
|
| 275 |
+
"special": true
|
| 276 |
+
},
|
| 277 |
+
"156925": {
|
| 278 |
+
"content": "<|reserved_token_30|>",
|
| 279 |
+
"lstrip": false,
|
| 280 |
+
"normalized": false,
|
| 281 |
+
"rstrip": false,
|
| 282 |
+
"single_word": false,
|
| 283 |
+
"special": true
|
| 284 |
+
},
|
| 285 |
+
"156926": {
|
| 286 |
+
"content": "<|reserved_token_31|>",
|
| 287 |
+
"lstrip": false,
|
| 288 |
+
"normalized": false,
|
| 289 |
+
"rstrip": false,
|
| 290 |
+
"single_word": false,
|
| 291 |
+
"special": true
|
| 292 |
+
},
|
| 293 |
+
"156927": {
|
| 294 |
+
"content": "<|reserved_token_32|>",
|
| 295 |
+
"lstrip": false,
|
| 296 |
+
"normalized": false,
|
| 297 |
+
"rstrip": false,
|
| 298 |
+
"single_word": false,
|
| 299 |
+
"special": true
|
| 300 |
+
},
|
| 301 |
+
"156928": {
|
| 302 |
+
"content": "<|reserved_token_33|>",
|
| 303 |
+
"lstrip": false,
|
| 304 |
+
"normalized": false,
|
| 305 |
+
"rstrip": false,
|
| 306 |
+
"single_word": false,
|
| 307 |
+
"special": true
|
| 308 |
+
},
|
| 309 |
+
"156929": {
|
| 310 |
+
"content": "<|reserved_token_34|>",
|
| 311 |
+
"lstrip": false,
|
| 312 |
+
"normalized": false,
|
| 313 |
+
"rstrip": false,
|
| 314 |
+
"single_word": false,
|
| 315 |
+
"special": true
|
| 316 |
+
},
|
| 317 |
+
"156930": {
|
| 318 |
+
"content": "<|reserved_token_35|>",
|
| 319 |
+
"lstrip": false,
|
| 320 |
+
"normalized": false,
|
| 321 |
+
"rstrip": false,
|
| 322 |
+
"single_word": false,
|
| 323 |
+
"special": true
|
| 324 |
+
},
|
| 325 |
+
"156931": {
|
| 326 |
+
"content": "<|reserved_token_36|>",
|
| 327 |
+
"lstrip": false,
|
| 328 |
+
"normalized": false,
|
| 329 |
+
"rstrip": false,
|
| 330 |
+
"single_word": false,
|
| 331 |
+
"special": true
|
| 332 |
+
},
|
| 333 |
+
"156932": {
|
| 334 |
+
"content": "<|reserved_token_37|>",
|
| 335 |
+
"lstrip": false,
|
| 336 |
+
"normalized": false,
|
| 337 |
+
"rstrip": false,
|
| 338 |
+
"single_word": false,
|
| 339 |
+
"special": true
|
| 340 |
+
},
|
| 341 |
+
"156933": {
|
| 342 |
+
"content": "<|reserved_token_38|>",
|
| 343 |
+
"lstrip": false,
|
| 344 |
+
"normalized": false,
|
| 345 |
+
"rstrip": false,
|
| 346 |
+
"single_word": false,
|
| 347 |
+
"special": true
|
| 348 |
+
},
|
| 349 |
+
"156934": {
|
| 350 |
+
"content": "<|reserved_token_39|>",
|
| 351 |
+
"lstrip": false,
|
| 352 |
+
"normalized": false,
|
| 353 |
+
"rstrip": false,
|
| 354 |
+
"single_word": false,
|
| 355 |
+
"special": true
|
| 356 |
+
},
|
| 357 |
+
"156935": {
|
| 358 |
+
"content": "<|reserved_token_40|>",
|
| 359 |
+
"lstrip": false,
|
| 360 |
+
"normalized": false,
|
| 361 |
+
"rstrip": false,
|
| 362 |
+
"single_word": false,
|
| 363 |
+
"special": true
|
| 364 |
+
},
|
| 365 |
+
"156936": {
|
| 366 |
+
"content": "<|reserved_token_41|>",
|
| 367 |
+
"lstrip": false,
|
| 368 |
+
"normalized": false,
|
| 369 |
+
"rstrip": false,
|
| 370 |
+
"single_word": false,
|
| 371 |
+
"special": true
|
| 372 |
+
},
|
| 373 |
+
"156937": {
|
| 374 |
+
"content": "<|reserved_token_42|>",
|
| 375 |
+
"lstrip": false,
|
| 376 |
+
"normalized": false,
|
| 377 |
+
"rstrip": false,
|
| 378 |
+
"single_word": false,
|
| 379 |
+
"special": true
|
| 380 |
+
},
|
| 381 |
+
"156938": {
|
| 382 |
+
"content": "<|reserved_token_43|>",
|
| 383 |
+
"lstrip": false,
|
| 384 |
+
"normalized": false,
|
| 385 |
+
"rstrip": false,
|
| 386 |
+
"single_word": false,
|
| 387 |
+
"special": true
|
| 388 |
+
},
|
| 389 |
+
"156939": {
|
| 390 |
+
"content": "<|reserved_token_44|>",
|
| 391 |
+
"lstrip": false,
|
| 392 |
+
"normalized": false,
|
| 393 |
+
"rstrip": false,
|
| 394 |
+
"single_word": false,
|
| 395 |
+
"special": true
|
| 396 |
+
},
|
| 397 |
+
"156940": {
|
| 398 |
+
"content": "<|reserved_token_45|>",
|
| 399 |
+
"lstrip": false,
|
| 400 |
+
"normalized": false,
|
| 401 |
+
"rstrip": false,
|
| 402 |
+
"single_word": false,
|
| 403 |
+
"special": true
|
| 404 |
+
},
|
| 405 |
+
"156941": {
|
| 406 |
+
"content": "<|reserved_token_46|>",
|
| 407 |
+
"lstrip": false,
|
| 408 |
+
"normalized": false,
|
| 409 |
+
"rstrip": false,
|
| 410 |
+
"single_word": false,
|
| 411 |
+
"special": true
|
| 412 |
+
},
|
| 413 |
+
"156942": {
|
| 414 |
+
"content": "<|reserved_token_47|>",
|
| 415 |
+
"lstrip": false,
|
| 416 |
+
"normalized": false,
|
| 417 |
+
"rstrip": false,
|
| 418 |
+
"single_word": false,
|
| 419 |
+
"special": true
|
| 420 |
+
},
|
| 421 |
+
"156943": {
|
| 422 |
+
"content": "<|reserved_token_48|>",
|
| 423 |
+
"lstrip": false,
|
| 424 |
+
"normalized": false,
|
| 425 |
+
"rstrip": false,
|
| 426 |
+
"single_word": false,
|
| 427 |
+
"special": true
|
| 428 |
+
},
|
| 429 |
+
"156944": {
|
| 430 |
+
"content": "<|reserved_token_49|>",
|
| 431 |
+
"lstrip": false,
|
| 432 |
+
"normalized": false,
|
| 433 |
+
"rstrip": false,
|
| 434 |
+
"single_word": false,
|
| 435 |
+
"special": true
|
| 436 |
+
},
|
| 437 |
+
"156945": {
|
| 438 |
+
"content": "<|reserved_token_50|>",
|
| 439 |
+
"lstrip": false,
|
| 440 |
+
"normalized": false,
|
| 441 |
+
"rstrip": false,
|
| 442 |
+
"single_word": false,
|
| 443 |
+
"special": true
|
| 444 |
+
},
|
| 445 |
+
"156946": {
|
| 446 |
+
"content": "<|reserved_token_51|>",
|
| 447 |
+
"lstrip": false,
|
| 448 |
+
"normalized": false,
|
| 449 |
+
"rstrip": false,
|
| 450 |
+
"single_word": false,
|
| 451 |
+
"special": true
|
| 452 |
+
},
|
| 453 |
+
"156947": {
|
| 454 |
+
"content": "<|reserved_token_52|>",
|
| 455 |
+
"lstrip": false,
|
| 456 |
+
"normalized": false,
|
| 457 |
+
"rstrip": false,
|
| 458 |
+
"single_word": false,
|
| 459 |
+
"special": true
|
| 460 |
+
},
|
| 461 |
+
"156948": {
|
| 462 |
+
"content": "<|reserved_token_53|>",
|
| 463 |
+
"lstrip": false,
|
| 464 |
+
"normalized": false,
|
| 465 |
+
"rstrip": false,
|
| 466 |
+
"single_word": false,
|
| 467 |
+
"special": true
|
| 468 |
+
},
|
| 469 |
+
"156949": {
|
| 470 |
+
"content": "<|reserved_token_54|>",
|
| 471 |
+
"lstrip": false,
|
| 472 |
+
"normalized": false,
|
| 473 |
+
"rstrip": false,
|
| 474 |
+
"single_word": false,
|
| 475 |
+
"special": true
|
| 476 |
+
},
|
| 477 |
+
"156950": {
|
| 478 |
+
"content": "<|reserved_token_55|>",
|
| 479 |
+
"lstrip": false,
|
| 480 |
+
"normalized": false,
|
| 481 |
+
"rstrip": false,
|
| 482 |
+
"single_word": false,
|
| 483 |
+
"special": true
|
| 484 |
+
},
|
| 485 |
+
"156951": {
|
| 486 |
+
"content": "<|reserved_token_56|>",
|
| 487 |
+
"lstrip": false,
|
| 488 |
+
"normalized": false,
|
| 489 |
+
"rstrip": false,
|
| 490 |
+
"single_word": false,
|
| 491 |
+
"special": true
|
| 492 |
+
},
|
| 493 |
+
"156952": {
|
| 494 |
+
"content": "<|reserved_token_57|>",
|
| 495 |
+
"lstrip": false,
|
| 496 |
+
"normalized": false,
|
| 497 |
+
"rstrip": false,
|
| 498 |
+
"single_word": false,
|
| 499 |
+
"special": true
|
| 500 |
+
},
|
| 501 |
+
"156953": {
|
| 502 |
+
"content": "<|reserved_token_58|>",
|
| 503 |
+
"lstrip": false,
|
| 504 |
+
"normalized": false,
|
| 505 |
+
"rstrip": false,
|
| 506 |
+
"single_word": false,
|
| 507 |
+
"special": true
|
| 508 |
+
},
|
| 509 |
+
"156954": {
|
| 510 |
+
"content": "<|reserved_token_59|>",
|
| 511 |
+
"lstrip": false,
|
| 512 |
+
"normalized": false,
|
| 513 |
+
"rstrip": false,
|
| 514 |
+
"single_word": false,
|
| 515 |
+
"special": true
|
| 516 |
+
},
|
| 517 |
+
"156955": {
|
| 518 |
+
"content": "<|reserved_token_60|>",
|
| 519 |
+
"lstrip": false,
|
| 520 |
+
"normalized": false,
|
| 521 |
+
"rstrip": false,
|
| 522 |
+
"single_word": false,
|
| 523 |
+
"special": true
|
| 524 |
+
},
|
| 525 |
+
"156956": {
|
| 526 |
+
"content": "<|reserved_token_61|>",
|
| 527 |
+
"lstrip": false,
|
| 528 |
+
"normalized": false,
|
| 529 |
+
"rstrip": false,
|
| 530 |
+
"single_word": false,
|
| 531 |
+
"special": true
|
| 532 |
+
},
|
| 533 |
+
"156957": {
|
| 534 |
+
"content": "<|reserved_token_62|>",
|
| 535 |
+
"lstrip": false,
|
| 536 |
+
"normalized": false,
|
| 537 |
+
"rstrip": false,
|
| 538 |
+
"single_word": false,
|
| 539 |
+
"special": true
|
| 540 |
+
},
|
| 541 |
+
"156958": {
|
| 542 |
+
"content": "<|reserved_token_63|>",
|
| 543 |
+
"lstrip": false,
|
| 544 |
+
"normalized": false,
|
| 545 |
+
"rstrip": false,
|
| 546 |
+
"single_word": false,
|
| 547 |
+
"special": true
|
| 548 |
+
},
|
| 549 |
+
"156959": {
|
| 550 |
+
"content": "<|reserved_token_64|>",
|
| 551 |
+
"lstrip": false,
|
| 552 |
+
"normalized": false,
|
| 553 |
+
"rstrip": false,
|
| 554 |
+
"single_word": false,
|
| 555 |
+
"special": true
|
| 556 |
+
},
|
| 557 |
+
"156960": {
|
| 558 |
+
"content": "<|reserved_token_65|>",
|
| 559 |
+
"lstrip": false,
|
| 560 |
+
"normalized": false,
|
| 561 |
+
"rstrip": false,
|
| 562 |
+
"single_word": false,
|
| 563 |
+
"special": true
|
| 564 |
+
},
|
| 565 |
+
"156961": {
|
| 566 |
+
"content": "<|reserved_token_66|>",
|
| 567 |
+
"lstrip": false,
|
| 568 |
+
"normalized": false,
|
| 569 |
+
"rstrip": false,
|
| 570 |
+
"single_word": false,
|
| 571 |
+
"special": true
|
| 572 |
+
},
|
| 573 |
+
"156962": {
|
| 574 |
+
"content": "<|reserved_token_67|>",
|
| 575 |
+
"lstrip": false,
|
| 576 |
+
"normalized": false,
|
| 577 |
+
"rstrip": false,
|
| 578 |
+
"single_word": false,
|
| 579 |
+
"special": true
|
| 580 |
+
},
|
| 581 |
+
"156963": {
|
| 582 |
+
"content": "<|reserved_token_68|>",
|
| 583 |
+
"lstrip": false,
|
| 584 |
+
"normalized": false,
|
| 585 |
+
"rstrip": false,
|
| 586 |
+
"single_word": false,
|
| 587 |
+
"special": true
|
| 588 |
+
},
|
| 589 |
+
"156964": {
|
| 590 |
+
"content": "<|reserved_token_69|>",
|
| 591 |
+
"lstrip": false,
|
| 592 |
+
"normalized": false,
|
| 593 |
+
"rstrip": false,
|
| 594 |
+
"single_word": false,
|
| 595 |
+
"special": true
|
| 596 |
+
},
|
| 597 |
+
"156965": {
|
| 598 |
+
"content": "<|reserved_token_70|>",
|
| 599 |
+
"lstrip": false,
|
| 600 |
+
"normalized": false,
|
| 601 |
+
"rstrip": false,
|
| 602 |
+
"single_word": false,
|
| 603 |
+
"special": true
|
| 604 |
+
},
|
| 605 |
+
"156966": {
|
| 606 |
+
"content": "<|reserved_token_71|>",
|
| 607 |
+
"lstrip": false,
|
| 608 |
+
"normalized": false,
|
| 609 |
+
"rstrip": false,
|
| 610 |
+
"single_word": false,
|
| 611 |
+
"special": true
|
| 612 |
+
},
|
| 613 |
+
"156967": {
|
| 614 |
+
"content": "<|reserved_token_72|>",
|
| 615 |
+
"lstrip": false,
|
| 616 |
+
"normalized": false,
|
| 617 |
+
"rstrip": false,
|
| 618 |
+
"single_word": false,
|
| 619 |
+
"special": true
|
| 620 |
+
},
|
| 621 |
+
"156968": {
|
| 622 |
+
"content": "<|reserved_token_73|>",
|
| 623 |
+
"lstrip": false,
|
| 624 |
+
"normalized": false,
|
| 625 |
+
"rstrip": false,
|
| 626 |
+
"single_word": false,
|
| 627 |
+
"special": true
|
| 628 |
+
},
|
| 629 |
+
"156969": {
|
| 630 |
+
"content": "<|reserved_token_74|>",
|
| 631 |
+
"lstrip": false,
|
| 632 |
+
"normalized": false,
|
| 633 |
+
"rstrip": false,
|
| 634 |
+
"single_word": false,
|
| 635 |
+
"special": true
|
| 636 |
+
},
|
| 637 |
+
"156970": {
|
| 638 |
+
"content": "<|reserved_token_75|>",
|
| 639 |
+
"lstrip": false,
|
| 640 |
+
"normalized": false,
|
| 641 |
+
"rstrip": false,
|
| 642 |
+
"single_word": false,
|
| 643 |
+
"special": true
|
| 644 |
+
},
|
| 645 |
+
"156971": {
|
| 646 |
+
"content": "<|reserved_token_76|>",
|
| 647 |
+
"lstrip": false,
|
| 648 |
+
"normalized": false,
|
| 649 |
+
"rstrip": false,
|
| 650 |
+
"single_word": false,
|
| 651 |
+
"special": true
|
| 652 |
+
},
|
| 653 |
+
"156972": {
|
| 654 |
+
"content": "<|reserved_token_77|>",
|
| 655 |
+
"lstrip": false,
|
| 656 |
+
"normalized": false,
|
| 657 |
+
"rstrip": false,
|
| 658 |
+
"single_word": false,
|
| 659 |
+
"special": true
|
| 660 |
+
},
|
| 661 |
+
"156973": {
|
| 662 |
+
"content": "<|reserved_token_78|>",
|
| 663 |
+
"lstrip": false,
|
| 664 |
+
"normalized": false,
|
| 665 |
+
"rstrip": false,
|
| 666 |
+
"single_word": false,
|
| 667 |
+
"special": true
|
| 668 |
+
},
|
| 669 |
+
"156974": {
|
| 670 |
+
"content": "<|reserved_token_79|>",
|
| 671 |
+
"lstrip": false,
|
| 672 |
+
"normalized": false,
|
| 673 |
+
"rstrip": false,
|
| 674 |
+
"single_word": false,
|
| 675 |
+
"special": true
|
| 676 |
+
},
|
| 677 |
+
"156975": {
|
| 678 |
+
"content": "<|reserved_token_80|>",
|
| 679 |
+
"lstrip": false,
|
| 680 |
+
"normalized": false,
|
| 681 |
+
"rstrip": false,
|
| 682 |
+
"single_word": false,
|
| 683 |
+
"special": true
|
| 684 |
+
},
|
| 685 |
+
"156976": {
|
| 686 |
+
"content": "<|reserved_token_81|>",
|
| 687 |
+
"lstrip": false,
|
| 688 |
+
"normalized": false,
|
| 689 |
+
"rstrip": false,
|
| 690 |
+
"single_word": false,
|
| 691 |
+
"special": true
|
| 692 |
+
},
|
| 693 |
+
"156977": {
|
| 694 |
+
"content": "<|reserved_token_82|>",
|
| 695 |
+
"lstrip": false,
|
| 696 |
+
"normalized": false,
|
| 697 |
+
"rstrip": false,
|
| 698 |
+
"single_word": false,
|
| 699 |
+
"special": true
|
| 700 |
+
},
|
| 701 |
+
"156978": {
|
| 702 |
+
"content": "<|reserved_token_83|>",
|
| 703 |
+
"lstrip": false,
|
| 704 |
+
"normalized": false,
|
| 705 |
+
"rstrip": false,
|
| 706 |
+
"single_word": false,
|
| 707 |
+
"special": true
|
| 708 |
+
},
|
| 709 |
+
"156979": {
|
| 710 |
+
"content": "<|reserved_token_84|>",
|
| 711 |
+
"lstrip": false,
|
| 712 |
+
"normalized": false,
|
| 713 |
+
"rstrip": false,
|
| 714 |
+
"single_word": false,
|
| 715 |
+
"special": true
|
| 716 |
+
},
|
| 717 |
+
"156980": {
|
| 718 |
+
"content": "<|reserved_token_85|>",
|
| 719 |
+
"lstrip": false,
|
| 720 |
+
"normalized": false,
|
| 721 |
+
"rstrip": false,
|
| 722 |
+
"single_word": false,
|
| 723 |
+
"special": true
|
| 724 |
+
},
|
| 725 |
+
"156981": {
|
| 726 |
+
"content": "<|reserved_token_86|>",
|
| 727 |
+
"lstrip": false,
|
| 728 |
+
"normalized": false,
|
| 729 |
+
"rstrip": false,
|
| 730 |
+
"single_word": false,
|
| 731 |
+
"special": true
|
| 732 |
+
},
|
| 733 |
+
"156982": {
|
| 734 |
+
"content": "<|reserved_token_87|>",
|
| 735 |
+
"lstrip": false,
|
| 736 |
+
"normalized": false,
|
| 737 |
+
"rstrip": false,
|
| 738 |
+
"single_word": false,
|
| 739 |
+
"special": true
|
| 740 |
+
},
|
| 741 |
+
"156983": {
|
| 742 |
+
"content": "<|reserved_token_88|>",
|
| 743 |
+
"lstrip": false,
|
| 744 |
+
"normalized": false,
|
| 745 |
+
"rstrip": false,
|
| 746 |
+
"single_word": false,
|
| 747 |
+
"special": true
|
| 748 |
+
},
|
| 749 |
+
"156984": {
|
| 750 |
+
"content": "<|reserved_token_89|>",
|
| 751 |
+
"lstrip": false,
|
| 752 |
+
"normalized": false,
|
| 753 |
+
"rstrip": false,
|
| 754 |
+
"single_word": false,
|
| 755 |
+
"special": true
|
| 756 |
+
},
|
| 757 |
+
"156985": {
|
| 758 |
+
"content": "<|reserved_token_90|>",
|
| 759 |
+
"lstrip": false,
|
| 760 |
+
"normalized": false,
|
| 761 |
+
"rstrip": false,
|
| 762 |
+
"single_word": false,
|
| 763 |
+
"special": true
|
| 764 |
+
},
|
| 765 |
+
"156986": {
|
| 766 |
+
"content": "<|reserved_token_91|>",
|
| 767 |
+
"lstrip": false,
|
| 768 |
+
"normalized": false,
|
| 769 |
+
"rstrip": false,
|
| 770 |
+
"single_word": false,
|
| 771 |
+
"special": true
|
| 772 |
+
},
|
| 773 |
+
"156987": {
|
| 774 |
+
"content": "<|reserved_token_92|>",
|
| 775 |
+
"lstrip": false,
|
| 776 |
+
"normalized": false,
|
| 777 |
+
"rstrip": false,
|
| 778 |
+
"single_word": false,
|
| 779 |
+
"special": true
|
| 780 |
+
},
|
| 781 |
+
"156988": {
|
| 782 |
+
"content": "<|reserved_token_93|>",
|
| 783 |
+
"lstrip": false,
|
| 784 |
+
"normalized": false,
|
| 785 |
+
"rstrip": false,
|
| 786 |
+
"single_word": false,
|
| 787 |
+
"special": true
|
| 788 |
+
},
|
| 789 |
+
"156989": {
|
| 790 |
+
"content": "<|reserved_token_94|>",
|
| 791 |
+
"lstrip": false,
|
| 792 |
+
"normalized": false,
|
| 793 |
+
"rstrip": false,
|
| 794 |
+
"single_word": false,
|
| 795 |
+
"special": true
|
| 796 |
+
},
|
| 797 |
+
"156990": {
|
| 798 |
+
"content": "<|reserved_token_95|>",
|
| 799 |
+
"lstrip": false,
|
| 800 |
+
"normalized": false,
|
| 801 |
+
"rstrip": false,
|
| 802 |
+
"single_word": false,
|
| 803 |
+
"special": true
|
| 804 |
+
},
|
| 805 |
+
"156991": {
|
| 806 |
+
"content": "<|reserved_token_96|>",
|
| 807 |
+
"lstrip": false,
|
| 808 |
+
"normalized": false,
|
| 809 |
+
"rstrip": false,
|
| 810 |
+
"single_word": false,
|
| 811 |
+
"special": true
|
| 812 |
+
},
|
| 813 |
+
"156992": {
|
| 814 |
+
"content": "<|reserved_token_97|>",
|
| 815 |
+
"lstrip": false,
|
| 816 |
+
"normalized": false,
|
| 817 |
+
"rstrip": false,
|
| 818 |
+
"single_word": false,
|
| 819 |
+
"special": true
|
| 820 |
+
},
|
| 821 |
+
"156993": {
|
| 822 |
+
"content": "<|reserved_token_98|>",
|
| 823 |
+
"lstrip": false,
|
| 824 |
+
"normalized": false,
|
| 825 |
+
"rstrip": false,
|
| 826 |
+
"single_word": false,
|
| 827 |
+
"special": true
|
| 828 |
+
},
|
| 829 |
+
"156994": {
|
| 830 |
+
"content": "<|reserved_token_99|>",
|
| 831 |
+
"lstrip": false,
|
| 832 |
+
"normalized": false,
|
| 833 |
+
"rstrip": false,
|
| 834 |
+
"single_word": false,
|
| 835 |
+
"special": true
|
| 836 |
+
},
|
| 837 |
+
"156995": {
|
| 838 |
+
"content": "<|reserved_token_100|>",
|
| 839 |
+
"lstrip": false,
|
| 840 |
+
"normalized": false,
|
| 841 |
+
"rstrip": false,
|
| 842 |
+
"single_word": false,
|
| 843 |
+
"special": true
|
| 844 |
+
},
|
| 845 |
+
"156996": {
|
| 846 |
+
"content": "<|reserved_token_101|>",
|
| 847 |
+
"lstrip": false,
|
| 848 |
+
"normalized": false,
|
| 849 |
+
"rstrip": false,
|
| 850 |
+
"single_word": false,
|
| 851 |
+
"special": true
|
| 852 |
+
},
|
| 853 |
+
"156997": {
|
| 854 |
+
"content": "<|reserved_token_102|>",
|
| 855 |
+
"lstrip": false,
|
| 856 |
+
"normalized": false,
|
| 857 |
+
"rstrip": false,
|
| 858 |
+
"single_word": false,
|
| 859 |
+
"special": true
|
| 860 |
+
},
|
| 861 |
+
"156998": {
|
| 862 |
+
"content": "<|reserved_token_103|>",
|
| 863 |
+
"lstrip": false,
|
| 864 |
+
"normalized": false,
|
| 865 |
+
"rstrip": false,
|
| 866 |
+
"single_word": false,
|
| 867 |
+
"special": true
|
| 868 |
+
},
|
| 869 |
+
"156999": {
|
| 870 |
+
"content": "<|reserved_token_104|>",
|
| 871 |
+
"lstrip": false,
|
| 872 |
+
"normalized": false,
|
| 873 |
+
"rstrip": false,
|
| 874 |
+
"single_word": false,
|
| 875 |
+
"special": true
|
| 876 |
+
},
|
| 877 |
+
"157000": {
|
| 878 |
+
"content": "<|reserved_token_105|>",
|
| 879 |
+
"lstrip": false,
|
| 880 |
+
"normalized": false,
|
| 881 |
+
"rstrip": false,
|
| 882 |
+
"single_word": false,
|
| 883 |
+
"special": true
|
| 884 |
+
},
|
| 885 |
+
"157001": {
|
| 886 |
+
"content": "<|reserved_token_106|>",
|
| 887 |
+
"lstrip": false,
|
| 888 |
+
"normalized": false,
|
| 889 |
+
"rstrip": false,
|
| 890 |
+
"single_word": false,
|
| 891 |
+
"special": true
|
| 892 |
+
},
|
| 893 |
+
"157002": {
|
| 894 |
+
"content": "<|reserved_token_107|>",
|
| 895 |
+
"lstrip": false,
|
| 896 |
+
"normalized": false,
|
| 897 |
+
"rstrip": false,
|
| 898 |
+
"single_word": false,
|
| 899 |
+
"special": true
|
| 900 |
+
},
|
| 901 |
+
"157003": {
|
| 902 |
+
"content": "<|reserved_token_108|>",
|
| 903 |
+
"lstrip": false,
|
| 904 |
+
"normalized": false,
|
| 905 |
+
"rstrip": false,
|
| 906 |
+
"single_word": false,
|
| 907 |
+
"special": true
|
| 908 |
+
},
|
| 909 |
+
"157004": {
|
| 910 |
+
"content": "<|reserved_token_109|>",
|
| 911 |
+
"lstrip": false,
|
| 912 |
+
"normalized": false,
|
| 913 |
+
"rstrip": false,
|
| 914 |
+
"single_word": false,
|
| 915 |
+
"special": true
|
| 916 |
+
},
|
| 917 |
+
"157005": {
|
| 918 |
+
"content": "<|reserved_token_110|>",
|
| 919 |
+
"lstrip": false,
|
| 920 |
+
"normalized": false,
|
| 921 |
+
"rstrip": false,
|
| 922 |
+
"single_word": false,
|
| 923 |
+
"special": true
|
| 924 |
+
},
|
| 925 |
+
"157006": {
|
| 926 |
+
"content": "<|reserved_token_111|>",
|
| 927 |
+
"lstrip": false,
|
| 928 |
+
"normalized": false,
|
| 929 |
+
"rstrip": false,
|
| 930 |
+
"single_word": false,
|
| 931 |
+
"special": true
|
| 932 |
+
},
|
| 933 |
+
"157007": {
|
| 934 |
+
"content": "<|reserved_token_112|>",
|
| 935 |
+
"lstrip": false,
|
| 936 |
+
"normalized": false,
|
| 937 |
+
"rstrip": false,
|
| 938 |
+
"single_word": false,
|
| 939 |
+
"special": true
|
| 940 |
+
},
|
| 941 |
+
"157008": {
|
| 942 |
+
"content": "<|reserved_token_113|>",
|
| 943 |
+
"lstrip": false,
|
| 944 |
+
"normalized": false,
|
| 945 |
+
"rstrip": false,
|
| 946 |
+
"single_word": false,
|
| 947 |
+
"special": true
|
| 948 |
+
},
|
| 949 |
+
"157009": {
|
| 950 |
+
"content": "<|reserved_token_114|>",
|
| 951 |
+
"lstrip": false,
|
| 952 |
+
"normalized": false,
|
| 953 |
+
"rstrip": false,
|
| 954 |
+
"single_word": false,
|
| 955 |
+
"special": true
|
| 956 |
+
},
|
| 957 |
+
"157010": {
|
| 958 |
+
"content": "<|reserved_token_115|>",
|
| 959 |
+
"lstrip": false,
|
| 960 |
+
"normalized": false,
|
| 961 |
+
"rstrip": false,
|
| 962 |
+
"single_word": false,
|
| 963 |
+
"special": true
|
| 964 |
+
},
|
| 965 |
+
"157011": {
|
| 966 |
+
"content": "<|reserved_token_116|>",
|
| 967 |
+
"lstrip": false,
|
| 968 |
+
"normalized": false,
|
| 969 |
+
"rstrip": false,
|
| 970 |
+
"single_word": false,
|
| 971 |
+
"special": true
|
| 972 |
+
},
|
| 973 |
+
"157012": {
|
| 974 |
+
"content": "<|reserved_token_117|>",
|
| 975 |
+
"lstrip": false,
|
| 976 |
+
"normalized": false,
|
| 977 |
+
"rstrip": false,
|
| 978 |
+
"single_word": false,
|
| 979 |
+
"special": true
|
| 980 |
+
},
|
| 981 |
+
"157013": {
|
| 982 |
+
"content": "<|reserved_token_118|>",
|
| 983 |
+
"lstrip": false,
|
| 984 |
+
"normalized": false,
|
| 985 |
+
"rstrip": false,
|
| 986 |
+
"single_word": false,
|
| 987 |
+
"special": true
|
| 988 |
+
},
|
| 989 |
+
"157014": {
|
| 990 |
+
"content": "<|reserved_token_119|>",
|
| 991 |
+
"lstrip": false,
|
| 992 |
+
"normalized": false,
|
| 993 |
+
"rstrip": false,
|
| 994 |
+
"single_word": false,
|
| 995 |
+
"special": true
|
| 996 |
+
},
|
| 997 |
+
"157015": {
|
| 998 |
+
"content": "<|reserved_token_120|>",
|
| 999 |
+
"lstrip": false,
|
| 1000 |
+
"normalized": false,
|
| 1001 |
+
"rstrip": false,
|
| 1002 |
+
"single_word": false,
|
| 1003 |
+
"special": true
|
| 1004 |
+
},
|
| 1005 |
+
"157016": {
|
| 1006 |
+
"content": "<|reserved_token_121|>",
|
| 1007 |
+
"lstrip": false,
|
| 1008 |
+
"normalized": false,
|
| 1009 |
+
"rstrip": false,
|
| 1010 |
+
"single_word": false,
|
| 1011 |
+
"special": true
|
| 1012 |
+
},
|
| 1013 |
+
"157017": {
|
| 1014 |
+
"content": "<|reserved_token_122|>",
|
| 1015 |
+
"lstrip": false,
|
| 1016 |
+
"normalized": false,
|
| 1017 |
+
"rstrip": false,
|
| 1018 |
+
"single_word": false,
|
| 1019 |
+
"special": true
|
| 1020 |
+
},
|
| 1021 |
+
"157018": {
|
| 1022 |
+
"content": "<|reserved_token_123|>",
|
| 1023 |
+
"lstrip": false,
|
| 1024 |
+
"normalized": false,
|
| 1025 |
+
"rstrip": false,
|
| 1026 |
+
"single_word": false,
|
| 1027 |
+
"special": true
|
| 1028 |
+
},
|
| 1029 |
+
"157019": {
|
| 1030 |
+
"content": "<|reserved_token_124|>",
|
| 1031 |
+
"lstrip": false,
|
| 1032 |
+
"normalized": false,
|
| 1033 |
+
"rstrip": false,
|
| 1034 |
+
"single_word": false,
|
| 1035 |
+
"special": true
|
| 1036 |
+
},
|
| 1037 |
+
"157020": {
|
| 1038 |
+
"content": "<|reserved_token_125|>",
|
| 1039 |
+
"lstrip": false,
|
| 1040 |
+
"normalized": false,
|
| 1041 |
+
"rstrip": false,
|
| 1042 |
+
"single_word": false,
|
| 1043 |
+
"special": true
|
| 1044 |
+
},
|
| 1045 |
+
"157021": {
|
| 1046 |
+
"content": "<|reserved_token_126|>",
|
| 1047 |
+
"lstrip": false,
|
| 1048 |
+
"normalized": false,
|
| 1049 |
+
"rstrip": false,
|
| 1050 |
+
"single_word": false,
|
| 1051 |
+
"special": true
|
| 1052 |
+
},
|
| 1053 |
+
"157022": {
|
| 1054 |
+
"content": "<|reserved_token_127|>",
|
| 1055 |
+
"lstrip": false,
|
| 1056 |
+
"normalized": false,
|
| 1057 |
+
"rstrip": false,
|
| 1058 |
+
"single_word": false,
|
| 1059 |
+
"special": true
|
| 1060 |
+
},
|
| 1061 |
+
"157023": {
|
| 1062 |
+
"content": "<|reserved_token_128|>",
|
| 1063 |
+
"lstrip": false,
|
| 1064 |
+
"normalized": false,
|
| 1065 |
+
"rstrip": false,
|
| 1066 |
+
"single_word": false,
|
| 1067 |
+
"special": true
|
| 1068 |
+
},
|
| 1069 |
+
"157024": {
|
| 1070 |
+
"content": "<|reserved_token_129|>",
|
| 1071 |
+
"lstrip": false,
|
| 1072 |
+
"normalized": false,
|
| 1073 |
+
"rstrip": false,
|
| 1074 |
+
"single_word": false,
|
| 1075 |
+
"special": true
|
| 1076 |
+
},
|
| 1077 |
+
"157025": {
|
| 1078 |
+
"content": "<|reserved_token_130|>",
|
| 1079 |
+
"lstrip": false,
|
| 1080 |
+
"normalized": false,
|
| 1081 |
+
"rstrip": false,
|
| 1082 |
+
"single_word": false,
|
| 1083 |
+
"special": true
|
| 1084 |
+
},
|
| 1085 |
+
"157026": {
|
| 1086 |
+
"content": "<|reserved_token_131|>",
|
| 1087 |
+
"lstrip": false,
|
| 1088 |
+
"normalized": false,
|
| 1089 |
+
"rstrip": false,
|
| 1090 |
+
"single_word": false,
|
| 1091 |
+
"special": true
|
| 1092 |
+
},
|
| 1093 |
+
"157027": {
|
| 1094 |
+
"content": "<|reserved_token_132|>",
|
| 1095 |
+
"lstrip": false,
|
| 1096 |
+
"normalized": false,
|
| 1097 |
+
"rstrip": false,
|
| 1098 |
+
"single_word": false,
|
| 1099 |
+
"special": true
|
| 1100 |
+
},
|
| 1101 |
+
"157028": {
|
| 1102 |
+
"content": "<|reserved_token_133|>",
|
| 1103 |
+
"lstrip": false,
|
| 1104 |
+
"normalized": false,
|
| 1105 |
+
"rstrip": false,
|
| 1106 |
+
"single_word": false,
|
| 1107 |
+
"special": true
|
| 1108 |
+
},
|
| 1109 |
+
"157029": {
|
| 1110 |
+
"content": "<|reserved_token_134|>",
|
| 1111 |
+
"lstrip": false,
|
| 1112 |
+
"normalized": false,
|
| 1113 |
+
"rstrip": false,
|
| 1114 |
+
"single_word": false,
|
| 1115 |
+
"special": true
|
| 1116 |
+
},
|
| 1117 |
+
"157030": {
|
| 1118 |
+
"content": "<|reserved_token_135|>",
|
| 1119 |
+
"lstrip": false,
|
| 1120 |
+
"normalized": false,
|
| 1121 |
+
"rstrip": false,
|
| 1122 |
+
"single_word": false,
|
| 1123 |
+
"special": true
|
| 1124 |
+
},
|
| 1125 |
+
"157031": {
|
| 1126 |
+
"content": "<|reserved_token_136|>",
|
| 1127 |
+
"lstrip": false,
|
| 1128 |
+
"normalized": false,
|
| 1129 |
+
"rstrip": false,
|
| 1130 |
+
"single_word": false,
|
| 1131 |
+
"special": true
|
| 1132 |
+
},
|
| 1133 |
+
"157032": {
|
| 1134 |
+
"content": "<|reserved_token_137|>",
|
| 1135 |
+
"lstrip": false,
|
| 1136 |
+
"normalized": false,
|
| 1137 |
+
"rstrip": false,
|
| 1138 |
+
"single_word": false,
|
| 1139 |
+
"special": true
|
| 1140 |
+
},
|
| 1141 |
+
"157033": {
|
| 1142 |
+
"content": "<|reserved_token_138|>",
|
| 1143 |
+
"lstrip": false,
|
| 1144 |
+
"normalized": false,
|
| 1145 |
+
"rstrip": false,
|
| 1146 |
+
"single_word": false,
|
| 1147 |
+
"special": true
|
| 1148 |
+
},
|
| 1149 |
+
"157034": {
|
| 1150 |
+
"content": "<|reserved_token_139|>",
|
| 1151 |
+
"lstrip": false,
|
| 1152 |
+
"normalized": false,
|
| 1153 |
+
"rstrip": false,
|
| 1154 |
+
"single_word": false,
|
| 1155 |
+
"special": true
|
| 1156 |
+
},
|
| 1157 |
+
"157035": {
|
| 1158 |
+
"content": "<|reserved_token_140|>",
|
| 1159 |
+
"lstrip": false,
|
| 1160 |
+
"normalized": false,
|
| 1161 |
+
"rstrip": false,
|
| 1162 |
+
"single_word": false,
|
| 1163 |
+
"special": true
|
| 1164 |
+
},
|
| 1165 |
+
"157036": {
|
| 1166 |
+
"content": "<|reserved_token_141|>",
|
| 1167 |
+
"lstrip": false,
|
| 1168 |
+
"normalized": false,
|
| 1169 |
+
"rstrip": false,
|
| 1170 |
+
"single_word": false,
|
| 1171 |
+
"special": true
|
| 1172 |
+
},
|
| 1173 |
+
"157037": {
|
| 1174 |
+
"content": "<|reserved_token_142|>",
|
| 1175 |
+
"lstrip": false,
|
| 1176 |
+
"normalized": false,
|
| 1177 |
+
"rstrip": false,
|
| 1178 |
+
"single_word": false,
|
| 1179 |
+
"special": true
|
| 1180 |
+
},
|
| 1181 |
+
"157038": {
|
| 1182 |
+
"content": "<|reserved_token_143|>",
|
| 1183 |
+
"lstrip": false,
|
| 1184 |
+
"normalized": false,
|
| 1185 |
+
"rstrip": false,
|
| 1186 |
+
"single_word": false,
|
| 1187 |
+
"special": true
|
| 1188 |
+
},
|
| 1189 |
+
"157039": {
|
| 1190 |
+
"content": "<|reserved_token_144|>",
|
| 1191 |
+
"lstrip": false,
|
| 1192 |
+
"normalized": false,
|
| 1193 |
+
"rstrip": false,
|
| 1194 |
+
"single_word": false,
|
| 1195 |
+
"special": true
|
| 1196 |
+
},
|
| 1197 |
+
"157040": {
|
| 1198 |
+
"content": "<|reserved_token_145|>",
|
| 1199 |
+
"lstrip": false,
|
| 1200 |
+
"normalized": false,
|
| 1201 |
+
"rstrip": false,
|
| 1202 |
+
"single_word": false,
|
| 1203 |
+
"special": true
|
| 1204 |
+
},
|
| 1205 |
+
"157041": {
|
| 1206 |
+
"content": "<|reserved_token_146|>",
|
| 1207 |
+
"lstrip": false,
|
| 1208 |
+
"normalized": false,
|
| 1209 |
+
"rstrip": false,
|
| 1210 |
+
"single_word": false,
|
| 1211 |
+
"special": true
|
| 1212 |
+
},
|
| 1213 |
+
"157042": {
|
| 1214 |
+
"content": "<|reserved_token_147|>",
|
| 1215 |
+
"lstrip": false,
|
| 1216 |
+
"normalized": false,
|
| 1217 |
+
"rstrip": false,
|
| 1218 |
+
"single_word": false,
|
| 1219 |
+
"special": true
|
| 1220 |
+
},
|
| 1221 |
+
"157043": {
|
| 1222 |
+
"content": "<|reserved_token_148|>",
|
| 1223 |
+
"lstrip": false,
|
| 1224 |
+
"normalized": false,
|
| 1225 |
+
"rstrip": false,
|
| 1226 |
+
"single_word": false,
|
| 1227 |
+
"special": true
|
| 1228 |
+
},
|
| 1229 |
+
"157044": {
|
| 1230 |
+
"content": "<|reserved_token_149|>",
|
| 1231 |
+
"lstrip": false,
|
| 1232 |
+
"normalized": false,
|
| 1233 |
+
"rstrip": false,
|
| 1234 |
+
"single_word": false,
|
| 1235 |
+
"special": true
|
| 1236 |
+
},
|
| 1237 |
+
"157045": {
|
| 1238 |
+
"content": "<|reserved_token_150|>",
|
| 1239 |
+
"lstrip": false,
|
| 1240 |
+
"normalized": false,
|
| 1241 |
+
"rstrip": false,
|
| 1242 |
+
"single_word": false,
|
| 1243 |
+
"special": true
|
| 1244 |
+
},
|
| 1245 |
+
"157046": {
|
| 1246 |
+
"content": "<|reserved_token_151|>",
|
| 1247 |
+
"lstrip": false,
|
| 1248 |
+
"normalized": false,
|
| 1249 |
+
"rstrip": false,
|
| 1250 |
+
"single_word": false,
|
| 1251 |
+
"special": true
|
| 1252 |
+
},
|
| 1253 |
+
"157047": {
|
| 1254 |
+
"content": "<|reserved_token_152|>",
|
| 1255 |
+
"lstrip": false,
|
| 1256 |
+
"normalized": false,
|
| 1257 |
+
"rstrip": false,
|
| 1258 |
+
"single_word": false,
|
| 1259 |
+
"special": true
|
| 1260 |
+
},
|
| 1261 |
+
"157048": {
|
| 1262 |
+
"content": "<|reserved_token_153|>",
|
| 1263 |
+
"lstrip": false,
|
| 1264 |
+
"normalized": false,
|
| 1265 |
+
"rstrip": false,
|
| 1266 |
+
"single_word": false,
|
| 1267 |
+
"special": true
|
| 1268 |
+
},
|
| 1269 |
+
"157049": {
|
| 1270 |
+
"content": "<|reserved_token_154|>",
|
| 1271 |
+
"lstrip": false,
|
| 1272 |
+
"normalized": false,
|
| 1273 |
+
"rstrip": false,
|
| 1274 |
+
"single_word": false,
|
| 1275 |
+
"special": true
|
| 1276 |
+
},
|
| 1277 |
+
"157050": {
|
| 1278 |
+
"content": "<|reserved_token_155|>",
|
| 1279 |
+
"lstrip": false,
|
| 1280 |
+
"normalized": false,
|
| 1281 |
+
"rstrip": false,
|
| 1282 |
+
"single_word": false,
|
| 1283 |
+
"special": true
|
| 1284 |
+
},
|
| 1285 |
+
"157051": {
|
| 1286 |
+
"content": "<|reserved_token_156|>",
|
| 1287 |
+
"lstrip": false,
|
| 1288 |
+
"normalized": false,
|
| 1289 |
+
"rstrip": false,
|
| 1290 |
+
"single_word": false,
|
| 1291 |
+
"special": true
|
| 1292 |
+
},
|
| 1293 |
+
"157052": {
|
| 1294 |
+
"content": "<|reserved_token_157|>",
|
| 1295 |
+
"lstrip": false,
|
| 1296 |
+
"normalized": false,
|
| 1297 |
+
"rstrip": false,
|
| 1298 |
+
"single_word": false,
|
| 1299 |
+
"special": true
|
| 1300 |
+
},
|
| 1301 |
+
"157053": {
|
| 1302 |
+
"content": "<|reserved_token_158|>",
|
| 1303 |
+
"lstrip": false,
|
| 1304 |
+
"normalized": false,
|
| 1305 |
+
"rstrip": false,
|
| 1306 |
+
"single_word": false,
|
| 1307 |
+
"special": true
|
| 1308 |
+
},
|
| 1309 |
+
"157054": {
|
| 1310 |
+
"content": "<|reserved_token_159|>",
|
| 1311 |
+
"lstrip": false,
|
| 1312 |
+
"normalized": false,
|
| 1313 |
+
"rstrip": false,
|
| 1314 |
+
"single_word": false,
|
| 1315 |
+
"special": true
|
| 1316 |
+
},
|
| 1317 |
+
"157055": {
|
| 1318 |
+
"content": "<|reserved_token_160|>",
|
| 1319 |
+
"lstrip": false,
|
| 1320 |
+
"normalized": false,
|
| 1321 |
+
"rstrip": false,
|
| 1322 |
+
"single_word": false,
|
| 1323 |
+
"special": true
|
| 1324 |
+
},
|
| 1325 |
+
"157056": {
|
| 1326 |
+
"content": "<|reserved_token_161|>",
|
| 1327 |
+
"lstrip": false,
|
| 1328 |
+
"normalized": false,
|
| 1329 |
+
"rstrip": false,
|
| 1330 |
+
"single_word": false,
|
| 1331 |
+
"special": true
|
| 1332 |
+
},
|
| 1333 |
+
"157057": {
|
| 1334 |
+
"content": "<|reserved_token_162|>",
|
| 1335 |
+
"lstrip": false,
|
| 1336 |
+
"normalized": false,
|
| 1337 |
+
"rstrip": false,
|
| 1338 |
+
"single_word": false,
|
| 1339 |
+
"special": true
|
| 1340 |
+
},
|
| 1341 |
+
"157058": {
|
| 1342 |
+
"content": "<|reserved_token_163|>",
|
| 1343 |
+
"lstrip": false,
|
| 1344 |
+
"normalized": false,
|
| 1345 |
+
"rstrip": false,
|
| 1346 |
+
"single_word": false,
|
| 1347 |
+
"special": true
|
| 1348 |
+
},
|
| 1349 |
+
"157059": {
|
| 1350 |
+
"content": "<|reserved_token_164|>",
|
| 1351 |
+
"lstrip": false,
|
| 1352 |
+
"normalized": false,
|
| 1353 |
+
"rstrip": false,
|
| 1354 |
+
"single_word": false,
|
| 1355 |
+
"special": true
|
| 1356 |
+
},
|
| 1357 |
+
"157060": {
|
| 1358 |
+
"content": "<|reserved_token_165|>",
|
| 1359 |
+
"lstrip": false,
|
| 1360 |
+
"normalized": false,
|
| 1361 |
+
"rstrip": false,
|
| 1362 |
+
"single_word": false,
|
| 1363 |
+
"special": true
|
| 1364 |
+
},
|
| 1365 |
+
"157061": {
|
| 1366 |
+
"content": "<|reserved_token_166|>",
|
| 1367 |
+
"lstrip": false,
|
| 1368 |
+
"normalized": false,
|
| 1369 |
+
"rstrip": false,
|
| 1370 |
+
"single_word": false,
|
| 1371 |
+
"special": true
|
| 1372 |
+
},
|
| 1373 |
+
"157062": {
|
| 1374 |
+
"content": "<|reserved_token_167|>",
|
| 1375 |
+
"lstrip": false,
|
| 1376 |
+
"normalized": false,
|
| 1377 |
+
"rstrip": false,
|
| 1378 |
+
"single_word": false,
|
| 1379 |
+
"special": true
|
| 1380 |
+
},
|
| 1381 |
+
"157063": {
|
| 1382 |
+
"content": "<|reserved_token_168|>",
|
| 1383 |
+
"lstrip": false,
|
| 1384 |
+
"normalized": false,
|
| 1385 |
+
"rstrip": false,
|
| 1386 |
+
"single_word": false,
|
| 1387 |
+
"special": true
|
| 1388 |
+
},
|
| 1389 |
+
"157064": {
|
| 1390 |
+
"content": "<|reserved_token_169|>",
|
| 1391 |
+
"lstrip": false,
|
| 1392 |
+
"normalized": false,
|
| 1393 |
+
"rstrip": false,
|
| 1394 |
+
"single_word": false,
|
| 1395 |
+
"special": true
|
| 1396 |
+
},
|
| 1397 |
+
"157065": {
|
| 1398 |
+
"content": "<|reserved_token_170|>",
|
| 1399 |
+
"lstrip": false,
|
| 1400 |
+
"normalized": false,
|
| 1401 |
+
"rstrip": false,
|
| 1402 |
+
"single_word": false,
|
| 1403 |
+
"special": true
|
| 1404 |
+
},
|
| 1405 |
+
"157066": {
|
| 1406 |
+
"content": "<|reserved_token_171|>",
|
| 1407 |
+
"lstrip": false,
|
| 1408 |
+
"normalized": false,
|
| 1409 |
+
"rstrip": false,
|
| 1410 |
+
"single_word": false,
|
| 1411 |
+
"special": true
|
| 1412 |
+
},
|
| 1413 |
+
"157067": {
|
| 1414 |
+
"content": "<|reserved_token_172|>",
|
| 1415 |
+
"lstrip": false,
|
| 1416 |
+
"normalized": false,
|
| 1417 |
+
"rstrip": false,
|
| 1418 |
+
"single_word": false,
|
| 1419 |
+
"special": true
|
| 1420 |
+
},
|
| 1421 |
+
"157068": {
|
| 1422 |
+
"content": "<|reserved_token_173|>",
|
| 1423 |
+
"lstrip": false,
|
| 1424 |
+
"normalized": false,
|
| 1425 |
+
"rstrip": false,
|
| 1426 |
+
"single_word": false,
|
| 1427 |
+
"special": true
|
| 1428 |
+
},
|
| 1429 |
+
"157069": {
|
| 1430 |
+
"content": "<|reserved_token_174|>",
|
| 1431 |
+
"lstrip": false,
|
| 1432 |
+
"normalized": false,
|
| 1433 |
+
"rstrip": false,
|
| 1434 |
+
"single_word": false,
|
| 1435 |
+
"special": true
|
| 1436 |
+
},
|
| 1437 |
+
"157070": {
|
| 1438 |
+
"content": "<|reserved_token_175|>",
|
| 1439 |
+
"lstrip": false,
|
| 1440 |
+
"normalized": false,
|
| 1441 |
+
"rstrip": false,
|
| 1442 |
+
"single_word": false,
|
| 1443 |
+
"special": true
|
| 1444 |
+
},
|
| 1445 |
+
"157071": {
|
| 1446 |
+
"content": "<|reserved_token_176|>",
|
| 1447 |
+
"lstrip": false,
|
| 1448 |
+
"normalized": false,
|
| 1449 |
+
"rstrip": false,
|
| 1450 |
+
"single_word": false,
|
| 1451 |
+
"special": true
|
| 1452 |
+
},
|
| 1453 |
+
"157072": {
|
| 1454 |
+
"content": "<|reserved_token_177|>",
|
| 1455 |
+
"lstrip": false,
|
| 1456 |
+
"normalized": false,
|
| 1457 |
+
"rstrip": false,
|
| 1458 |
+
"single_word": false,
|
| 1459 |
+
"special": true
|
| 1460 |
+
},
|
| 1461 |
+
"157073": {
|
| 1462 |
+
"content": "<|reserved_token_178|>",
|
| 1463 |
+
"lstrip": false,
|
| 1464 |
+
"normalized": false,
|
| 1465 |
+
"rstrip": false,
|
| 1466 |
+
"single_word": false,
|
| 1467 |
+
"special": true
|
| 1468 |
+
},
|
| 1469 |
+
"157074": {
|
| 1470 |
+
"content": "<|reserved_token_179|>",
|
| 1471 |
+
"lstrip": false,
|
| 1472 |
+
"normalized": false,
|
| 1473 |
+
"rstrip": false,
|
| 1474 |
+
"single_word": false,
|
| 1475 |
+
"special": true
|
| 1476 |
+
},
|
| 1477 |
+
"157075": {
|
| 1478 |
+
"content": "<|reserved_token_180|>",
|
| 1479 |
+
"lstrip": false,
|
| 1480 |
+
"normalized": false,
|
| 1481 |
+
"rstrip": false,
|
| 1482 |
+
"single_word": false,
|
| 1483 |
+
"special": true
|
| 1484 |
+
},
|
| 1485 |
+
"157076": {
|
| 1486 |
+
"content": "<|reserved_token_181|>",
|
| 1487 |
+
"lstrip": false,
|
| 1488 |
+
"normalized": false,
|
| 1489 |
+
"rstrip": false,
|
| 1490 |
+
"single_word": false,
|
| 1491 |
+
"special": true
|
| 1492 |
+
},
|
| 1493 |
+
"157077": {
|
| 1494 |
+
"content": "<|reserved_token_182|>",
|
| 1495 |
+
"lstrip": false,
|
| 1496 |
+
"normalized": false,
|
| 1497 |
+
"rstrip": false,
|
| 1498 |
+
"single_word": false,
|
| 1499 |
+
"special": true
|
| 1500 |
+
},
|
| 1501 |
+
"157078": {
|
| 1502 |
+
"content": "<|reserved_token_183|>",
|
| 1503 |
+
"lstrip": false,
|
| 1504 |
+
"normalized": false,
|
| 1505 |
+
"rstrip": false,
|
| 1506 |
+
"single_word": false,
|
| 1507 |
+
"special": true
|
| 1508 |
+
},
|
| 1509 |
+
"157079": {
|
| 1510 |
+
"content": "<|reserved_token_184|>",
|
| 1511 |
+
"lstrip": false,
|
| 1512 |
+
"normalized": false,
|
| 1513 |
+
"rstrip": false,
|
| 1514 |
+
"single_word": false,
|
| 1515 |
+
"special": true
|
| 1516 |
+
},
|
| 1517 |
+
"157080": {
|
| 1518 |
+
"content": "<|reserved_token_185|>",
|
| 1519 |
+
"lstrip": false,
|
| 1520 |
+
"normalized": false,
|
| 1521 |
+
"rstrip": false,
|
| 1522 |
+
"single_word": false,
|
| 1523 |
+
"special": true
|
| 1524 |
+
},
|
| 1525 |
+
"157081": {
|
| 1526 |
+
"content": "<|reserved_token_186|>",
|
| 1527 |
+
"lstrip": false,
|
| 1528 |
+
"normalized": false,
|
| 1529 |
+
"rstrip": false,
|
| 1530 |
+
"single_word": false,
|
| 1531 |
+
"special": true
|
| 1532 |
+
},
|
| 1533 |
+
"157082": {
|
| 1534 |
+
"content": "<|reserved_token_187|>",
|
| 1535 |
+
"lstrip": false,
|
| 1536 |
+
"normalized": false,
|
| 1537 |
+
"rstrip": false,
|
| 1538 |
+
"single_word": false,
|
| 1539 |
+
"special": true
|
| 1540 |
+
},
|
| 1541 |
+
"157083": {
|
| 1542 |
+
"content": "<|reserved_token_188|>",
|
| 1543 |
+
"lstrip": false,
|
| 1544 |
+
"normalized": false,
|
| 1545 |
+
"rstrip": false,
|
| 1546 |
+
"single_word": false,
|
| 1547 |
+
"special": true
|
| 1548 |
+
},
|
| 1549 |
+
"157084": {
|
| 1550 |
+
"content": "<|reserved_token_189|>",
|
| 1551 |
+
"lstrip": false,
|
| 1552 |
+
"normalized": false,
|
| 1553 |
+
"rstrip": false,
|
| 1554 |
+
"single_word": false,
|
| 1555 |
+
"special": true
|
| 1556 |
+
},
|
| 1557 |
+
"157085": {
|
| 1558 |
+
"content": "<|reserved_token_190|>",
|
| 1559 |
+
"lstrip": false,
|
| 1560 |
+
"normalized": false,
|
| 1561 |
+
"rstrip": false,
|
| 1562 |
+
"single_word": false,
|
| 1563 |
+
"special": true
|
| 1564 |
+
},
|
| 1565 |
+
"157086": {
|
| 1566 |
+
"content": "<|reserved_token_191|>",
|
| 1567 |
+
"lstrip": false,
|
| 1568 |
+
"normalized": false,
|
| 1569 |
+
"rstrip": false,
|
| 1570 |
+
"single_word": false,
|
| 1571 |
+
"special": true
|
| 1572 |
+
},
|
| 1573 |
+
"157087": {
|
| 1574 |
+
"content": "<|reserved_token_192|>",
|
| 1575 |
+
"lstrip": false,
|
| 1576 |
+
"normalized": false,
|
| 1577 |
+
"rstrip": false,
|
| 1578 |
+
"single_word": false,
|
| 1579 |
+
"special": true
|
| 1580 |
+
},
|
| 1581 |
+
"157088": {
|
| 1582 |
+
"content": "<|reserved_token_193|>",
|
| 1583 |
+
"lstrip": false,
|
| 1584 |
+
"normalized": false,
|
| 1585 |
+
"rstrip": false,
|
| 1586 |
+
"single_word": false,
|
| 1587 |
+
"special": true
|
| 1588 |
+
},
|
| 1589 |
+
"157089": {
|
| 1590 |
+
"content": "<|reserved_token_194|>",
|
| 1591 |
+
"lstrip": false,
|
| 1592 |
+
"normalized": false,
|
| 1593 |
+
"rstrip": false,
|
| 1594 |
+
"single_word": false,
|
| 1595 |
+
"special": true
|
| 1596 |
+
},
|
| 1597 |
+
"157090": {
|
| 1598 |
+
"content": "<|reserved_token_195|>",
|
| 1599 |
+
"lstrip": false,
|
| 1600 |
+
"normalized": false,
|
| 1601 |
+
"rstrip": false,
|
| 1602 |
+
"single_word": false,
|
| 1603 |
+
"special": true
|
| 1604 |
+
},
|
| 1605 |
+
"157091": {
|
| 1606 |
+
"content": "<|reserved_token_196|>",
|
| 1607 |
+
"lstrip": false,
|
| 1608 |
+
"normalized": false,
|
| 1609 |
+
"rstrip": false,
|
| 1610 |
+
"single_word": false,
|
| 1611 |
+
"special": true
|
| 1612 |
+
},
|
| 1613 |
+
"157092": {
|
| 1614 |
+
"content": "<|reserved_token_197|>",
|
| 1615 |
+
"lstrip": false,
|
| 1616 |
+
"normalized": false,
|
| 1617 |
+
"rstrip": false,
|
| 1618 |
+
"single_word": false,
|
| 1619 |
+
"special": true
|
| 1620 |
+
},
|
| 1621 |
+
"157093": {
|
| 1622 |
+
"content": "<|reserved_token_198|>",
|
| 1623 |
+
"lstrip": false,
|
| 1624 |
+
"normalized": false,
|
| 1625 |
+
"rstrip": false,
|
| 1626 |
+
"single_word": false,
|
| 1627 |
+
"special": true
|
| 1628 |
+
},
|
| 1629 |
+
"157094": {
|
| 1630 |
+
"content": "<|reserved_token_199|>",
|
| 1631 |
+
"lstrip": false,
|
| 1632 |
+
"normalized": false,
|
| 1633 |
+
"rstrip": false,
|
| 1634 |
+
"single_word": false,
|
| 1635 |
+
"special": true
|
| 1636 |
+
},
|
| 1637 |
+
"157095": {
|
| 1638 |
+
"content": "<|reserved_token_200|>",
|
| 1639 |
+
"lstrip": false,
|
| 1640 |
+
"normalized": false,
|
| 1641 |
+
"rstrip": false,
|
| 1642 |
+
"single_word": false,
|
| 1643 |
+
"special": true
|
| 1644 |
+
},
|
| 1645 |
+
"157096": {
|
| 1646 |
+
"content": "<|reserved_token_201|>",
|
| 1647 |
+
"lstrip": false,
|
| 1648 |
+
"normalized": false,
|
| 1649 |
+
"rstrip": false,
|
| 1650 |
+
"single_word": false,
|
| 1651 |
+
"special": true
|
| 1652 |
+
},
|
| 1653 |
+
"157097": {
|
| 1654 |
+
"content": "<|reserved_token_202|>",
|
| 1655 |
+
"lstrip": false,
|
| 1656 |
+
"normalized": false,
|
| 1657 |
+
"rstrip": false,
|
| 1658 |
+
"single_word": false,
|
| 1659 |
+
"special": true
|
| 1660 |
+
},
|
| 1661 |
+
"157098": {
|
| 1662 |
+
"content": "<|reserved_token_203|>",
|
| 1663 |
+
"lstrip": false,
|
| 1664 |
+
"normalized": false,
|
| 1665 |
+
"rstrip": false,
|
| 1666 |
+
"single_word": false,
|
| 1667 |
+
"special": true
|
| 1668 |
+
},
|
| 1669 |
+
"157099": {
|
| 1670 |
+
"content": "<|reserved_token_204|>",
|
| 1671 |
+
"lstrip": false,
|
| 1672 |
+
"normalized": false,
|
| 1673 |
+
"rstrip": false,
|
| 1674 |
+
"single_word": false,
|
| 1675 |
+
"special": true
|
| 1676 |
+
},
|
| 1677 |
+
"157100": {
|
| 1678 |
+
"content": "<|reserved_token_205|>",
|
| 1679 |
+
"lstrip": false,
|
| 1680 |
+
"normalized": false,
|
| 1681 |
+
"rstrip": false,
|
| 1682 |
+
"single_word": false,
|
| 1683 |
+
"special": true
|
| 1684 |
+
},
|
| 1685 |
+
"157101": {
|
| 1686 |
+
"content": "<|reserved_token_206|>",
|
| 1687 |
+
"lstrip": false,
|
| 1688 |
+
"normalized": false,
|
| 1689 |
+
"rstrip": false,
|
| 1690 |
+
"single_word": false,
|
| 1691 |
+
"special": true
|
| 1692 |
+
},
|
| 1693 |
+
"157102": {
|
| 1694 |
+
"content": "<|reserved_token_207|>",
|
| 1695 |
+
"lstrip": false,
|
| 1696 |
+
"normalized": false,
|
| 1697 |
+
"rstrip": false,
|
| 1698 |
+
"single_word": false,
|
| 1699 |
+
"special": true
|
| 1700 |
+
},
|
| 1701 |
+
"157103": {
|
| 1702 |
+
"content": "<|reserved_token_208|>",
|
| 1703 |
+
"lstrip": false,
|
| 1704 |
+
"normalized": false,
|
| 1705 |
+
"rstrip": false,
|
| 1706 |
+
"single_word": false,
|
| 1707 |
+
"special": true
|
| 1708 |
+
},
|
| 1709 |
+
"157104": {
|
| 1710 |
+
"content": "<|reserved_token_209|>",
|
| 1711 |
+
"lstrip": false,
|
| 1712 |
+
"normalized": false,
|
| 1713 |
+
"rstrip": false,
|
| 1714 |
+
"single_word": false,
|
| 1715 |
+
"special": true
|
| 1716 |
+
},
|
| 1717 |
+
"157105": {
|
| 1718 |
+
"content": "<|reserved_token_210|>",
|
| 1719 |
+
"lstrip": false,
|
| 1720 |
+
"normalized": false,
|
| 1721 |
+
"rstrip": false,
|
| 1722 |
+
"single_word": false,
|
| 1723 |
+
"special": true
|
| 1724 |
+
},
|
| 1725 |
+
"157106": {
|
| 1726 |
+
"content": "<|reserved_token_211|>",
|
| 1727 |
+
"lstrip": false,
|
| 1728 |
+
"normalized": false,
|
| 1729 |
+
"rstrip": false,
|
| 1730 |
+
"single_word": false,
|
| 1731 |
+
"special": true
|
| 1732 |
+
},
|
| 1733 |
+
"157107": {
|
| 1734 |
+
"content": "<|reserved_token_212|>",
|
| 1735 |
+
"lstrip": false,
|
| 1736 |
+
"normalized": false,
|
| 1737 |
+
"rstrip": false,
|
| 1738 |
+
"single_word": false,
|
| 1739 |
+
"special": true
|
| 1740 |
+
},
|
| 1741 |
+
"157108": {
|
| 1742 |
+
"content": "<|reserved_token_213|>",
|
| 1743 |
+
"lstrip": false,
|
| 1744 |
+
"normalized": false,
|
| 1745 |
+
"rstrip": false,
|
| 1746 |
+
"single_word": false,
|
| 1747 |
+
"special": true
|
| 1748 |
+
},
|
| 1749 |
+
"157109": {
|
| 1750 |
+
"content": "<|reserved_token_214|>",
|
| 1751 |
+
"lstrip": false,
|
| 1752 |
+
"normalized": false,
|
| 1753 |
+
"rstrip": false,
|
| 1754 |
+
"single_word": false,
|
| 1755 |
+
"special": true
|
| 1756 |
+
},
|
| 1757 |
+
"157110": {
|
| 1758 |
+
"content": "<|reserved_token_215|>",
|
| 1759 |
+
"lstrip": false,
|
| 1760 |
+
"normalized": false,
|
| 1761 |
+
"rstrip": false,
|
| 1762 |
+
"single_word": false,
|
| 1763 |
+
"special": true
|
| 1764 |
+
},
|
| 1765 |
+
"157111": {
|
| 1766 |
+
"content": "<|reserved_token_216|>",
|
| 1767 |
+
"lstrip": false,
|
| 1768 |
+
"normalized": false,
|
| 1769 |
+
"rstrip": false,
|
| 1770 |
+
"single_word": false,
|
| 1771 |
+
"special": true
|
| 1772 |
+
},
|
| 1773 |
+
"157112": {
|
| 1774 |
+
"content": "<|reserved_token_217|>",
|
| 1775 |
+
"lstrip": false,
|
| 1776 |
+
"normalized": false,
|
| 1777 |
+
"rstrip": false,
|
| 1778 |
+
"single_word": false,
|
| 1779 |
+
"special": true
|
| 1780 |
+
},
|
| 1781 |
+
"157113": {
|
| 1782 |
+
"content": "<|reserved_token_218|>",
|
| 1783 |
+
"lstrip": false,
|
| 1784 |
+
"normalized": false,
|
| 1785 |
+
"rstrip": false,
|
| 1786 |
+
"single_word": false,
|
| 1787 |
+
"special": true
|
| 1788 |
+
},
|
| 1789 |
+
"157114": {
|
| 1790 |
+
"content": "<|reserved_token_219|>",
|
| 1791 |
+
"lstrip": false,
|
| 1792 |
+
"normalized": false,
|
| 1793 |
+
"rstrip": false,
|
| 1794 |
+
"single_word": false,
|
| 1795 |
+
"special": true
|
| 1796 |
+
},
|
| 1797 |
+
"157115": {
|
| 1798 |
+
"content": "<|reserved_token_220|>",
|
| 1799 |
+
"lstrip": false,
|
| 1800 |
+
"normalized": false,
|
| 1801 |
+
"rstrip": false,
|
| 1802 |
+
"single_word": false,
|
| 1803 |
+
"special": true
|
| 1804 |
+
},
|
| 1805 |
+
"157116": {
|
| 1806 |
+
"content": "<|reserved_token_221|>",
|
| 1807 |
+
"lstrip": false,
|
| 1808 |
+
"normalized": false,
|
| 1809 |
+
"rstrip": false,
|
| 1810 |
+
"single_word": false,
|
| 1811 |
+
"special": true
|
| 1812 |
+
},
|
| 1813 |
+
"157117": {
|
| 1814 |
+
"content": "<|reserved_token_222|>",
|
| 1815 |
+
"lstrip": false,
|
| 1816 |
+
"normalized": false,
|
| 1817 |
+
"rstrip": false,
|
| 1818 |
+
"single_word": false,
|
| 1819 |
+
"special": true
|
| 1820 |
+
},
|
| 1821 |
+
"157118": {
|
| 1822 |
+
"content": "<|reserved_token_223|>",
|
| 1823 |
+
"lstrip": false,
|
| 1824 |
+
"normalized": false,
|
| 1825 |
+
"rstrip": false,
|
| 1826 |
+
"single_word": false,
|
| 1827 |
+
"special": true
|
| 1828 |
+
},
|
| 1829 |
+
"157119": {
|
| 1830 |
+
"content": "<|reserved_token_224|>",
|
| 1831 |
+
"lstrip": false,
|
| 1832 |
+
"normalized": false,
|
| 1833 |
+
"rstrip": false,
|
| 1834 |
+
"single_word": false,
|
| 1835 |
+
"special": true
|
| 1836 |
+
},
|
| 1837 |
+
"157120": {
|
| 1838 |
+
"content": "<|reserved_token_225|>",
|
| 1839 |
+
"lstrip": false,
|
| 1840 |
+
"normalized": false,
|
| 1841 |
+
"rstrip": false,
|
| 1842 |
+
"single_word": false,
|
| 1843 |
+
"special": true
|
| 1844 |
+
},
|
| 1845 |
+
"157121": {
|
| 1846 |
+
"content": "<|reserved_token_226|>",
|
| 1847 |
+
"lstrip": false,
|
| 1848 |
+
"normalized": false,
|
| 1849 |
+
"rstrip": false,
|
| 1850 |
+
"single_word": false,
|
| 1851 |
+
"special": true
|
| 1852 |
+
},
|
| 1853 |
+
"157122": {
|
| 1854 |
+
"content": "<|reserved_token_227|>",
|
| 1855 |
+
"lstrip": false,
|
| 1856 |
+
"normalized": false,
|
| 1857 |
+
"rstrip": false,
|
| 1858 |
+
"single_word": false,
|
| 1859 |
+
"special": true
|
| 1860 |
+
},
|
| 1861 |
+
"157123": {
|
| 1862 |
+
"content": "<|reserved_token_228|>",
|
| 1863 |
+
"lstrip": false,
|
| 1864 |
+
"normalized": false,
|
| 1865 |
+
"rstrip": false,
|
| 1866 |
+
"single_word": false,
|
| 1867 |
+
"special": true
|
| 1868 |
+
},
|
| 1869 |
+
"157124": {
|
| 1870 |
+
"content": "<|reserved_token_229|>",
|
| 1871 |
+
"lstrip": false,
|
| 1872 |
+
"normalized": false,
|
| 1873 |
+
"rstrip": false,
|
| 1874 |
+
"single_word": false,
|
| 1875 |
+
"special": true
|
| 1876 |
+
},
|
| 1877 |
+
"157125": {
|
| 1878 |
+
"content": "<|reserved_token_230|>",
|
| 1879 |
+
"lstrip": false,
|
| 1880 |
+
"normalized": false,
|
| 1881 |
+
"rstrip": false,
|
| 1882 |
+
"single_word": false,
|
| 1883 |
+
"special": true
|
| 1884 |
+
},
|
| 1885 |
+
"157126": {
|
| 1886 |
+
"content": "<|reserved_token_231|>",
|
| 1887 |
+
"lstrip": false,
|
| 1888 |
+
"normalized": false,
|
| 1889 |
+
"rstrip": false,
|
| 1890 |
+
"single_word": false,
|
| 1891 |
+
"special": true
|
| 1892 |
+
},
|
| 1893 |
+
"157127": {
|
| 1894 |
+
"content": "<|reserved_token_232|>",
|
| 1895 |
+
"lstrip": false,
|
| 1896 |
+
"normalized": false,
|
| 1897 |
+
"rstrip": false,
|
| 1898 |
+
"single_word": false,
|
| 1899 |
+
"special": true
|
| 1900 |
+
},
|
| 1901 |
+
"157128": {
|
| 1902 |
+
"content": "<|reserved_token_233|>",
|
| 1903 |
+
"lstrip": false,
|
| 1904 |
+
"normalized": false,
|
| 1905 |
+
"rstrip": false,
|
| 1906 |
+
"single_word": false,
|
| 1907 |
+
"special": true
|
| 1908 |
+
},
|
| 1909 |
+
"157129": {
|
| 1910 |
+
"content": "<|reserved_token_234|>",
|
| 1911 |
+
"lstrip": false,
|
| 1912 |
+
"normalized": false,
|
| 1913 |
+
"rstrip": false,
|
| 1914 |
+
"single_word": false,
|
| 1915 |
+
"special": true
|
| 1916 |
+
},
|
| 1917 |
+
"157130": {
|
| 1918 |
+
"content": "<|reserved_token_235|>",
|
| 1919 |
+
"lstrip": false,
|
| 1920 |
+
"normalized": false,
|
| 1921 |
+
"rstrip": false,
|
| 1922 |
+
"single_word": false,
|
| 1923 |
+
"special": true
|
| 1924 |
+
},
|
| 1925 |
+
"157131": {
|
| 1926 |
+
"content": "<|reserved_token_236|>",
|
| 1927 |
+
"lstrip": false,
|
| 1928 |
+
"normalized": false,
|
| 1929 |
+
"rstrip": false,
|
| 1930 |
+
"single_word": false,
|
| 1931 |
+
"special": true
|
| 1932 |
+
},
|
| 1933 |
+
"157132": {
|
| 1934 |
+
"content": "<|reserved_token_237|>",
|
| 1935 |
+
"lstrip": false,
|
| 1936 |
+
"normalized": false,
|
| 1937 |
+
"rstrip": false,
|
| 1938 |
+
"single_word": false,
|
| 1939 |
+
"special": true
|
| 1940 |
+
},
|
| 1941 |
+
"157133": {
|
| 1942 |
+
"content": "<|reserved_token_238|>",
|
| 1943 |
+
"lstrip": false,
|
| 1944 |
+
"normalized": false,
|
| 1945 |
+
"rstrip": false,
|
| 1946 |
+
"single_word": false,
|
| 1947 |
+
"special": true
|
| 1948 |
+
},
|
| 1949 |
+
"157134": {
|
| 1950 |
+
"content": "<|reserved_token_239|>",
|
| 1951 |
+
"lstrip": false,
|
| 1952 |
+
"normalized": false,
|
| 1953 |
+
"rstrip": false,
|
| 1954 |
+
"single_word": false,
|
| 1955 |
+
"special": true
|
| 1956 |
+
},
|
| 1957 |
+
"157135": {
|
| 1958 |
+
"content": "<|reserved_token_240|>",
|
| 1959 |
+
"lstrip": false,
|
| 1960 |
+
"normalized": false,
|
| 1961 |
+
"rstrip": false,
|
| 1962 |
+
"single_word": false,
|
| 1963 |
+
"special": true
|
| 1964 |
+
},
|
| 1965 |
+
"157136": {
|
| 1966 |
+
"content": "<|reserved_token_241|>",
|
| 1967 |
+
"lstrip": false,
|
| 1968 |
+
"normalized": false,
|
| 1969 |
+
"rstrip": false,
|
| 1970 |
+
"single_word": false,
|
| 1971 |
+
"special": true
|
| 1972 |
+
},
|
| 1973 |
+
"157137": {
|
| 1974 |
+
"content": "<|reserved_token_242|>",
|
| 1975 |
+
"lstrip": false,
|
| 1976 |
+
"normalized": false,
|
| 1977 |
+
"rstrip": false,
|
| 1978 |
+
"single_word": false,
|
| 1979 |
+
"special": true
|
| 1980 |
+
},
|
| 1981 |
+
"157138": {
|
| 1982 |
+
"content": "<|reserved_token_243|>",
|
| 1983 |
+
"lstrip": false,
|
| 1984 |
+
"normalized": false,
|
| 1985 |
+
"rstrip": false,
|
| 1986 |
+
"single_word": false,
|
| 1987 |
+
"special": true
|
| 1988 |
+
},
|
| 1989 |
+
"157139": {
|
| 1990 |
+
"content": "<|reserved_token_244|>",
|
| 1991 |
+
"lstrip": false,
|
| 1992 |
+
"normalized": false,
|
| 1993 |
+
"rstrip": false,
|
| 1994 |
+
"single_word": false,
|
| 1995 |
+
"special": true
|
| 1996 |
+
},
|
| 1997 |
+
"157140": {
|
| 1998 |
+
"content": "<|reserved_token_245|>",
|
| 1999 |
+
"lstrip": false,
|
| 2000 |
+
"normalized": false,
|
| 2001 |
+
"rstrip": false,
|
| 2002 |
+
"single_word": false,
|
| 2003 |
+
"special": true
|
| 2004 |
+
},
|
| 2005 |
+
"157141": {
|
| 2006 |
+
"content": "<|reserved_token_246|>",
|
| 2007 |
+
"lstrip": false,
|
| 2008 |
+
"normalized": false,
|
| 2009 |
+
"rstrip": false,
|
| 2010 |
+
"single_word": false,
|
| 2011 |
+
"special": true
|
| 2012 |
+
},
|
| 2013 |
+
"157142": {
|
| 2014 |
+
"content": "<|reserved_token_247|>",
|
| 2015 |
+
"lstrip": false,
|
| 2016 |
+
"normalized": false,
|
| 2017 |
+
"rstrip": false,
|
| 2018 |
+
"single_word": false,
|
| 2019 |
+
"special": true
|
| 2020 |
+
},
|
| 2021 |
+
"157143": {
|
| 2022 |
+
"content": "<|reserved_token_248|>",
|
| 2023 |
+
"lstrip": false,
|
| 2024 |
+
"normalized": false,
|
| 2025 |
+
"rstrip": false,
|
| 2026 |
+
"single_word": false,
|
| 2027 |
+
"special": true
|
| 2028 |
+
},
|
| 2029 |
+
"157144": {
|
| 2030 |
+
"content": "<|reserved_token_249|>",
|
| 2031 |
+
"lstrip": false,
|
| 2032 |
+
"normalized": false,
|
| 2033 |
+
"rstrip": false,
|
| 2034 |
+
"single_word": false,
|
| 2035 |
+
"special": true
|
| 2036 |
+
},
|
| 2037 |
+
"157145": {
|
| 2038 |
+
"content": "<|reserved_token_250|>",
|
| 2039 |
+
"lstrip": false,
|
| 2040 |
+
"normalized": false,
|
| 2041 |
+
"rstrip": false,
|
| 2042 |
+
"single_word": false,
|
| 2043 |
+
"special": true
|
| 2044 |
+
},
|
| 2045 |
+
"157146": {
|
| 2046 |
+
"content": "<|reserved_token_251|>",
|
| 2047 |
+
"lstrip": false,
|
| 2048 |
+
"normalized": false,
|
| 2049 |
+
"rstrip": false,
|
| 2050 |
+
"single_word": false,
|
| 2051 |
+
"special": true
|
| 2052 |
+
},
|
| 2053 |
+
"157147": {
|
| 2054 |
+
"content": "<|reserved_token_252|>",
|
| 2055 |
+
"lstrip": false,
|
| 2056 |
+
"normalized": false,
|
| 2057 |
+
"rstrip": false,
|
| 2058 |
+
"single_word": false,
|
| 2059 |
+
"special": true
|
| 2060 |
+
},
|
| 2061 |
+
"157148": {
|
| 2062 |
+
"content": "<|reserved_token_253|>",
|
| 2063 |
+
"lstrip": false,
|
| 2064 |
+
"normalized": false,
|
| 2065 |
+
"rstrip": false,
|
| 2066 |
+
"single_word": false,
|
| 2067 |
+
"special": true
|
| 2068 |
+
},
|
| 2069 |
+
"157149": {
|
| 2070 |
+
"content": "<|reserved_token_254|>",
|
| 2071 |
+
"lstrip": false,
|
| 2072 |
+
"normalized": false,
|
| 2073 |
+
"rstrip": false,
|
| 2074 |
+
"single_word": false,
|
| 2075 |
+
"special": true
|
| 2076 |
+
},
|
| 2077 |
+
"157150": {
|
| 2078 |
+
"content": "<|reserved_token_255|>",
|
| 2079 |
+
"lstrip": false,
|
| 2080 |
+
"normalized": false,
|
| 2081 |
+
"rstrip": false,
|
| 2082 |
+
"single_word": false,
|
| 2083 |
+
"special": true
|
| 2084 |
+
},
|
| 2085 |
+
"157151": {
|
| 2086 |
+
"content": "<role>",
|
| 2087 |
+
"lstrip": false,
|
| 2088 |
+
"normalized": false,
|
| 2089 |
+
"rstrip": false,
|
| 2090 |
+
"single_word": false,
|
| 2091 |
+
"special": true
|
| 2092 |
+
},
|
| 2093 |
+
"157152": {
|
| 2094 |
+
"content": "</role>",
|
| 2095 |
+
"lstrip": false,
|
| 2096 |
+
"normalized": false,
|
| 2097 |
+
"rstrip": false,
|
| 2098 |
+
"single_word": false,
|
| 2099 |
+
"special": true
|
| 2100 |
+
}
|
| 2101 |
+
},
|
| 2102 |
+
"bos_token": "<|startoftext|>",
|
| 2103 |
+
"clean_up_tokenization_spaces": false,
|
| 2104 |
+
"cls_token": "[CLS]",
|
| 2105 |
+
"eos_token": "<|endoftext|>",
|
| 2106 |
+
"extra_special_tokens": {},
|
| 2107 |
+
"fast_tokenizer": true,
|
| 2108 |
+
"gmask_token": "[gMASK]",
|
| 2109 |
+
"merges_file": null,
|
| 2110 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 2111 |
+
"pad_token": "<|endoftext|>",
|
| 2112 |
+
"tokenizer_class": "PreTrainedTokenizerFast",
|
| 2113 |
+
"trust_remote_code": true
|
| 2114 |
+
}
|