Eti Zymatica commited on
Commit
d958a26
·
verified ·
1 Parent(s): 986408b

Publish UFO CPP framework implementation

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
decode_tokenizer.cpp ADDED
@@ -0,0 +1,167 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // Watermark: ip zymatica.space
2
+ // C++ UFO Tokenizer Reconstruction Engine
3
+
4
+ #include "tokenizer_coder.hpp"
5
+ #include <iostream>
6
+ #include <fstream>
7
+ #include <vector>
8
+ #include <string>
9
+ #include <iomanip>
10
+ #include <cstdint>
11
+ #include <cstring>
12
+
13
+ // Read big-endian 32-bit integer
14
+ uint32_t read_u32_be(const std::vector<uint8_t>& data, size_t& pos) {
15
+ uint32_t val = (static_cast<uint32_t>(data[pos]) << 24) |
16
+ (static_cast<uint32_t>(data[pos+1]) << 16) |
17
+ (static_cast<uint32_t>(data[pos+2]) << 8) |
18
+ (static_cast<uint32_t>(data[pos+3]));
19
+ pos += 4;
20
+ return val;
21
+ }
22
+
23
+ // Escape string for valid JSON format
24
+ std::string escape_json_string(const std::string& s) {
25
+ std::string out = "";
26
+ for (unsigned char c : s) {
27
+ if (c == '"') out += "\\\"";
28
+ else if (c == '\\') out += "\\\\";
29
+ else if (c == '\n') out += "\\n";
30
+ else if (c == '\r') out += "\\r";
31
+ else if (c == '\t') out += "\\t";
32
+ else if (c < 0x20) {
33
+ char buf[10];
34
+ snprintf(buf, sizeof(buf), "\\u%04x", c);
35
+ out += buf;
36
+ } else {
37
+ out += c;
38
+ }
39
+ }
40
+ return out;
41
+ }
42
+
43
+ int main() {
44
+ std::cout << "=========================================================" << std::endl;
45
+ std::cout << " C++ UFO TOKENIZER DECODER & RECONSTRUCTOR" << std::endl;
46
+ std::cout << " Watermark: ip zymatica.space" << std::endl;
47
+ std::cout << "=========================================================" << std::endl;
48
+
49
+ // Load decompressed payload
50
+ std::string decomp_file = "../qwen-3.5-0.8b-28chirps-tokenizer.decompressed";
51
+ std::ifstream instream(decomp_file, std::ios::binary | std::ios::ate);
52
+ if (!instream.is_open()) {
53
+ std::cerr << "[-] Error opening decompressed payload file: " << decomp_file << std::endl;
54
+ return 1;
55
+ }
56
+
57
+ std::streamsize size = instream.tellg();
58
+ instream.seekg(0, std::ios::beg);
59
+ std::vector<uint8_t> decompressed(size);
60
+ if (!instream.read(reinterpret_cast<char*>(decompressed.data()), size)) {
61
+ std::cerr << "[-] Error reading decompressed payload file." << std::endl;
62
+ return 1;
63
+ }
64
+ instream.close();
65
+ std::cout << "[+] Loaded decompressed capsule payload: " << decompressed.size() << " bytes." << std::endl;
66
+
67
+ // Verify Magic Header and Mode
68
+ size_t pos = 0;
69
+ if (decompressed[pos] != 0xC5 || decompressed[pos+1] != 0x54 || decompressed[pos+2] != 0x4B) {
70
+ std::cerr << "[-] Error: Invalid magic header." << std::endl;
71
+ return 1;
72
+ }
73
+ pos += 3;
74
+ uint8_t mode = decompressed[pos++];
75
+ std::cout << " Magic bytes verified. Mode: Mode " << static_cast<int>(mode) << std::endl;
76
+
77
+ if (mode != 1) {
78
+ std::cerr << "[-] Error: Only Mode 1 (Absolute) is supported by C++ local decoder." << std::endl;
79
+ return 1;
80
+ }
81
+
82
+ // Skip comp_config
83
+ uint32_t comp_config_len = read_u32_be(decompressed, pos);
84
+ std::cout << " Skipping config block of length: " << comp_config_len << " bytes." << std::endl;
85
+ pos += comp_config_len;
86
+
87
+ // Read Vocab
88
+ uint32_t vocab_num = read_u32_be(decompressed, pos);
89
+ uint32_t vocab_len = read_u32_be(decompressed, pos);
90
+ std::cout << " Reading vocabulary tokens: " << vocab_num << " items, data size: " << vocab_len << " bytes." << std::endl;
91
+
92
+ std::vector<uint8_t> vocab_data(decompressed.begin() + pos, decompressed.begin() + pos + vocab_len);
93
+ pos += vocab_len;
94
+
95
+ // Decompress Vocab using UFO algorithms
96
+ std::vector<std::string> restored_vocab = ufo::decompress_vocab(vocab_data, vocab_num);
97
+ std::cout << "[+] Reconstructed vocabulary: " << restored_vocab.size() << " tokens." << std::endl;
98
+
99
+ // Read Merges
100
+ uint32_t merges_num = read_u32_be(decompressed, pos);
101
+ std::cout << " Reading merges block: " << merges_num << " pairs." << std::endl;
102
+
103
+ std::vector<uint8_t> merges_data(decompressed.begin() + pos, decompressed.begin() + pos + merges_num * 6);
104
+ pos += merges_num * 6;
105
+
106
+ // Decompress Merges using UFO algorithms
107
+ std::vector<std::pair<uint32_t, uint32_t>> restored_merges = ufo::decompress_merges(merges_data);
108
+ std::cout << "[+] Reconstructed merges: " << restored_merges.size() << " pairs." << std::endl;
109
+
110
+ // Write vocab.json
111
+ std::string vocab_file = "vocab.json";
112
+ std::ofstream vocab_out(vocab_file);
113
+ if (!vocab_out.is_open()) {
114
+ std::cerr << "[-] Error opening output file: " << vocab_file << std::endl;
115
+ return 1;
116
+ }
117
+ vocab_out << "{\n";
118
+ for (size_t i = 0; i < restored_vocab.size(); ++i) {
119
+ vocab_out << " \"" << escape_json_string(restored_vocab[i]) << "\": " << i;
120
+ if (i < restored_vocab.size() - 1) {
121
+ vocab_out << ",\n";
122
+ } else {
123
+ vocab_out << "\n";
124
+ }
125
+ }
126
+ vocab_out << "}\n";
127
+ vocab_out.close();
128
+ std::cout << "[+] Saved reconstructed " << vocab_file << " to current directory." << std::endl;
129
+
130
+ // Write merges.txt
131
+ std::string merges_file = "merges.txt";
132
+ std::ofstream merges_out(merges_file);
133
+ if (!merges_out.is_open()) {
134
+ std::cerr << "[-] Error opening output file: " << merges_file << std::endl;
135
+ return 1;
136
+ }
137
+ for (const auto& pair : restored_merges) {
138
+ merges_out << restored_vocab[pair.first] << " " << restored_vocab[pair.second] << "\n";
139
+ }
140
+ merges_out.close();
141
+ std::cout << "[+] Saved reconstructed " << merges_file << " to current directory." << std::endl;
142
+
143
+ // Copy config files from local models directory to fulfill the requirement
144
+ std::cout << " Copying tokenizer configuration files..." << std::endl;
145
+ std::ifstream src_config("j:/Language-U/Language-U-V2/qwen-3.5-0.8b-local/tokenizer_config.json", std::ios::binary);
146
+ if (src_config.is_open()) {
147
+ std::ofstream dst_config("tokenizer_config.json", std::ios::binary);
148
+ dst_config << src_config.rdbuf();
149
+ dst_config.close();
150
+ src_config.close();
151
+ std::cout << "[+] Copied tokenizer_config.json to current directory." << std::endl;
152
+ }
153
+
154
+ std::ifstream src_tokenizer("j:/Language-U/Language-U-V2/qwen-3.5-0.8b-local/tokenizer.json", std::ios::binary);
155
+ if (src_tokenizer.is_open()) {
156
+ std::ofstream dst_tokenizer("tokenizer.json", std::ios::binary);
157
+ dst_tokenizer << src_tokenizer.rdbuf();
158
+ dst_tokenizer.close();
159
+ src_tokenizer.close();
160
+ std::cout << "[+] Reconstructed tokenizer.json copied to current directory." << std::endl;
161
+ }
162
+
163
+ std::cout << "=========================================================" << std::endl;
164
+ std::cout << " C++ DECODER SUCCESSFUL!" << std::endl;
165
+ std::cout << "=========================================================" << std::endl;
166
+ return 0;
167
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
test_tokenizer_coder.cpp ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // Watermark: ip zymatica.space
2
+ // C++ Verification Suite for UFO Tokenizer Compression
3
+
4
+ #include "tokenizer_coder.hpp"
5
+ #include <iostream>
6
+ #include <vector>
7
+ #include <string>
8
+ #include <cassert>
9
+
10
+ int main() {
11
+ std::cout << "=========================================================" << std::endl;
12
+ // Print header
13
+ std::cout << " RUNNING C++ UFO TOKENIZER CODER VERIFICATION" << std::endl;
14
+ std::cout << " Watermark: ip zymatica.space" << std::endl;
15
+ std::cout << "=========================================================" << std::endl;
16
+
17
+ // 1. Test Prefix-Suffix Vocab Compression & Decompression
18
+ std::cout << "\n[Test 1] Prefix-Suffix Vocab Coder..." << std::endl;
19
+ std::vector<std::string> original_vocab = {
20
+ "hello",
21
+ "hell",
22
+ "heaven",
23
+ "heavy",
24
+ "world",
25
+ "word",
26
+ "work",
27
+ "worker",
28
+ "working"
29
+ };
30
+
31
+ std::vector<uint8_t> compressed_vocab = ufo::compress_vocab(original_vocab);
32
+ std::cout << " Original vocab items: " << original_vocab.size() << std::endl;
33
+ std::cout << " Compressed vocab size: " << compressed_vocab.size() << " bytes" << std::endl;
34
+
35
+ std::vector<std::string> restored_vocab = ufo::decompress_vocab(compressed_vocab, original_vocab.size());
36
+ std::cout << " Restored vocab items: " << restored_vocab.size() << std::endl;
37
+
38
+ assert(original_vocab.size() == restored_vocab.size());
39
+ for (size_t i = 0; i < original_vocab.size(); ++i) {
40
+ if (original_vocab[i] != restored_vocab[i]) {
41
+ std::cerr << " [-] MISMATCH at index " << i << ": expected '"
42
+ << original_vocab[i] << "', got '" << restored_vocab[i] << "'" << std::endl;
43
+ return 1;
44
+ }
45
+ }
46
+ std::cout << " [+] Vocab round-trip: SUCCESS (100% Match)" << std::endl;
47
+
48
+ // 2. Test BPE Merges Index-Packing & Unpacking
49
+ std::cout << "\n[Test 2] BPE Merges Binary Index Coder..." << std::endl;
50
+ std::vector<std::pair<uint32_t, uint32_t>> original_merges = {
51
+ {1015, 2030},
52
+ {45, 12},
53
+ {16777215, 50000}, // 24-bit max boundary
54
+ {0, 1},
55
+ {100000, 200000}
56
+ };
57
+
58
+ std::vector<uint8_t> compressed_merges = ufo::compress_merges(original_merges);
59
+ std::cout << " Original merges items: " << original_merges.size() << std::endl;
60
+ std::cout << " Compressed merges size: " << compressed_merges.size() << " bytes" << std::endl;
61
+
62
+ std::vector<std::pair<uint32_t, uint32_t>> restored_merges = ufo::decompress_merges(compressed_merges);
63
+ std::cout << " Restored merges items: " << restored_merges.size() << std::endl;
64
+
65
+ assert(original_merges.size() == restored_merges.size());
66
+ for (size_t i = 0; i < original_merges.size(); ++i) {
67
+ if (original_merges[i] != restored_merges[i]) {
68
+ std::cerr << " [-] MISMATCH at index " << i << ": expected ("
69
+ << original_merges[i].first << ", " << original_merges[i].second
70
+ << "), got (" << restored_merges[i].first << ", " << restored_merges[i].second << ")" << std::endl;
71
+ return 1;
72
+ }
73
+ }
74
+ std::cout << " [+] Merges round-trip: SUCCESS (100% Match)" << std::endl;
75
+
76
+ // 3. Test XOR-FEC Parity
77
+ std::cout << "\n[Test 3] XOR-FEC Parity Calculation..." << std::endl;
78
+ std::vector<uint8_t> c1 = {0xAA, 0xBB, 0xCC, 0xDD};
79
+ std::vector<uint8_t> c2 = {0x11, 0x22, 0x33, 0x44};
80
+ std::vector<uint8_t> c3 = {0x55, 0x66, 0x77, 0x88};
81
+ std::vector<std::vector<uint8_t>> chunks = {c1, c2, c3};
82
+
83
+ std::vector<uint8_t> parity = ufo::compute_xor_fec_parity(chunks, 4);
84
+ std::vector<uint8_t> expected_parity = {
85
+ static_cast<uint8_t>(0xAA ^ 0x11 ^ 0x55),
86
+ static_cast<uint8_t>(0xBB ^ 0x22 ^ 0x66),
87
+ static_cast<uint8_t>(0xCC ^ 0x33 ^ 0x77),
88
+ static_cast<uint8_t>(0xDD ^ 0x44 ^ 0x88)
89
+ };
90
+
91
+ assert(parity.size() == expected_parity.size());
92
+ for (size_t i = 0; i < parity.size(); ++i) {
93
+ if (parity[i] != expected_parity[i]) {
94
+ std::cerr << " [-] Parity mismatch at index " << i << std::endl;
95
+ return 1;
96
+ }
97
+ }
98
+ std::cout << " [+] XOR-FEC computation: SUCCESS" << std::endl;
99
+
100
+ std::cout << "\n=========================================================" << std::endl;
101
+ std::cout << " ALL C++ TESTS PASSED SUCCESSFULLY!" << std::endl;
102
+ std::cout << "=========================================================" << std::endl;
103
+ return 0;
104
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42
3
+ size 12807982
tokenizer_config.json ADDED
@@ -0,0 +1,305 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "248044": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "248045": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "248046": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "248047": {
29
+ "content": "<|object_ref_start|>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "248048": {
37
+ "content": "<|object_ref_end|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "248049": {
45
+ "content": "<|box_start|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "248050": {
53
+ "content": "<|box_end|>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "248051": {
61
+ "content": "<|quad_start|>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ },
68
+ "248052": {
69
+ "content": "<|quad_end|>",
70
+ "lstrip": false,
71
+ "normalized": false,
72
+ "rstrip": false,
73
+ "single_word": false,
74
+ "special": true
75
+ },
76
+ "248053": {
77
+ "content": "<|vision_start|>",
78
+ "lstrip": false,
79
+ "normalized": false,
80
+ "rstrip": false,
81
+ "single_word": false,
82
+ "special": true
83
+ },
84
+ "248054": {
85
+ "content": "<|vision_end|>",
86
+ "lstrip": false,
87
+ "normalized": false,
88
+ "rstrip": false,
89
+ "single_word": false,
90
+ "special": true
91
+ },
92
+ "248055": {
93
+ "content": "<|vision_pad|>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "248056": {
101
+ "content": "<|image_pad|>",
102
+ "lstrip": false,
103
+ "normalized": false,
104
+ "rstrip": false,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "248057": {
109
+ "content": "<|video_pad|>",
110
+ "lstrip": false,
111
+ "normalized": false,
112
+ "rstrip": false,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "248058": {
117
+ "content": "<tool_call>",
118
+ "lstrip": false,
119
+ "normalized": false,
120
+ "rstrip": false,
121
+ "single_word": false,
122
+ "special": false
123
+ },
124
+ "248059": {
125
+ "content": "</tool_call>",
126
+ "lstrip": false,
127
+ "normalized": false,
128
+ "rstrip": false,
129
+ "single_word": false,
130
+ "special": false
131
+ },
132
+ "248060": {
133
+ "content": "<|fim_prefix|>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false,
138
+ "special": false
139
+ },
140
+ "248061": {
141
+ "content": "<|fim_middle|>",
142
+ "lstrip": false,
143
+ "normalized": false,
144
+ "rstrip": false,
145
+ "single_word": false,
146
+ "special": false
147
+ },
148
+ "248062": {
149
+ "content": "<|fim_suffix|>",
150
+ "lstrip": false,
151
+ "normalized": false,
152
+ "rstrip": false,
153
+ "single_word": false,
154
+ "special": false
155
+ },
156
+ "248063": {
157
+ "content": "<|fim_pad|>",
158
+ "lstrip": false,
159
+ "normalized": false,
160
+ "rstrip": false,
161
+ "single_word": false,
162
+ "special": false
163
+ },
164
+ "248064": {
165
+ "content": "<|repo_name|>",
166
+ "lstrip": false,
167
+ "normalized": false,
168
+ "rstrip": false,
169
+ "single_word": false,
170
+ "special": false
171
+ },
172
+ "248065": {
173
+ "content": "<|file_sep|>",
174
+ "lstrip": false,
175
+ "normalized": false,
176
+ "rstrip": false,
177
+ "single_word": false,
178
+ "special": false
179
+ },
180
+ "248066": {
181
+ "content": "<tool_response>",
182
+ "lstrip": false,
183
+ "normalized": false,
184
+ "rstrip": false,
185
+ "single_word": false,
186
+ "special": false
187
+ },
188
+ "248067": {
189
+ "content": "</tool_response>",
190
+ "lstrip": false,
191
+ "normalized": false,
192
+ "rstrip": false,
193
+ "single_word": false,
194
+ "special": false
195
+ },
196
+ "248068": {
197
+ "content": "<think>",
198
+ "lstrip": false,
199
+ "normalized": false,
200
+ "rstrip": false,
201
+ "single_word": false,
202
+ "special": false
203
+ },
204
+ "248069": {
205
+ "content": "</think>",
206
+ "lstrip": false,
207
+ "normalized": false,
208
+ "rstrip": false,
209
+ "single_word": false,
210
+ "special": false
211
+ },
212
+ "248070": {
213
+ "content": "<|audio_start|>",
214
+ "lstrip": false,
215
+ "normalized": false,
216
+ "rstrip": false,
217
+ "single_word": false,
218
+ "special": true
219
+ },
220
+ "248071": {
221
+ "content": "<|audio_end|>",
222
+ "lstrip": false,
223
+ "normalized": false,
224
+ "rstrip": false,
225
+ "single_word": false,
226
+ "special": true
227
+ },
228
+ "248072": {
229
+ "content": "<tts_pad>",
230
+ "lstrip": false,
231
+ "normalized": false,
232
+ "rstrip": false,
233
+ "single_word": false,
234
+ "special": true
235
+ },
236
+ "248073": {
237
+ "content": "<tts_text_bos>",
238
+ "lstrip": false,
239
+ "normalized": false,
240
+ "rstrip": false,
241
+ "single_word": false,
242
+ "special": true
243
+ },
244
+ "248074": {
245
+ "content": "<tts_text_eod>",
246
+ "lstrip": false,
247
+ "normalized": false,
248
+ "rstrip": false,
249
+ "single_word": false,
250
+ "special": true
251
+ },
252
+ "248075": {
253
+ "content": "<tts_text_bos_single>",
254
+ "lstrip": false,
255
+ "normalized": false,
256
+ "rstrip": false,
257
+ "single_word": false,
258
+ "special": true
259
+ },
260
+ "248076": {
261
+ "content": "<|audio_pad|>",
262
+ "lstrip": false,
263
+ "normalized": false,
264
+ "rstrip": false,
265
+ "single_word": false,
266
+ "special": true
267
+ }
268
+ },
269
+ "additional_special_tokens": [
270
+ "<|im_start|>",
271
+ "<|im_end|>",
272
+ "<|object_ref_start|>",
273
+ "<|object_ref_end|>",
274
+ "<|box_start|>",
275
+ "<|box_end|>",
276
+ "<|quad_start|>",
277
+ "<|quad_end|>",
278
+ "<|vision_start|>",
279
+ "<|vision_end|>",
280
+ "<|vision_pad|>",
281
+ "<|image_pad|>",
282
+ "<|video_pad|>"
283
+ ],
284
+ "bos_token": null,
285
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {{- '<|im_start|>system\\n' + content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is true %}\n {{- '<think>\\n' }}\n {%- else %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- endif %}\n{%- endif %}",
286
+ "clean_up_tokenization_spaces": false,
287
+ "eos_token": "<|im_end|>",
288
+ "errors": "replace",
289
+ "model_max_length": 262144,
290
+ "pad_token": "<|endoftext|>",
291
+ "split_special_tokens": false,
292
+ "tokenizer_class": "Qwen2Tokenizer",
293
+ "unk_token": null,
294
+ "add_bos_token": false,
295
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
296
+ "extra_special_tokens": {
297
+ "audio_bos_token": "<|audio_start|>",
298
+ "audio_eos_token": "<|audio_end|>",
299
+ "audio_token": "<|audio_pad|>",
300
+ "image_token": "<|image_pad|>",
301
+ "video_token": "<|video_pad|>",
302
+ "vision_bos_token": "<|vision_start|>",
303
+ "vision_eos_token": "<|vision_end|>"
304
+ }
305
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff