gsaltintas commited on
Commit
3d669b5
·
verified ·
1 Parent(s): 8179655

Upload folder using huggingface_hub

Browse files
Files changed (6) hide show
  1. README.md +44 -0
  2. merges.txt +11 -0
  3. special_tokens_map.json +10 -0
  4. tokenizer.json +246 -0
  5. tokenizer_config.json +144 -0
  6. vocab.json +29 -0
README.md ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ language:
4
+ - und # ISO 639-3 code or "und" if not identifiable
5
+ tags:
6
+ - tokenizer
7
+ - bpe
8
+ - flexitok
9
+ - fineweb2
10
+ ---
11
+
12
+ # Byte-Level BPE Tokenizer: numeric (0K)
13
+
14
+ A **Byte-Level BPE** tokenizer trained on **numeric** data from Fineweb-2-HQ.
15
+
16
+ ## Training Details
17
+
18
+ | Parameter | Value |
19
+ |-----------|-------|
20
+ | Algorithm | Byte-Level BPE |
21
+ | Language | `numeric` |
22
+ | Target Vocab Size | 268 |
23
+ | Final Vocab Size | 27 |
24
+ | Pre-tokenizer | byte_level |
25
+ | Number handling | rtl_1digit |
26
+ | Contraction handling | False |
27
+ | Normalizer | NONE |
28
+ | Special Tokens | `<s>`, `</s>`, `<pad>`, `<unk>` |
29
+ | Training Shards | 1 |
30
+
31
+ ## Usage
32
+
33
+ ```python
34
+ from transformers import AutoTokenizer
35
+
36
+ tokenizer = AutoTokenizer.from_pretrained("flexitok/mod-tokenizers-bpe_numeric_268")
37
+ tokens = tokenizer.encode("Hello, world!")
38
+ ```
39
+
40
+ ## Files
41
+
42
+ - `tokenizer.json` — Full HuggingFace tokenizer
43
+ - `vocab.json` — Vocabulary mapping
44
+ - `merges.txt` — BPE merge rules
merges.txt ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #version: 0.2
2
+ ['Ġ', '0']
3
+ ['Ġ', '1']
4
+ ['Ġ', '2']
5
+ ['Ġ', '3']
6
+ ['Ġ', '4']
7
+ ['Ġ', '5']
8
+ ['Ġ', '6']
9
+ ['Ġ', '7']
10
+ ['Ġ', '8']
11
+ ['Ġ', '9']
special_tokens_map.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "mod",
4
+ "="
5
+ ],
6
+ "bos_token": "<s>",
7
+ "eos_token": "</s>",
8
+ "pad_token": "<pad>",
9
+ "unk_token": "<unk>"
10
+ }
tokenizer.json ADDED
@@ -0,0 +1,246 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "1.0",
3
+ "truncation": null,
4
+ "padding": null,
5
+ "added_tokens": [
6
+ {
7
+ "id": 0,
8
+ "content": "<unk>",
9
+ "single_word": false,
10
+ "lstrip": false,
11
+ "rstrip": false,
12
+ "normalized": false,
13
+ "special": true
14
+ },
15
+ {
16
+ "id": 1,
17
+ "content": "<s>",
18
+ "single_word": false,
19
+ "lstrip": false,
20
+ "rstrip": false,
21
+ "normalized": false,
22
+ "special": true
23
+ },
24
+ {
25
+ "id": 2,
26
+ "content": "</s>",
27
+ "single_word": false,
28
+ "lstrip": false,
29
+ "rstrip": false,
30
+ "normalized": false,
31
+ "special": true
32
+ },
33
+ {
34
+ "id": 3,
35
+ "content": "<pad>",
36
+ "single_word": false,
37
+ "lstrip": false,
38
+ "rstrip": false,
39
+ "normalized": false,
40
+ "special": true
41
+ },
42
+ {
43
+ "id": 4,
44
+ "content": "mod",
45
+ "single_word": false,
46
+ "lstrip": false,
47
+ "rstrip": false,
48
+ "normalized": false,
49
+ "special": true
50
+ },
51
+ {
52
+ "id": 5,
53
+ "content": "=",
54
+ "single_word": false,
55
+ "lstrip": false,
56
+ "rstrip": false,
57
+ "normalized": false,
58
+ "special": true
59
+ },
60
+ {
61
+ "id": 6,
62
+ "content": "0",
63
+ "single_word": false,
64
+ "lstrip": false,
65
+ "rstrip": false,
66
+ "normalized": true,
67
+ "special": false
68
+ },
69
+ {
70
+ "id": 7,
71
+ "content": "1",
72
+ "single_word": false,
73
+ "lstrip": false,
74
+ "rstrip": false,
75
+ "normalized": true,
76
+ "special": false
77
+ },
78
+ {
79
+ "id": 8,
80
+ "content": "2",
81
+ "single_word": false,
82
+ "lstrip": false,
83
+ "rstrip": false,
84
+ "normalized": true,
85
+ "special": false
86
+ },
87
+ {
88
+ "id": 9,
89
+ "content": "3",
90
+ "single_word": false,
91
+ "lstrip": false,
92
+ "rstrip": false,
93
+ "normalized": true,
94
+ "special": false
95
+ },
96
+ {
97
+ "id": 10,
98
+ "content": "4",
99
+ "single_word": false,
100
+ "lstrip": false,
101
+ "rstrip": false,
102
+ "normalized": true,
103
+ "special": false
104
+ },
105
+ {
106
+ "id": 11,
107
+ "content": "5",
108
+ "single_word": false,
109
+ "lstrip": false,
110
+ "rstrip": false,
111
+ "normalized": true,
112
+ "special": false
113
+ },
114
+ {
115
+ "id": 12,
116
+ "content": "6",
117
+ "single_word": false,
118
+ "lstrip": false,
119
+ "rstrip": false,
120
+ "normalized": true,
121
+ "special": false
122
+ },
123
+ {
124
+ "id": 13,
125
+ "content": "7",
126
+ "single_word": false,
127
+ "lstrip": false,
128
+ "rstrip": false,
129
+ "normalized": true,
130
+ "special": false
131
+ },
132
+ {
133
+ "id": 14,
134
+ "content": "8",
135
+ "single_word": false,
136
+ "lstrip": false,
137
+ "rstrip": false,
138
+ "normalized": true,
139
+ "special": false
140
+ },
141
+ {
142
+ "id": 15,
143
+ "content": "9",
144
+ "single_word": false,
145
+ "lstrip": false,
146
+ "rstrip": false,
147
+ "normalized": true,
148
+ "special": false
149
+ }
150
+ ],
151
+ "normalizer": null,
152
+ "pre_tokenizer": {
153
+ "type": "ByteLevel",
154
+ "add_prefix_space": false,
155
+ "trim_offsets": true,
156
+ "use_regex": true
157
+ },
158
+ "post_processor": null,
159
+ "decoder": {
160
+ "type": "ByteLevel",
161
+ "add_prefix_space": true,
162
+ "trim_offsets": true,
163
+ "use_regex": true
164
+ },
165
+ "model": {
166
+ "type": "BPE",
167
+ "dropout": null,
168
+ "unk_token": "<unk>",
169
+ "continuing_subword_prefix": null,
170
+ "end_of_word_suffix": null,
171
+ "fuse_unk": false,
172
+ "byte_fallback": false,
173
+ "ignore_merges": false,
174
+ "vocab": {
175
+ "<unk>": 0,
176
+ "<s>": 1,
177
+ "</s>": 2,
178
+ "<pad>": 3,
179
+ "mod": 4,
180
+ "=": 5,
181
+ "0": 6,
182
+ "1": 7,
183
+ "2": 8,
184
+ "3": 9,
185
+ "4": 10,
186
+ "5": 11,
187
+ "6": 12,
188
+ "7": 13,
189
+ "8": 14,
190
+ "9": 15,
191
+ "Ġ": 16,
192
+ "Ġ0": 17,
193
+ "Ġ1": 18,
194
+ "Ġ2": 19,
195
+ "Ġ3": 20,
196
+ "Ġ4": 21,
197
+ "Ġ5": 22,
198
+ "Ġ6": 23,
199
+ "Ġ7": 24,
200
+ "Ġ8": 25,
201
+ "Ġ9": 26
202
+ },
203
+ "merges": [
204
+ [
205
+ "Ġ",
206
+ "0"
207
+ ],
208
+ [
209
+ "Ġ",
210
+ "1"
211
+ ],
212
+ [
213
+ "Ġ",
214
+ "2"
215
+ ],
216
+ [
217
+ "Ġ",
218
+ "3"
219
+ ],
220
+ [
221
+ "Ġ",
222
+ "4"
223
+ ],
224
+ [
225
+ "Ġ",
226
+ "5"
227
+ ],
228
+ [
229
+ "Ġ",
230
+ "6"
231
+ ],
232
+ [
233
+ "Ġ",
234
+ "7"
235
+ ],
236
+ [
237
+ "Ġ",
238
+ "8"
239
+ ],
240
+ [
241
+ "Ġ",
242
+ "9"
243
+ ]
244
+ ]
245
+ }
246
+ }
tokenizer_config.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "<unk>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<s>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "2": {
20
+ "content": "</s>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "3": {
28
+ "content": "<pad>",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "4": {
36
+ "content": "mod",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ },
43
+ "5": {
44
+ "content": "=",
45
+ "lstrip": false,
46
+ "normalized": false,
47
+ "rstrip": false,
48
+ "single_word": false,
49
+ "special": true
50
+ },
51
+ "6": {
52
+ "content": "0",
53
+ "lstrip": false,
54
+ "normalized": true,
55
+ "rstrip": false,
56
+ "single_word": false,
57
+ "special": false
58
+ },
59
+ "7": {
60
+ "content": "1",
61
+ "lstrip": false,
62
+ "normalized": true,
63
+ "rstrip": false,
64
+ "single_word": false,
65
+ "special": false
66
+ },
67
+ "8": {
68
+ "content": "2",
69
+ "lstrip": false,
70
+ "normalized": true,
71
+ "rstrip": false,
72
+ "single_word": false,
73
+ "special": false
74
+ },
75
+ "9": {
76
+ "content": "3",
77
+ "lstrip": false,
78
+ "normalized": true,
79
+ "rstrip": false,
80
+ "single_word": false,
81
+ "special": false
82
+ },
83
+ "10": {
84
+ "content": "4",
85
+ "lstrip": false,
86
+ "normalized": true,
87
+ "rstrip": false,
88
+ "single_word": false,
89
+ "special": false
90
+ },
91
+ "11": {
92
+ "content": "5",
93
+ "lstrip": false,
94
+ "normalized": true,
95
+ "rstrip": false,
96
+ "single_word": false,
97
+ "special": false
98
+ },
99
+ "12": {
100
+ "content": "6",
101
+ "lstrip": false,
102
+ "normalized": true,
103
+ "rstrip": false,
104
+ "single_word": false,
105
+ "special": false
106
+ },
107
+ "13": {
108
+ "content": "7",
109
+ "lstrip": false,
110
+ "normalized": true,
111
+ "rstrip": false,
112
+ "single_word": false,
113
+ "special": false
114
+ },
115
+ "14": {
116
+ "content": "8",
117
+ "lstrip": false,
118
+ "normalized": true,
119
+ "rstrip": false,
120
+ "single_word": false,
121
+ "special": false
122
+ },
123
+ "15": {
124
+ "content": "9",
125
+ "lstrip": false,
126
+ "normalized": true,
127
+ "rstrip": false,
128
+ "single_word": false,
129
+ "special": false
130
+ }
131
+ },
132
+ "additional_special_tokens": [
133
+ "mod",
134
+ "="
135
+ ],
136
+ "bos_token": "<s>",
137
+ "clean_up_tokenization_spaces": false,
138
+ "eos_token": "</s>",
139
+ "extra_special_tokens": {},
140
+ "model_max_length": 1000000000000000019884624838656,
141
+ "pad_token": "<pad>",
142
+ "tokenizer_class": "PreTrainedTokenizerFast",
143
+ "unk_token": "<unk>"
144
+ }
vocab.json ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "Ġ2": 19,
3
+ "Ġ0": 17,
4
+ "Ġ9": 26,
5
+ "Ġ6": 23,
6
+ "1": 7,
7
+ "Ġ7": 24,
8
+ "Ġ4": 21,
9
+ "4": 10,
10
+ "=": 5,
11
+ "2": 8,
12
+ "Ġ3": 20,
13
+ "5": 11,
14
+ "3": 9,
15
+ "9": 15,
16
+ "mod": 4,
17
+ "0": 6,
18
+ "8": 14,
19
+ "Ġ": 16,
20
+ "Ġ1": 18,
21
+ "<pad>": 3,
22
+ "</s>": 2,
23
+ "6": 12,
24
+ "Ġ5": 22,
25
+ "<s>": 1,
26
+ "Ġ8": 25,
27
+ "7": 13,
28
+ "<unk>": 0
29
+ }