nono1224 commited on
Commit
424b2f2
·
verified ·
1 Parent(s): b831083

Upload tokenizer

Browse files
Files changed (4) hide show
  1. special_tokens_map.json +10 -40
  2. tokenizer.json +2 -2
  3. tokenizer.model +3 -0
  4. tokenizer_config.json +121 -38
special_tokens_map.json CHANGED
@@ -1,80 +1,50 @@
1
  {
2
- "additional_special_tokens": [
3
- {
4
- "content": "<ent>",
5
- "lstrip": false,
6
- "normalized": true,
7
- "rstrip": false,
8
- "single_word": false
9
- },
10
- {
11
- "content": "<ent2>",
12
- "lstrip": false,
13
- "normalized": true,
14
- "rstrip": false,
15
- "single_word": false
16
- },
17
- {
18
- "content": "<ent>",
19
- "lstrip": false,
20
- "normalized": true,
21
- "rstrip": false,
22
- "single_word": false
23
- },
24
- {
25
- "content": "<ent2>",
26
- "lstrip": false,
27
- "normalized": true,
28
- "rstrip": false,
29
- "single_word": false
30
- }
31
- ],
32
  "bos_token": {
33
  "content": "<s>",
34
  "lstrip": false,
35
- "normalized": true,
36
  "rstrip": false,
37
  "single_word": false
38
  },
39
  "cls_token": {
40
- "content": "<s>",
41
  "lstrip": false,
42
- "normalized": true,
43
  "rstrip": false,
44
  "single_word": false
45
  },
46
  "eos_token": {
47
  "content": "</s>",
48
  "lstrip": false,
49
- "normalized": true,
50
  "rstrip": false,
51
  "single_word": false
52
  },
53
  "mask_token": {
54
  "content": "<mask>",
55
- "lstrip": true,
56
- "normalized": true,
57
  "rstrip": false,
58
  "single_word": false
59
  },
60
  "pad_token": {
61
  "content": "<pad>",
62
  "lstrip": false,
63
- "normalized": true,
64
  "rstrip": false,
65
  "single_word": false
66
  },
67
  "sep_token": {
68
- "content": "</s>",
69
  "lstrip": false,
70
- "normalized": true,
71
  "rstrip": false,
72
  "single_word": false
73
  },
74
  "unk_token": {
75
  "content": "<unk>",
76
  "lstrip": false,
77
- "normalized": true,
78
  "rstrip": false,
79
  "single_word": false
80
  }
 
1
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  "bos_token": {
3
  "content": "<s>",
4
  "lstrip": false,
5
+ "normalized": false,
6
  "rstrip": false,
7
  "single_word": false
8
  },
9
  "cls_token": {
10
+ "content": "<cls>",
11
  "lstrip": false,
12
+ "normalized": false,
13
  "rstrip": false,
14
  "single_word": false
15
  },
16
  "eos_token": {
17
  "content": "</s>",
18
  "lstrip": false,
19
+ "normalized": false,
20
  "rstrip": false,
21
  "single_word": false
22
  },
23
  "mask_token": {
24
  "content": "<mask>",
25
+ "lstrip": false,
26
+ "normalized": false,
27
  "rstrip": false,
28
  "single_word": false
29
  },
30
  "pad_token": {
31
  "content": "<pad>",
32
  "lstrip": false,
33
+ "normalized": false,
34
  "rstrip": false,
35
  "single_word": false
36
  },
37
  "sep_token": {
38
+ "content": "<sep>",
39
  "lstrip": false,
40
+ "normalized": false,
41
  "rstrip": false,
42
  "single_word": false
43
  },
44
  "unk_token": {
45
  "content": "<unk>",
46
  "lstrip": false,
47
+ "normalized": false,
48
  "rstrip": false,
49
  "single_word": false
50
  }
tokenizer.json CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ab06a40b6fb706ccde8547f0f2fca96968d9af24f49a214da6693d954f3ee4fa
3
- size 8656721
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9640c4cc0aba69ddb3b0d70601a75cb9fc55688868539ce240c5031d6ce6acec
3
+ size 6724970
tokenizer.model ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:008293028e1a9d9a1038d9b63d989a2319797dfeaa03f171093a57b33a3a8277
3
+ size 1831879
tokenizer_config.json CHANGED
@@ -1,18 +1,21 @@
1
  {
 
 
 
2
  "add_prefix_space": false,
3
  "added_tokens_decoder": {
4
  "0": {
5
- "content": "<s>",
6
  "lstrip": false,
7
- "normalized": true,
8
  "rstrip": false,
9
  "single_word": false,
10
  "special": true
11
  },
12
  "1": {
13
- "content": "<pad>",
14
  "lstrip": false,
15
- "normalized": true,
16
  "rstrip": false,
17
  "single_word": false,
18
  "special": true
@@ -20,69 +23,149 @@
20
  "2": {
21
  "content": "</s>",
22
  "lstrip": false,
23
- "normalized": true,
24
  "rstrip": false,
25
  "single_word": false,
26
  "special": true
27
  },
28
  "3": {
29
- "content": "<unk>",
30
  "lstrip": false,
31
- "normalized": true,
32
  "rstrip": false,
33
  "single_word": false,
34
  "special": true
35
  },
36
- "50264": {
37
- "content": "<mask>",
38
- "lstrip": true,
39
- "normalized": true,
40
  "rstrip": false,
41
  "single_word": false,
42
  "special": true
43
  },
44
- "50265": {
45
- "content": "<ent>",
46
  "lstrip": false,
47
- "normalized": true,
48
  "rstrip": false,
49
  "single_word": false,
50
  "special": true
51
  },
52
- "50266": {
53
- "content": "<ent2>",
54
  "lstrip": false,
55
- "normalized": true,
56
  "rstrip": false,
57
  "single_word": false,
58
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
59
  }
60
  },
61
- "additional_special_tokens": [
62
- "<ent>",
63
- "<ent2>",
64
- "<ent>",
65
- "<ent2>"
66
- ],
67
  "bos_token": "<s>",
68
  "clean_up_tokenization_spaces": false,
69
- "cls_token": "<s>",
70
- "entity_mask2_token": "[MASK2]",
71
- "entity_mask_token": "[MASK]",
72
- "entity_pad_token": "[PAD]",
73
- "entity_token_1": "<ent>",
74
- "entity_token_2": "<ent2>",
75
- "entity_unk_token": "[UNK]",
76
  "eos_token": "</s>",
77
- "errors": "replace",
78
  "extra_special_tokens": {},
 
 
79
  "mask_token": "<mask>",
80
- "max_entity_length": 32,
81
- "max_mention_length": 30,
82
- "model_max_length": 512,
83
  "pad_token": "<pad>",
84
- "sep_token": "</s>",
85
- "task": null,
86
- "tokenizer_class": "LukeTokenizer",
87
- "unk_token": "<unk>"
 
 
 
88
  }
 
1
  {
2
+ "add_bos_token": true,
3
+ "add_dummy_prefix_space": false,
4
+ "add_eos_token": true,
5
  "add_prefix_space": false,
6
  "added_tokens_decoder": {
7
  "0": {
8
+ "content": "<unk>",
9
  "lstrip": false,
10
+ "normalized": false,
11
  "rstrip": false,
12
  "single_word": false,
13
  "special": true
14
  },
15
  "1": {
16
+ "content": "<s>",
17
  "lstrip": false,
18
+ "normalized": false,
19
  "rstrip": false,
20
  "single_word": false,
21
  "special": true
 
23
  "2": {
24
  "content": "</s>",
25
  "lstrip": false,
26
+ "normalized": false,
27
  "rstrip": false,
28
  "single_word": false,
29
  "special": true
30
  },
31
  "3": {
32
+ "content": "<pad>",
33
  "lstrip": false,
34
+ "normalized": false,
35
  "rstrip": false,
36
  "single_word": false,
37
  "special": true
38
  },
39
+ "4": {
40
+ "content": "<sep>",
41
+ "lstrip": false,
42
+ "normalized": false,
43
  "rstrip": false,
44
  "single_word": false,
45
  "special": true
46
  },
47
+ "5": {
48
+ "content": "<mask>",
49
  "lstrip": false,
50
+ "normalized": false,
51
  "rstrip": false,
52
  "single_word": false,
53
  "special": true
54
  },
55
+ "6": {
56
+ "content": "<cls>",
57
  "lstrip": false,
58
+ "normalized": false,
59
  "rstrip": false,
60
  "single_word": false,
61
  "special": true
62
+ },
63
+ "7": {
64
+ "content": "<|system|>",
65
+ "lstrip": false,
66
+ "normalized": false,
67
+ "rstrip": false,
68
+ "single_word": false,
69
+ "special": false
70
+ },
71
+ "8": {
72
+ "content": "<|assistant|>",
73
+ "lstrip": false,
74
+ "normalized": false,
75
+ "rstrip": false,
76
+ "single_word": false,
77
+ "special": false
78
+ },
79
+ "9": {
80
+ "content": "<|user|>",
81
+ "lstrip": false,
82
+ "normalized": false,
83
+ "rstrip": false,
84
+ "single_word": false,
85
+ "special": false
86
+ },
87
+ "10": {
88
+ "content": "<|available_tools|>",
89
+ "lstrip": false,
90
+ "normalized": false,
91
+ "rstrip": false,
92
+ "single_word": false,
93
+ "special": false
94
+ },
95
+ "11": {
96
+ "content": "<|tool_calls|>",
97
+ "lstrip": false,
98
+ "normalized": false,
99
+ "rstrip": false,
100
+ "single_word": false,
101
+ "special": false
102
+ },
103
+ "12": {
104
+ "content": "<|tool_results|>",
105
+ "lstrip": false,
106
+ "normalized": false,
107
+ "rstrip": false,
108
+ "single_word": false,
109
+ "special": false
110
+ },
111
+ "13": {
112
+ "content": "<|code|>",
113
+ "lstrip": false,
114
+ "normalized": false,
115
+ "rstrip": false,
116
+ "single_word": false,
117
+ "special": false
118
+ },
119
+ "14": {
120
+ "content": "<|file|>",
121
+ "lstrip": false,
122
+ "normalized": false,
123
+ "rstrip": false,
124
+ "single_word": false,
125
+ "special": false
126
+ },
127
+ "102397": {
128
+ "content": "<|prefix|>",
129
+ "lstrip": false,
130
+ "normalized": false,
131
+ "rstrip": false,
132
+ "single_word": false,
133
+ "special": false
134
+ },
135
+ "102398": {
136
+ "content": "<|suffix|>",
137
+ "lstrip": false,
138
+ "normalized": false,
139
+ "rstrip": false,
140
+ "single_word": false,
141
+ "special": false
142
+ },
143
+ "102399": {
144
+ "content": "<|middle|>",
145
+ "lstrip": false,
146
+ "normalized": false,
147
+ "rstrip": false,
148
+ "single_word": false,
149
+ "special": false
150
  }
151
  },
 
 
 
 
 
 
152
  "bos_token": "<s>",
153
  "clean_up_tokenization_spaces": false,
154
+ "cls_token": "<cls>",
155
+ "do_lower_case": false,
 
 
 
 
 
156
  "eos_token": "</s>",
157
+ "extra_ids": 0,
158
  "extra_special_tokens": {},
159
+ "keep_accents": true,
160
+ "legacy": false,
161
  "mask_token": "<mask>",
162
+ "model_max_length": 1000000000000000019884624838656,
 
 
163
  "pad_token": "<pad>",
164
+ "padding_side": "right",
165
+ "sep_token": "<sep>",
166
+ "sp_model_kwargs": {},
167
+ "spaces_between_special_tokens": false,
168
+ "tokenizer_class": "LlamaTokenizer",
169
+ "unk_token": "<unk>",
170
+ "use_default_system_prompt": false
171
  }