nono1224 commited on
Commit
bbf996b
·
verified ·
1 Parent(s): d0a7659

Upload tokenizer

Browse files
sentencepiece.bpe.model CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d8b73a5e054936c920cf5b7d1ec21ce9c281977078269963beb821c6c86fbff7
3
- size 841889
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cfc8146abe2a0488e9e2a0c56de7952f7c11ab059eca145a0a727afce0db2865
3
+ size 5069051
special_tokens_map.json CHANGED
@@ -1,95 +1,15 @@
1
  {
2
- "additional_special_tokens": [
3
- {
4
- "content": "<ent>",
5
- "lstrip": false,
6
- "normalized": true,
7
- "rstrip": false,
8
- "single_word": false
9
- },
10
- {
11
- "content": "<ent2>",
12
- "lstrip": false,
13
- "normalized": true,
14
- "rstrip": false,
15
- "single_word": false
16
- },
17
- {
18
- "content": "<ent>",
19
- "lstrip": false,
20
- "normalized": false,
21
- "rstrip": false,
22
- "single_word": false
23
- },
24
- {
25
- "content": "<ent2>",
26
- "lstrip": false,
27
- "normalized": false,
28
- "rstrip": false,
29
- "single_word": false
30
- },
31
- {
32
- "content": "<ent>",
33
- "lstrip": false,
34
- "normalized": true,
35
- "rstrip": false,
36
- "single_word": false
37
- },
38
- {
39
- "content": "<ent2>",
40
- "lstrip": false,
41
- "normalized": true,
42
- "rstrip": false,
43
- "single_word": false
44
- }
45
- ],
46
- "bos_token": {
47
- "content": "<s>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false
52
- },
53
- "cls_token": {
54
- "content": "<s>",
55
- "lstrip": false,
56
- "normalized": false,
57
- "rstrip": false,
58
- "single_word": false
59
- },
60
- "eos_token": {
61
- "content": "</s>",
62
- "lstrip": false,
63
- "normalized": false,
64
- "rstrip": false,
65
- "single_word": false
66
- },
67
  "mask_token": {
68
  "content": "<mask>",
69
  "lstrip": true,
70
- "normalized": true,
71
- "rstrip": false,
72
- "single_word": false
73
- },
74
- "pad_token": {
75
- "content": "<pad>",
76
- "lstrip": false,
77
  "normalized": false,
78
  "rstrip": false,
79
  "single_word": false
80
  },
81
- "sep_token": {
82
- "content": "</s>",
83
- "lstrip": false,
84
- "normalized": false,
85
- "rstrip": false,
86
- "single_word": false
87
- },
88
- "unk_token": {
89
- "content": "<unk>",
90
- "lstrip": false,
91
- "normalized": false,
92
- "rstrip": false,
93
- "single_word": false
94
- }
95
  }
 
1
  {
2
+ "bos_token": "<s>",
3
+ "cls_token": "<s>",
4
+ "eos_token": "</s>",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  "mask_token": {
6
  "content": "<mask>",
7
  "lstrip": true,
 
 
 
 
 
 
 
8
  "normalized": false,
9
  "rstrip": false,
10
  "single_word": false
11
  },
12
+ "pad_token": "<pad>",
13
+ "sep_token": "</s>",
14
+ "unk_token": "<unk>"
 
 
 
 
 
 
 
 
 
 
 
15
  }
tokenizer_config.json CHANGED
@@ -32,74 +32,24 @@
32
  "single_word": false,
33
  "special": true
34
  },
35
- "32769": {
36
  "content": "<mask>",
37
  "lstrip": true,
38
- "normalized": true,
39
- "rstrip": false,
40
- "single_word": false,
41
- "special": true
42
- },
43
- "32770": {
44
- "content": "<ent>",
45
- "lstrip": false,
46
- "normalized": true,
47
- "rstrip": false,
48
- "single_word": false,
49
- "special": true
50
- },
51
- "32771": {
52
- "content": "<ent2>",
53
- "lstrip": false,
54
- "normalized": true,
55
  "rstrip": false,
56
  "single_word": false,
57
  "special": true
58
  }
59
  },
60
- "additional_special_tokens": [
61
- "<ent>",
62
- "<ent2>",
63
- "<ent>",
64
- "<ent2>",
65
- "<ent>",
66
- "<ent2>"
67
- ],
68
  "bos_token": "<s>",
69
  "clean_up_tokenization_spaces": false,
70
  "cls_token": "<s>",
71
- "entity_mask2_token": "[MASK2]",
72
- "entity_mask_token": "[MASK]",
73
- "entity_pad_token": "[PAD]",
74
- "entity_token_1": {
75
- "__type": "AddedToken",
76
- "content": "<ent>",
77
- "lstrip": false,
78
- "normalized": true,
79
- "rstrip": false,
80
- "single_word": false,
81
- "special": false
82
- },
83
- "entity_token_2": {
84
- "__type": "AddedToken",
85
- "content": "<ent2>",
86
- "lstrip": false,
87
- "normalized": true,
88
- "rstrip": false,
89
- "single_word": false,
90
- "special": false
91
- },
92
- "entity_unk_token": "[UNK]",
93
  "eos_token": "</s>",
94
  "extra_special_tokens": {},
95
  "mask_token": "<mask>",
96
- "max_entity_length": 32,
97
- "max_mention_length": 30,
98
  "model_max_length": 512,
99
  "pad_token": "<pad>",
100
  "sep_token": "</s>",
101
- "sp_model_kwargs": {},
102
- "task": null,
103
- "tokenizer_class": "MLukeTokenizer",
104
  "unk_token": "<unk>"
105
  }
 
32
  "single_word": false,
33
  "special": true
34
  },
35
+ "250001": {
36
  "content": "<mask>",
37
  "lstrip": true,
38
+ "normalized": false,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
39
  "rstrip": false,
40
  "single_word": false,
41
  "special": true
42
  }
43
  },
 
 
 
 
 
 
 
 
44
  "bos_token": "<s>",
45
  "clean_up_tokenization_spaces": false,
46
  "cls_token": "<s>",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
47
  "eos_token": "</s>",
48
  "extra_special_tokens": {},
49
  "mask_token": "<mask>",
 
 
50
  "model_max_length": 512,
51
  "pad_token": "<pad>",
52
  "sep_token": "</s>",
53
+ "tokenizer_class": "XLMRobertaTokenizer",
 
 
54
  "unk_token": "<unk>"
55
  }