Bingsu commited on
Commit
b075619
·
1 Parent(s): ec5c88c

Upload tokenizer

Browse files
special_tokens_map.json CHANGED
@@ -2,13 +2,7 @@
2
  "bos_token": "<s>",
3
  "cls_token": "<s>",
4
  "eos_token": "</s>",
5
- "mask_token": {
6
- "content": "<mask>",
7
- "lstrip": false,
8
- "normalized": false,
9
- "rstrip": false,
10
- "single_word": false
11
- },
12
  "pad_token": "<pad>",
13
  "sep_token": "</s>",
14
  "unk_token": "<unk>"
 
2
  "bos_token": "<s>",
3
  "cls_token": "<s>",
4
  "eos_token": "</s>",
5
+ "mask_token": "<mask>",
 
 
 
 
 
 
6
  "pad_token": "<pad>",
7
  "sep_token": "</s>",
8
  "unk_token": "<unk>"
spiece.model CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6c1ead4e1bb780e472f416568f5ffd6bcff1bbb28cd4e1aa0b8fbbb80da35b12
3
- size 1175992
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:50540e8103c6e358c241308c20c33e3785aa69af5ca1c294faea9e262276802f
3
+ size 1175148
tokenizer.json CHANGED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json CHANGED
@@ -7,18 +7,18 @@
7
  "mask_token": {
8
  "__type": "AddedToken",
9
  "content": "<mask>",
10
- "lstrip": false,
11
  "normalized": false,
12
  "rstrip": false,
13
  "single_word": false
14
  },
15
  "model_max_length": 512,
16
- "name_or_path": "Bingsu/test_albert_tokenizer",
17
  "pad_token": "<pad>",
18
  "remove_space": true,
19
  "sep_token": "</s>",
20
  "sp_model_kwargs": {},
21
- "special_tokens_map_file": "./special_tokens_map.json",
22
  "tokenizer_class": "AlbertTokenizer",
23
  "unk_token": "<unk>"
24
  }
 
7
  "mask_token": {
8
  "__type": "AddedToken",
9
  "content": "<mask>",
10
+ "lstrip": true,
11
  "normalized": false,
12
  "rstrip": false,
13
  "single_word": false
14
  },
15
  "model_max_length": 512,
16
+ "name_or_path": "albert_tokenizer_02",
17
  "pad_token": "<pad>",
18
  "remove_space": true,
19
  "sep_token": "</s>",
20
  "sp_model_kwargs": {},
21
+ "special_tokens_map_file": "albert_tokenizer_01\\special_tokens_map.json",
22
  "tokenizer_class": "AlbertTokenizer",
23
  "unk_token": "<unk>"
24
  }