junghwanjkim commited on
Commit
4aef030
·
1 Parent(s): a8cf191

Fix tokenizer by adding pad and mask tokens

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +14 -0
  2. tokenizer_config.json +2 -0
special_tokens_map.json CHANGED
@@ -12,5 +12,19 @@
12
  "normalized": false,
13
  "rstrip": false,
14
  "single_word": false
 
 
 
 
 
 
 
 
 
 
 
 
 
 
15
  }
16
  }
 
12
  "normalized": false,
13
  "rstrip": false,
14
  "single_word": false
15
+ },
16
+ "mask_token": {
17
+ "content": "<|reserved_special_token_247|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ },
23
+ "pad_token": {
24
+ "content": "<|reserved_special_token_247|>",
25
+ "lstrip": false,
26
+ "normalized": false,
27
+ "rstrip": false,
28
+ "single_word": false
29
  }
30
  }
tokenizer_config.json CHANGED
@@ -2053,10 +2053,12 @@
2053
  "clean_up_tokenization_spaces": true,
2054
  "eos_token": "<|end_of_text|>",
2055
  "extra_special_tokens": {},
 
2056
  "model_input_names": [
2057
  "input_ids",
2058
  "attention_mask"
2059
  ],
2060
  "model_max_length": 131072,
 
2061
  "tokenizer_class": "PreTrainedTokenizerFast"
2062
  }
 
2053
  "clean_up_tokenization_spaces": true,
2054
  "eos_token": "<|end_of_text|>",
2055
  "extra_special_tokens": {},
2056
+ "mask_token": "<|reserved_special_token_247|>",
2057
  "model_input_names": [
2058
  "input_ids",
2059
  "attention_mask"
2060
  ],
2061
  "model_max_length": 131072,
2062
+ "pad_token": "<|reserved_special_token_247|>",
2063
  "tokenizer_class": "PreTrainedTokenizerFast"
2064
  }