windowsartes commited on
Commit
53b50a7
·
verified ·
1 Parent(s): 2bd4943

Upload tokenizer

Browse files
added_tokens.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "</s>": 912,
3
+ "<s>": 911
4
+ }
special_tokens_map.json CHANGED
@@ -1,5 +1,7 @@
1
  {
 
2
  "cls_token": "[CLS]",
 
3
  "mask_token": "[MASK]",
4
  "pad_token": "[PAD]",
5
  "sep_token": "[SEP]",
 
1
  {
2
+ "bos_token": "<s>",
3
  "cls_token": "[CLS]",
4
+ "eos_token": "</s>",
5
  "mask_token": "[MASK]",
6
  "pad_token": "[PAD]",
7
  "sep_token": "[SEP]",
tokenizer_config.json CHANGED
@@ -39,12 +39,30 @@
39
  "rstrip": false,
40
  "single_word": false,
41
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
42
  }
43
  },
 
44
  "clean_up_tokenization_spaces": true,
45
  "cls_token": "[CLS]",
46
  "do_basic_tokenize": true,
47
  "do_lower_case": false,
 
48
  "mask_token": "[MASK]",
49
  "model_max_length": 1000000000000000019884624838656,
50
  "never_split": null,
@@ -52,6 +70,6 @@
52
  "sep_token": "[SEP]",
53
  "strip_accents": null,
54
  "tokenize_chinese_chars": true,
55
- "tokenizer_class": "MobileBertTokenizer",
56
  "unk_token": "[UNK]"
57
  }
 
39
  "rstrip": false,
40
  "single_word": false,
41
  "special": true
42
+ },
43
+ "911": {
44
+ "content": "<s>",
45
+ "lstrip": false,
46
+ "normalized": false,
47
+ "rstrip": false,
48
+ "single_word": false,
49
+ "special": true
50
+ },
51
+ "912": {
52
+ "content": "</s>",
53
+ "lstrip": false,
54
+ "normalized": false,
55
+ "rstrip": false,
56
+ "single_word": false,
57
+ "special": true
58
  }
59
  },
60
+ "bos_token": "<s>",
61
  "clean_up_tokenization_spaces": true,
62
  "cls_token": "[CLS]",
63
  "do_basic_tokenize": true,
64
  "do_lower_case": false,
65
+ "eos_token": "</s>",
66
  "mask_token": "[MASK]",
67
  "model_max_length": 1000000000000000019884624838656,
68
  "never_split": null,
 
70
  "sep_token": "[SEP]",
71
  "strip_accents": null,
72
  "tokenize_chinese_chars": true,
73
+ "tokenizer_class": "FunnelTokenizer",
74
  "unk_token": "[UNK]"
75
  }