infi commited on
Commit
141867f
·
verified ·
1 Parent(s): 0b2d921

Upload tokenizer

Browse files
special_tokens_map.json CHANGED
@@ -1,6 +1,30 @@
1
  {
2
- "bos_token": "<|startbox|>",
3
- "eos_token": "<|endbox|>",
4
- "pad_token": "<pad>",
5
- "unk_token": "<|unkbox|>"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6
  }
 
1
  {
2
+ "bos_token": {
3
+ "content": "<|startbox|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|endbox|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<pad>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ },
23
+ "unk_token": {
24
+ "content": "<|unkbox|>",
25
+ "lstrip": false,
26
+ "normalized": false,
27
+ "rstrip": false,
28
+ "single_word": false
29
+ }
30
  }
tokenizer.json CHANGED
@@ -54,8 +54,8 @@
54
  "single_word": false,
55
  "lstrip": false,
56
  "rstrip": false,
57
- "normalized": false,
58
- "special": true
59
  },
60
  {
61
  "id": 34005,
@@ -63,8 +63,8 @@
63
  "single_word": false,
64
  "lstrip": false,
65
  "rstrip": false,
66
- "normalized": false,
67
- "special": true
68
  },
69
  {
70
  "id": 34006,
 
54
  "single_word": false,
55
  "lstrip": false,
56
  "rstrip": false,
57
+ "normalized": true,
58
+ "special": false
59
  },
60
  {
61
  "id": 34005,
 
63
  "single_word": false,
64
  "lstrip": false,
65
  "rstrip": false,
66
+ "normalized": true,
67
+ "special": false
68
  },
69
  {
70
  "id": 34006,
tokenizer_config.json CHANGED
@@ -44,18 +44,18 @@
44
  "34004": {
45
  "content": "<|startbox|>",
46
  "lstrip": false,
47
- "normalized": false,
48
  "rstrip": false,
49
  "single_word": false,
50
- "special": true
51
  },
52
  "34005": {
53
  "content": "<|endbox|>",
54
  "lstrip": false,
55
- "normalized": false,
56
  "rstrip": false,
57
  "single_word": false,
58
- "special": true
59
  },
60
  "34006": {
61
  "content": "<|para|>",
 
44
  "34004": {
45
  "content": "<|startbox|>",
46
  "lstrip": false,
47
+ "normalized": true,
48
  "rstrip": false,
49
  "single_word": false,
50
+ "special": false
51
  },
52
  "34005": {
53
  "content": "<|endbox|>",
54
  "lstrip": false,
55
+ "normalized": true,
56
  "rstrip": false,
57
  "single_word": false,
58
+ "special": false
59
  },
60
  "34006": {
61
  "content": "<|para|>",