sarangahebbar commited on
Commit
79d3eae
·
1 Parent(s): 6de89f1

Training done

Browse files
added_tokens.json CHANGED
@@ -1,14 +1,19 @@
1
  {
2
- "</s_1Pcode>": 57526,
3
- "</s_1Tcode>": 57528,
4
- "</s_Dcode>": 57532,
5
- "</s_Qcode>": 57530,
6
- "<s_1Pcode>": 57525,
7
- "<s_1Tcode>": 57527,
8
- "<s_Dcode>": 57531,
9
- "<s_Qcode>": 57529,
 
 
 
 
10
  "<s_iitcdip>": 57523,
11
  "<s_johnson_electric>": 57533,
12
  "<s_synthdog>": 57524,
13
- "<sep/>": 57522
 
14
  }
 
1
  {
2
+ "</s>": 2,
3
+ "</s_1Pcode>": 57532,
4
+ "</s_1Tcode>": 57530,
5
+ "</s_Dcode>": 57528,
6
+ "</s_Qcode>": 57526,
7
+ "<mask>": 57521,
8
+ "<pad>": 1,
9
+ "<s>": 0,
10
+ "<s_1Pcode>": 57531,
11
+ "<s_1Tcode>": 57529,
12
+ "<s_Dcode>": 57527,
13
+ "<s_Qcode>": 57525,
14
  "<s_iitcdip>": 57523,
15
  "<s_johnson_electric>": 57533,
16
  "<s_synthdog>": 57524,
17
+ "<sep/>": 57522,
18
+ "<unk>": 3
19
  }
special_tokens_map.json CHANGED
@@ -6,13 +6,7 @@
6
  "bos_token": "<s>",
7
  "cls_token": "<s>",
8
  "eos_token": "</s>",
9
- "mask_token": {
10
- "content": "<mask>",
11
- "lstrip": true,
12
- "normalized": true,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
  "pad_token": "<pad>",
17
  "sep_token": "</s>",
18
  "unk_token": "<unk>"
 
6
  "bos_token": "<s>",
7
  "cls_token": "<s>",
8
  "eos_token": "</s>",
9
+ "mask_token": "<mask>",
 
 
 
 
 
 
10
  "pad_token": "<pad>",
11
  "sep_token": "</s>",
12
  "unk_token": "<unk>"
tokenizer.json CHANGED
@@ -75,8 +75,8 @@
75
  "id": 57523,
76
  "content": "<s_iitcdip>",
77
  "single_word": false,
78
- "lstrip": false,
79
- "rstrip": false,
80
  "normalized": false,
81
  "special": true
82
  },
@@ -84,14 +84,14 @@
84
  "id": 57524,
85
  "content": "<s_synthdog>",
86
  "single_word": false,
87
- "lstrip": false,
88
- "rstrip": false,
89
  "normalized": false,
90
  "special": true
91
  },
92
  {
93
  "id": 57525,
94
- "content": "<s_1Pcode>",
95
  "single_word": false,
96
  "lstrip": false,
97
  "rstrip": false,
@@ -100,7 +100,7 @@
100
  },
101
  {
102
  "id": 57526,
103
- "content": "</s_1Pcode>",
104
  "single_word": false,
105
  "lstrip": false,
106
  "rstrip": false,
@@ -109,7 +109,7 @@
109
  },
110
  {
111
  "id": 57527,
112
- "content": "<s_1Tcode>",
113
  "single_word": false,
114
  "lstrip": false,
115
  "rstrip": false,
@@ -118,7 +118,7 @@
118
  },
119
  {
120
  "id": 57528,
121
- "content": "</s_1Tcode>",
122
  "single_word": false,
123
  "lstrip": false,
124
  "rstrip": false,
@@ -127,7 +127,7 @@
127
  },
128
  {
129
  "id": 57529,
130
- "content": "<s_Qcode>",
131
  "single_word": false,
132
  "lstrip": false,
133
  "rstrip": false,
@@ -136,7 +136,7 @@
136
  },
137
  {
138
  "id": 57530,
139
- "content": "</s_Qcode>",
140
  "single_word": false,
141
  "lstrip": false,
142
  "rstrip": false,
@@ -145,7 +145,7 @@
145
  },
146
  {
147
  "id": 57531,
148
- "content": "<s_Dcode>",
149
  "single_word": false,
150
  "lstrip": false,
151
  "rstrip": false,
@@ -154,7 +154,7 @@
154
  },
155
  {
156
  "id": 57532,
157
- "content": "</s_Dcode>",
158
  "single_word": false,
159
  "lstrip": false,
160
  "rstrip": false,
@@ -230370,6 +230370,7 @@
230370
  "<mask>",
230371
  0.0
230372
  ]
230373
- ]
 
230374
  }
230375
  }
 
75
  "id": 57523,
76
  "content": "<s_iitcdip>",
77
  "single_word": false,
78
+ "lstrip": true,
79
+ "rstrip": true,
80
  "normalized": false,
81
  "special": true
82
  },
 
84
  "id": 57524,
85
  "content": "<s_synthdog>",
86
  "single_word": false,
87
+ "lstrip": true,
88
+ "rstrip": true,
89
  "normalized": false,
90
  "special": true
91
  },
92
  {
93
  "id": 57525,
94
+ "content": "<s_Qcode>",
95
  "single_word": false,
96
  "lstrip": false,
97
  "rstrip": false,
 
100
  },
101
  {
102
  "id": 57526,
103
+ "content": "</s_Qcode>",
104
  "single_word": false,
105
  "lstrip": false,
106
  "rstrip": false,
 
109
  },
110
  {
111
  "id": 57527,
112
+ "content": "<s_Dcode>",
113
  "single_word": false,
114
  "lstrip": false,
115
  "rstrip": false,
 
118
  },
119
  {
120
  "id": 57528,
121
+ "content": "</s_Dcode>",
122
  "single_word": false,
123
  "lstrip": false,
124
  "rstrip": false,
 
127
  },
128
  {
129
  "id": 57529,
130
+ "content": "<s_1Tcode>",
131
  "single_word": false,
132
  "lstrip": false,
133
  "rstrip": false,
 
136
  },
137
  {
138
  "id": 57530,
139
+ "content": "</s_1Tcode>",
140
  "single_word": false,
141
  "lstrip": false,
142
  "rstrip": false,
 
145
  },
146
  {
147
  "id": 57531,
148
+ "content": "<s_1Pcode>",
149
  "single_word": false,
150
  "lstrip": false,
151
  "rstrip": false,
 
154
  },
155
  {
156
  "id": 57532,
157
+ "content": "</s_1Pcode>",
158
  "single_word": false,
159
  "lstrip": false,
160
  "rstrip": false,
 
230370
  "<mask>",
230371
  0.0
230372
  ]
230373
+ ],
230374
+ "byte_fallback": false
230375
  }
230376
  }
tokenizer_config.json CHANGED
@@ -1,16 +1,151 @@
1
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  "bos_token": "<s>",
3
  "clean_up_tokenization_spaces": true,
4
  "cls_token": "<s>",
5
  "eos_token": "</s>",
6
- "mask_token": {
7
- "__type": "AddedToken",
8
- "content": "<mask>",
9
- "lstrip": true,
10
- "normalized": true,
11
- "rstrip": false,
12
- "single_word": false
13
- },
14
  "model_max_length": 1000000000000000019884624838656,
15
  "pad_token": "<pad>",
16
  "processor_class": "DonutProcessor",
 
1
  {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "<s>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<pad>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "2": {
20
+ "content": "</s>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "3": {
28
+ "content": "<unk>",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "57521": {
36
+ "content": "<mask>",
37
+ "lstrip": true,
38
+ "normalized": true,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ },
43
+ "57522": {
44
+ "content": "<sep/>",
45
+ "lstrip": false,
46
+ "normalized": true,
47
+ "rstrip": false,
48
+ "single_word": false,
49
+ "special": false
50
+ },
51
+ "57523": {
52
+ "content": "<s_iitcdip>",
53
+ "lstrip": true,
54
+ "normalized": false,
55
+ "rstrip": true,
56
+ "single_word": false,
57
+ "special": true
58
+ },
59
+ "57524": {
60
+ "content": "<s_synthdog>",
61
+ "lstrip": true,
62
+ "normalized": false,
63
+ "rstrip": true,
64
+ "single_word": false,
65
+ "special": true
66
+ },
67
+ "57525": {
68
+ "content": "<s_Qcode>",
69
+ "lstrip": false,
70
+ "normalized": true,
71
+ "rstrip": false,
72
+ "single_word": false,
73
+ "special": false
74
+ },
75
+ "57526": {
76
+ "content": "</s_Qcode>",
77
+ "lstrip": false,
78
+ "normalized": true,
79
+ "rstrip": false,
80
+ "single_word": false,
81
+ "special": false
82
+ },
83
+ "57527": {
84
+ "content": "<s_Dcode>",
85
+ "lstrip": false,
86
+ "normalized": true,
87
+ "rstrip": false,
88
+ "single_word": false,
89
+ "special": false
90
+ },
91
+ "57528": {
92
+ "content": "</s_Dcode>",
93
+ "lstrip": false,
94
+ "normalized": true,
95
+ "rstrip": false,
96
+ "single_word": false,
97
+ "special": false
98
+ },
99
+ "57529": {
100
+ "content": "<s_1Tcode>",
101
+ "lstrip": false,
102
+ "normalized": true,
103
+ "rstrip": false,
104
+ "single_word": false,
105
+ "special": false
106
+ },
107
+ "57530": {
108
+ "content": "</s_1Tcode>",
109
+ "lstrip": false,
110
+ "normalized": true,
111
+ "rstrip": false,
112
+ "single_word": false,
113
+ "special": false
114
+ },
115
+ "57531": {
116
+ "content": "<s_1Pcode>",
117
+ "lstrip": false,
118
+ "normalized": true,
119
+ "rstrip": false,
120
+ "single_word": false,
121
+ "special": false
122
+ },
123
+ "57532": {
124
+ "content": "</s_1Pcode>",
125
+ "lstrip": false,
126
+ "normalized": true,
127
+ "rstrip": false,
128
+ "single_word": false,
129
+ "special": false
130
+ },
131
+ "57533": {
132
+ "content": "<s_johnson_electric>",
133
+ "lstrip": false,
134
+ "normalized": true,
135
+ "rstrip": false,
136
+ "single_word": false,
137
+ "special": false
138
+ }
139
+ },
140
+ "additional_special_tokens": [
141
+ "<s_iitcdip>",
142
+ "<s_synthdog>"
143
+ ],
144
  "bos_token": "<s>",
145
  "clean_up_tokenization_spaces": true,
146
  "cls_token": "<s>",
147
  "eos_token": "</s>",
148
+ "mask_token": "<mask>",
 
 
 
 
 
 
 
149
  "model_max_length": 1000000000000000019884624838656,
150
  "pad_token": "<pad>",
151
  "processor_class": "DonutProcessor",