madhuHuggingface commited on
Commit
a1a4279
·
verified ·
1 Parent(s): e1e8df5

Training in progress, step 300, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8c028f24f3067cda391a4b121ed7be318959121a2b2bff13b405b186b535836b
3
  size 60785144
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2263d7e8e8a2477ab01c8bc96a02918025fb0e463a289e0e0702e0a1d8d0fa72
3
  size 60785144
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:40e2a02437b046b7f3e8828a69d32979b1eaba5754a3b88b88ae3f1206dfe2f7
3
- size 31148949
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:11146f0c3474ab69b618d52c0df5dea20430be51d33835af31f41fe6774d4488
3
+ size 31149205
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f1d565802a8e26c4e8a31328752b7a7fdc186d9401aa008e65697d0ad8c22e33
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7c800b778fa7e115e4c34de8529902de8b61c9a1b4bab3eb8295d06dafff030e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9a2b848792e7bd656024255a488792ad400422248c9712097291155b6784a7c9
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9a82671fb137361a0604603511b9d8eaa9e60fd0f97a0a1ff5cfdebbdb224ef7
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.8,
6
  "eval_steps": 500,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -148,6 +148,76 @@
148
  "learning_rate": 0.0001717676913675962,
149
  "loss": 0.0231,
150
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
151
  }
152
  ],
153
  "logging_steps": 10,
@@ -167,7 +237,7 @@
167
  "attributes": {}
168
  }
169
  },
170
- "total_flos": 639813475138560.0,
171
  "train_batch_size": 2,
172
  "trial_name": null,
173
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.2,
6
  "eval_steps": 500,
7
+ "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
148
  "learning_rate": 0.0001717676913675962,
149
  "loss": 0.0231,
150
  "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.84,
154
+ "grad_norm": 0.36980199813842773,
155
+ "learning_rate": 0.0001687052767223667,
156
+ "loss": 0.019,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.88,
161
+ "grad_norm": 0.9401270151138306,
162
+ "learning_rate": 0.00016551563572090854,
163
+ "loss": 0.0263,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.92,
168
+ "grad_norm": 0.18841078877449036,
169
+ "learning_rate": 0.00016220467484408677,
170
+ "loss": 0.0295,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.96,
175
+ "grad_norm": 0.18036462366580963,
176
+ "learning_rate": 0.00015877852522924732,
177
+ "loss": 0.0255,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 1.0,
182
+ "grad_norm": 0.462505042552948,
183
+ "learning_rate": 0.000155243531316762,
184
+ "loss": 0.022,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 1.04,
189
+ "grad_norm": 0.3836606442928314,
190
+ "learning_rate": 0.00015160623910158528,
191
+ "loss": 0.02,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 1.08,
196
+ "grad_norm": 0.33333465456962585,
197
+ "learning_rate": 0.00014787338401157885,
198
+ "loss": 0.0247,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 1.12,
203
+ "grad_norm": 0.25382208824157715,
204
+ "learning_rate": 0.0001440518784350495,
205
+ "loss": 0.0102,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 1.16,
210
+ "grad_norm": 0.5277785062789917,
211
+ "learning_rate": 0.0001401487989205973,
212
+ "loss": 0.0226,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 1.2,
217
+ "grad_norm": 0.14403806626796722,
218
+ "learning_rate": 0.00013617137307297676,
219
+ "loss": 0.0175,
220
+ "step": 300
221
  }
222
  ],
223
  "logging_steps": 10,
 
237
  "attributes": {}
238
  }
239
  },
240
+ "total_flos": 960635083461120.0,
241
  "train_batch_size": 2,
242
  "trial_name": null,
243
  "trial_params": null