CodeIsAbstract commited on
Commit
c286dd7
·
verified ·
1 Parent(s): 909c125

Training in progress, step 4000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:70a32488c42d94bc68537838fd59bace50bad9a491961be86f3738a6910755e4
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d9744fca00a76a2b63dac0145112528d105510001811540108c40b285279761
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:47fd29bde813c33926e0c9c10cf9775d231d7930b7aa4abd47133c29f24ed7c5
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eda6f94a6ddf235700eba4f7ef325af055c1617aa52880aa4490a4d5390ce4cb
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a2a4adad934312ea001cd84349883cb6c15698df9bb6c090af6bf88181fb06b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:666c4ffc6f56dc33b9a78682d8f70b6048cb030ff6769652c67983bc0c55e9bb
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:649e8cc37e14e4911064da2b8cb9ef2ade87f6e78d1185d4b8e5b14453b1c83d
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b91a82f25eb2ba5e7ada463b6840a878d34219b311613ef51dfe56bd048532b9
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f6403efa0c30bbacae871a7df874ec4bbdb5b528b6d86d3b5609a837c3eb80b9
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae6fc782ac05ca9ced39cd4e6262e90acfd0088a765c64bc05286563954690b6
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fedd614c5b3fb7e22cbb42e3d69fcde0d7a32f2e678c37f7c99bb5a167b975e8
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cb48f272d3916d4dba504ddc3932b95ab25bac7736fe8342e16dc0cbc2a1adf
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f72141908b3038f5d527be0a4afe331900fd2cd98b763078b556a00ecca0eca1
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b090f294678dd201862ecf44cf91ce818a27b404a0eff4bef3b64dec57a7ecb8
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c958787b74a3aefeefe565a04ee3471c563655a4aa29b37f82c1b78d0c1b5a6c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:93564d987336a3da8adbded70bd5490c95a0d271364aab176adc5b13e849acb5
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8a198a4edc5b0a0c0735b02620e42eae859d7b0a942bf7c1a21b08b1885eff98
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8dd028df69482d761eb10c87e2921b635a7522d6e0534c883e77e6c5389b6ff4
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1403f69e2c99c4fcaf2999db32c4368d388bb9ac696855db926ab4209f647997
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da0f84dd27aa055253dd988727f836725b299f77ef0700780966bcc8184785ad
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:19317e67cea234fe3e615230a11e75ee190fdec27b0f95904042181fcbf16742
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f9a8388b6b51fab4a30d39efc19f0e1fc13cb83c585b1bd7d347d801a2ccc8e7
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -166,6 +166,164 @@
166
  "eval_samples_per_second": 17.231,
167
  "eval_steps_per_second": 0.551,
168
  "step": 2000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
169
  }
170
  ],
171
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
166
  "eval_samples_per_second": 17.231,
167
  "eval_steps_per_second": 0.551,
168
  "step": 2000
169
+ },
170
+ {
171
+ "epoch": 0.084,
172
+ "grad_norm": 0.5317835807800293,
173
+ "learning_rate": 0.0039314280243790255,
174
+ "loss": 51.5766015625,
175
+ "step": 2100
176
+ },
177
+ {
178
+ "epoch": 0.088,
179
+ "grad_norm": 0.7677087187767029,
180
+ "learning_rate": 0.003924748298765388,
181
+ "loss": 51.459482421875,
182
+ "step": 2200
183
+ },
184
+ {
185
+ "epoch": 0.092,
186
+ "grad_norm": 0.8987520933151245,
187
+ "learning_rate": 0.003917764389788164,
188
+ "loss": 51.2251513671875,
189
+ "step": 2300
190
+ },
191
+ {
192
+ "epoch": 0.096,
193
+ "grad_norm": 0.8509472608566284,
194
+ "learning_rate": 0.003910477401170333,
195
+ "loss": 51.086318359375,
196
+ "step": 2400
197
+ },
198
+ {
199
+ "epoch": 0.1,
200
+ "grad_norm": 1.0043303966522217,
201
+ "learning_rate": 0.003902888484532972,
202
+ "loss": 51.013447265625,
203
+ "step": 2500
204
+ },
205
+ {
206
+ "epoch": 0.104,
207
+ "grad_norm": 0.6577029228210449,
208
+ "learning_rate": 0.0038949988392132555,
209
+ "loss": 50.6555419921875,
210
+ "step": 2600
211
+ },
212
+ {
213
+ "epoch": 0.108,
214
+ "grad_norm": 0.9883126020431519,
215
+ "learning_rate": 0.0038868097120749187,
216
+ "loss": 50.6927490234375,
217
+ "step": 2700
218
+ },
219
+ {
220
+ "epoch": 0.112,
221
+ "grad_norm": 1.0359454154968262,
222
+ "learning_rate": 0.003878322397311201,
223
+ "loss": 50.8522021484375,
224
+ "step": 2800
225
+ },
226
+ {
227
+ "epoch": 0.116,
228
+ "grad_norm": 0.8272144198417664,
229
+ "learning_rate": 0.0038695382362403186,
230
+ "loss": 50.349677734375,
231
+ "step": 2900
232
+ },
233
+ {
234
+ "epoch": 0.12,
235
+ "grad_norm": 0.7374610304832458,
236
+ "learning_rate": 0.0038604586170934807,
237
+ "loss": 49.932939453125,
238
+ "step": 3000
239
+ },
240
+ {
241
+ "epoch": 0.12,
242
+ "eval_accuracy": 0.16219354838709676,
243
+ "eval_loss": 50.10387420654297,
244
+ "eval_runtime": 7.1742,
245
+ "eval_samples_per_second": 17.424,
246
+ "eval_steps_per_second": 0.558,
247
+ "step": 3000
248
+ },
249
+ {
250
+ "epoch": 0.124,
251
+ "grad_norm": 1.0479457378387451,
252
+ "learning_rate": 0.0038510849747955015,
253
+ "loss": 49.958212890625,
254
+ "step": 3100
255
+ },
256
+ {
257
+ "epoch": 0.128,
258
+ "grad_norm": 1.4167745113372803,
259
+ "learning_rate": 0.0038414187907380216,
260
+ "loss": 49.9526025390625,
261
+ "step": 3200
262
+ },
263
+ {
264
+ "epoch": 0.132,
265
+ "grad_norm": 1.264487385749817,
266
+ "learning_rate": 0.0038314615925453986,
267
+ "loss": 49.983779296875,
268
+ "step": 3300
269
+ },
270
+ {
271
+ "epoch": 0.136,
272
+ "grad_norm": 1.5543415546417236,
273
+ "learning_rate": 0.003821214953833277,
274
+ "loss": 49.699599609375,
275
+ "step": 3400
276
+ },
277
+ {
278
+ "epoch": 0.14,
279
+ "grad_norm": 1.4869952201843262,
280
+ "learning_rate": 0.0038106804939599037,
281
+ "loss": 49.46021484375,
282
+ "step": 3500
283
+ },
284
+ {
285
+ "epoch": 0.144,
286
+ "grad_norm": 1.2286678552627563,
287
+ "learning_rate": 0.003799859877770204,
288
+ "loss": 49.3774853515625,
289
+ "step": 3600
290
+ },
291
+ {
292
+ "epoch": 0.148,
293
+ "grad_norm": 2.0684049129486084,
294
+ "learning_rate": 0.003788754815332674,
295
+ "loss": 48.850126953125,
296
+ "step": 3700
297
+ },
298
+ {
299
+ "epoch": 0.152,
300
+ "grad_norm": 2.0739598274230957,
301
+ "learning_rate": 0.003777367061669124,
302
+ "loss": 48.7242919921875,
303
+ "step": 3800
304
+ },
305
+ {
306
+ "epoch": 0.156,
307
+ "grad_norm": 2.031198740005493,
308
+ "learning_rate": 0.0037656984164773206,
309
+ "loss": 48.9626513671875,
310
+ "step": 3900
311
+ },
312
+ {
313
+ "epoch": 0.16,
314
+ "grad_norm": 1.9467569589614868,
315
+ "learning_rate": 0.003753750723846564,
316
+ "loss": 48.71037109375,
317
+ "step": 4000
318
+ },
319
+ {
320
+ "epoch": 0.16,
321
+ "eval_accuracy": 0.1748435972629521,
322
+ "eval_loss": 48.56401824951172,
323
+ "eval_runtime": 7.0573,
324
+ "eval_samples_per_second": 17.712,
325
+ "eval_steps_per_second": 0.567,
326
+ "step": 4000
327
  }
328
  ],
329
  "logging_steps": 100,