CodeIsAbstract commited on
Commit
db57107
·
verified ·
1 Parent(s): a06358f

Training in progress, step 2500, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:482d033c346824d7aae735e0a30cb3e8c41e0c5f81d53f22ebe44258c3425d26
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15ff0e61e5052b90c67c4ed5d482def1c575e4af82a097620b99d9915b81b731
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2d011aeead383547fdef45678632afedc62a2962158c32e8cb4ce8e90d692ad0
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:55bf4bc111af8de23f7f2ce21aaa380a72f84bc8970796d140e42699e863f8e3
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6b1991cc99e86c95e5ef139f6421894a5757fd1b892dab95840668814509c4cb
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:976b136d7bb1a4d07310de167b239d43246911b3a7eb91fae3d795edd7156199
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:28158218281a80eaa5c1e06ab030551152161ff67e14ea3969da4953f1446edf
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12e3075e882ab3c95707e237853acd00dab10b273a7aca257247da539b166367
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a8532a948bb3a20cc440fe262d4aed275beeed50de257f08279070a99339d577
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c2360e8a0ee66690e03398af5b14c36fa14674949dfec6821223c9dd5b85b837
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:65cb6f2cd04253d39e6e8a66ab2ad12b1df978d3afeb82f6ffaa53f610295f4e
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:79f8b5fd8d8bcfc44dbd154c14b025870938d6c00c296f2ca879f51e298de13c
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:133b88ea6938699834d9214a043b01ec9753babf93f7342c4d965c731f9ca3fa
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:240a205a89bf5f59ded6ca0762e1b0916c54445eb25a1cc46eb8219e9a64c5a5
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2569b59cbd12a55fc23ea1de67cfa40d8ce399abd6061b317cf5cd83d9f589ad
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4477cc6ec8df9bbaad382fc5c5b0c63907c2ba994deea060adb01d78698e02b
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5ca0a826dbc599964473aabc25a5eff43d974a031a177512095eff6c593a4a29
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7888db6be40cddea55e6d51f3d215731350768df7a4f6a4212815955b5eb66e3
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:799cf010f91f5055e7bac3df01758db699f980ad15cd9440967d7e87ae060f04
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d71128673fbb9afdc89dcbb28747a722c01732d4d19593a193c081af2137e22
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32e12c26fc7187e12c6806b82566389f8bad1946f6607f89dddf407ce49f8742
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a1dc80710df0faed46c8038fb1842bb2490f6526d6a92e28a38baa9e8f7ef4ed
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6666666666666666,
6
  "eval_steps": 250,
7
- "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -360,6 +360,94 @@
360
  "eval_samples_per_second": 15.161,
361
  "eval_steps_per_second": 0.485,
362
  "step": 2000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
363
  }
364
  ],
365
  "logging_steps": 50,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.8333333333333334,
6
  "eval_steps": 250,
7
+ "global_step": 2500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
360
  "eval_samples_per_second": 15.161,
361
  "eval_steps_per_second": 0.485,
362
  "step": 2000
363
+ },
364
+ {
365
+ "epoch": 0.6833333333333333,
366
+ "grad_norm": 480.41583251953125,
367
+ "learning_rate": 0.00010019098696491591,
368
+ "loss": 1744.609375,
369
+ "step": 2050
370
+ },
371
+ {
372
+ "epoch": 0.7,
373
+ "grad_norm": 489.03533935546875,
374
+ "learning_rate": 9.079499873138675e-05,
375
+ "loss": 1743.00265625,
376
+ "step": 2100
377
+ },
378
+ {
379
+ "epoch": 0.7166666666666667,
380
+ "grad_norm": 281.31231689453125,
381
+ "learning_rate": 8.173066249750974e-05,
382
+ "loss": 1741.88375,
383
+ "step": 2150
384
+ },
385
+ {
386
+ "epoch": 0.7333333333333333,
387
+ "grad_norm": 367.38116455078125,
388
+ "learning_rate": 7.302550635451295e-05,
389
+ "loss": 1741.17609375,
390
+ "step": 2200
391
+ },
392
+ {
393
+ "epoch": 0.75,
394
+ "grad_norm": 373.6410827636719,
395
+ "learning_rate": 6.470596757549348e-05,
396
+ "loss": 1740.87671875,
397
+ "step": 2250
398
+ },
399
+ {
400
+ "epoch": 0.75,
401
+ "eval_accuracy": 0.17889638318670575,
402
+ "eval_loss": 57.56399917602539,
403
+ "eval_runtime": 8.5942,
404
+ "eval_samples_per_second": 14.545,
405
+ "eval_steps_per_second": 0.465,
406
+ "step": 2250
407
+ },
408
+ {
409
+ "epoch": 0.7666666666666667,
410
+ "grad_norm": 378.64324951171875,
411
+ "learning_rate": 5.6797312326288e-05,
412
+ "loss": 1741.790625,
413
+ "step": 2300
414
+ },
415
+ {
416
+ "epoch": 0.7833333333333333,
417
+ "grad_norm": 399.52471923828125,
418
+ "learning_rate": 4.932355893295748e-05,
419
+ "loss": 1739.6509375,
420
+ "step": 2350
421
+ },
422
+ {
423
+ "epoch": 0.8,
424
+ "grad_norm": 448.8892822265625,
425
+ "learning_rate": 4.230740493892038e-05,
426
+ "loss": 1739.64484375,
427
+ "step": 2400
428
+ },
429
+ {
430
+ "epoch": 0.8166666666666667,
431
+ "grad_norm": 246.4326171875,
432
+ "learning_rate": 3.57701581732592e-05,
433
+ "loss": 1738.5784375,
434
+ "step": 2450
435
+ },
436
+ {
437
+ "epoch": 0.8333333333333334,
438
+ "grad_norm": 420.97027587890625,
439
+ "learning_rate": 2.973167203954399e-05,
440
+ "loss": 1738.51265625,
441
+ "step": 2500
442
+ },
443
+ {
444
+ "epoch": 0.8333333333333334,
445
+ "eval_accuracy": 0.17922873900293254,
446
+ "eval_loss": 57.48527526855469,
447
+ "eval_runtime": 6.8955,
448
+ "eval_samples_per_second": 18.128,
449
+ "eval_steps_per_second": 0.58,
450
+ "step": 2500
451
  }
452
  ],
453
  "logging_steps": 50,