CodeIsAbstract commited on
Commit
0a4b086
·
verified ·
1 Parent(s): 1499a75

Training in progress, step 3000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15ff0e61e5052b90c67c4ed5d482def1c575e4af82a097620b99d9915b81b731
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3dbc2fb0a6c5f2970cd53453e764a649ac6607abbf79cfa4749975093698eaf6
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:55bf4bc111af8de23f7f2ce21aaa380a72f84bc8970796d140e42699e863f8e3
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e6a587f037c6fe6673f8514fdde6c05e7562af204e86df7c654d7dae4b255f1f
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:976b136d7bb1a4d07310de167b239d43246911b3a7eb91fae3d795edd7156199
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:218b503b380a0e1262fb31f7e6ad9fa8d4cf2f109ff7eda4aa5eed8f7aef324c
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12e3075e882ab3c95707e237853acd00dab10b273a7aca257247da539b166367
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:97a5947f809acf4e5bfead2eebc83edaff5624e9d14837c9fa503560db15c663
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c2360e8a0ee66690e03398af5b14c36fa14674949dfec6821223c9dd5b85b837
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c7885e48aad56e6bb142603863a10c62d22f7a2518b4baba80b8d015b666e48
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:79f8b5fd8d8bcfc44dbd154c14b025870938d6c00c296f2ca879f51e298de13c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c7610f49093646d2e22293764225fe9b8113bf86c1e3e290ea735ac5e1bd6ae7
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:240a205a89bf5f59ded6ca0762e1b0916c54445eb25a1cc46eb8219e9a64c5a5
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9ca1bdf5df09ff68c66a2f5cdc44a65ad89d1674b79918065e22e92ce4210da0
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f4477cc6ec8df9bbaad382fc5c5b0c63907c2ba994deea060adb01d78698e02b
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:112e61e6b9d44ab507241b2ef32d6cfa5970d93605a3430d819c9d6544dddd34
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7888db6be40cddea55e6d51f3d215731350768df7a4f6a4212815955b5eb66e3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2dc8f42a650b40b2f2252aaa98d697c4c891fd4cd9d7cc47c1c2b9764ba09934
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8d71128673fbb9afdc89dcbb28747a722c01732d4d19593a193c081af2137e22
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f8e75193fe6e093d812dbc12982db192f141515d07b4fe6ba85ffc695a84fb99
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a1dc80710df0faed46c8038fb1842bb2490f6526d6a92e28a38baa9e8f7ef4ed
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b36204b8c6666543761ec77ac720f6c626682baa52cec8cf26572629d35d9b79
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.8333333333333334,
6
  "eval_steps": 250,
7
- "global_step": 2500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -448,6 +448,94 @@
448
  "eval_samples_per_second": 18.128,
449
  "eval_steps_per_second": 0.58,
450
  "step": 2500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
451
  }
452
  ],
453
  "logging_steps": 50,
@@ -462,7 +550,7 @@
462
  "should_evaluate": false,
463
  "should_log": false,
464
  "should_save": true,
465
- "should_training_stop": false
466
  },
467
  "attributes": {}
468
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 250,
7
+ "global_step": 3000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
448
  "eval_samples_per_second": 18.128,
449
  "eval_steps_per_second": 0.58,
450
  "step": 2500
451
+ },
452
+ {
453
+ "epoch": 0.85,
454
+ "grad_norm": 455.0086975097656,
455
+ "learning_rate": 2.4210285221698124e-05,
456
+ "loss": 1739.21203125,
457
+ "step": 2550
458
+ },
459
+ {
460
+ "epoch": 0.8666666666666667,
461
+ "grad_norm": 460.4869079589844,
462
+ "learning_rate": 1.922276599001802e-05,
463
+ "loss": 1737.5628125,
464
+ "step": 2600
465
+ },
466
+ {
467
+ "epoch": 0.8833333333333333,
468
+ "grad_norm": 488.4925231933594,
469
+ "learning_rate": 1.4784261276487333e-05,
470
+ "loss": 1737.22265625,
471
+ "step": 2650
472
+ },
473
+ {
474
+ "epoch": 0.9,
475
+ "grad_norm": 602.5552978515625,
476
+ "learning_rate": 1.0908250674042331e-05,
477
+ "loss": 1736.81359375,
478
+ "step": 2700
479
+ },
480
+ {
481
+ "epoch": 0.9166666666666666,
482
+ "grad_norm": 375.4463195800781,
483
+ "learning_rate": 7.606505499491246e-06,
484
+ "loss": 1736.9728125,
485
+ "step": 2750
486
+ },
487
+ {
488
+ "epoch": 0.9166666666666666,
489
+ "eval_accuracy": 0.17933235581622678,
490
+ "eval_loss": 57.458404541015625,
491
+ "eval_runtime": 7.136,
492
+ "eval_samples_per_second": 17.517,
493
+ "eval_steps_per_second": 0.561,
494
+ "step": 2750
495
+ },
496
+ {
497
+ "epoch": 0.9333333333333333,
498
+ "grad_norm": 411.4505920410156,
499
+ "learning_rate": 4.88905304441194e-06,
500
+ "loss": 1736.17953125,
501
+ "step": 2800
502
+ },
503
+ {
504
+ "epoch": 0.95,
505
+ "grad_norm": 255.17337036132812,
506
+ "learning_rate": 2.7641461225969665e-06,
507
+ "loss": 1737.8128125,
508
+ "step": 2850
509
+ },
510
+ {
511
+ "epoch": 0.9666666666666667,
512
+ "grad_norm": 709.8735961914062,
513
+ "learning_rate": 1.2382380065293576e-06,
514
+ "loss": 1737.474375,
515
+ "step": 2900
516
+ },
517
+ {
518
+ "epoch": 0.9833333333333333,
519
+ "grad_norm": 485.8822326660156,
520
+ "learning_rate": 3.159628290061445e-07,
521
+ "loss": 1737.075,
522
+ "step": 2950
523
+ },
524
+ {
525
+ "epoch": 1.0,
526
+ "grad_norm": 478.07696533203125,
527
+ "learning_rate": 1.2150942938493615e-10,
528
+ "loss": 1737.3753125,
529
+ "step": 3000
530
+ },
531
+ {
532
+ "epoch": 1.0,
533
+ "eval_accuracy": 0.17936754643206257,
534
+ "eval_loss": 57.4544677734375,
535
+ "eval_runtime": 7.3114,
536
+ "eval_samples_per_second": 17.097,
537
+ "eval_steps_per_second": 0.547,
538
+ "step": 3000
539
  }
540
  ],
541
  "logging_steps": 50,
 
550
  "should_evaluate": false,
551
  "should_log": false,
552
  "should_save": true,
553
+ "should_training_stop": true
554
  },
555
  "attributes": {}
556
  }