CodeIsAbstract commited on
Commit
943b6e6
·
verified ·
1 Parent(s): 4179e5d

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:206c62db0b89449568ebdd8c25cd017d55ecb7bf6df86bab1e451b3b3fd8f88e
3
  size 579748776
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:782162ee692bc89212bb1bf2237fad94be0a4f8d8e6006bd2ffb4a44a2317764
3
  size 579748776
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5a9af1204c1f4878c612726d1537329b4144d6b734182cd55d2c644c1a10e4b7
3
  size 1159627083
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0b7fcb1417225ab4b260a3b69014227cbfd71b2c42ffb3dfa186c5847e2fbe41
3
  size 1159627083
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5445b36fdb78ff76c2ef206d44397c5aad17f330414e24d03bab0de9bf0e9222
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc46d826a65dffd9fffb4922acf95d98d089e8e092701fad0e51a5751ee5d30b
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:108a89056f023268994da88aac029418224df4448adc2b0cb4d1083cb65ac2b0
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4176a469ce53e5e5b4adef275d41a37a84a174b35795b06d37448f614386eab4
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6d0f2c84615336a1d1664a4244642bd13601391ec93a4749b9c0e0cdc17433aa
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c0d627c7f8a0fb48b948b9cb274ead2ddc9dfdd7223d39ba3cdfe65192c3410
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.9,
6
  "eval_steps": 100,
7
- "global_step": 1800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -674,6 +674,80 @@
674
  "eval_samples_per_second": 32.679,
675
  "eval_steps_per_second": 1.078,
676
  "step": 1800
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
677
  }
678
  ],
679
  "logging_steps": 25,
@@ -688,12 +762,12 @@
688
  "should_evaluate": false,
689
  "should_log": false,
690
  "should_save": true,
691
- "should_training_stop": false
692
  },
693
  "attributes": {}
694
  }
695
  },
696
- "total_flos": 1.2040032428752896e+18,
697
  "train_batch_size": 64,
698
  "trial_name": null,
699
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.0,
6
  "eval_steps": 100,
7
+ "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
674
  "eval_samples_per_second": 32.679,
675
  "eval_steps_per_second": 1.078,
676
  "step": 1800
677
+ },
678
+ {
679
+ "epoch": 0.9125,
680
+ "grad_norm": 300.458984375,
681
+ "learning_rate": 2.216595153886969e-06,
682
+ "loss": 96.5413,
683
+ "step": 1825
684
+ },
685
+ {
686
+ "epoch": 0.925,
687
+ "grad_norm": 1235.066162109375,
688
+ "learning_rate": 1.6348173407569557e-06,
689
+ "loss": 96.6509,
690
+ "step": 1850
691
+ },
692
+ {
693
+ "epoch": 0.9375,
694
+ "grad_norm": 1469.02978515625,
695
+ "learning_rate": 1.1401968558123976e-06,
696
+ "loss": 96.5367,
697
+ "step": 1875
698
+ },
699
+ {
700
+ "epoch": 0.95,
701
+ "grad_norm": 632.1422119140625,
702
+ "learning_rate": 7.336250385988896e-07,
703
+ "loss": 96.472,
704
+ "step": 1900
705
+ },
706
+ {
707
+ "epoch": 0.95,
708
+ "eval_accuracy": 0.15278578605365067,
709
+ "eval_loss": 6.0192670822143555,
710
+ "eval_runtime": 7.5713,
711
+ "eval_samples_per_second": 64.058,
712
+ "eval_steps_per_second": 2.113,
713
+ "step": 1900
714
+ },
715
+ {
716
+ "epoch": 0.9625,
717
+ "grad_norm": 2200.51904296875,
718
+ "learning_rate": 4.1583455901149647e-07,
719
+ "loss": 96.5353,
720
+ "step": 1925
721
+ },
722
+ {
723
+ "epoch": 0.975,
724
+ "grad_norm": 1819.2476806640625,
725
+ "learning_rate": 1.8739809697409517e-07,
726
+ "loss": 96.0391,
727
+ "step": 1950
728
+ },
729
+ {
730
+ "epoch": 0.9875,
731
+ "grad_norm": 1034.934326171875,
732
+ "learning_rate": 4.872731043143453e-08,
733
+ "loss": 96.1378,
734
+ "step": 1975
735
+ },
736
+ {
737
+ "epoch": 1.0,
738
+ "grad_norm": 477.2402648925781,
739
+ "learning_rate": 7.209351372550188e-11,
740
+ "loss": 96.5051,
741
+ "step": 2000
742
+ },
743
+ {
744
+ "epoch": 1.0,
745
+ "eval_accuracy": 0.15278376645343053,
746
+ "eval_loss": 6.019127368927002,
747
+ "eval_runtime": 7.2999,
748
+ "eval_samples_per_second": 66.44,
749
+ "eval_steps_per_second": 2.192,
750
+ "step": 2000
751
  }
752
  ],
753
  "logging_steps": 25,
 
762
  "should_evaluate": false,
763
  "should_log": false,
764
  "should_save": true,
765
+ "should_training_stop": true
766
  },
767
  "attributes": {}
768
  }
769
  },
770
+ "total_flos": 1.337781380972544e+18,
771
  "train_batch_size": 64,
772
  "trial_name": null,
773
  "trial_params": null