CodeIsAbstract commited on
Commit
01cca34
·
verified ·
1 Parent(s): 79d4955

Training in progress, step 52000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6bc9d91d9a60b7700203af96a974794d27be7eea3f420ff1ae5b370210ee2ea
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4e7ba8d2524bc9f928490251d5557e2326e4ee76b8165e53f83fb0334d18ddfe
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:84e7123b1ce19594fbb22be55c4498bf5de4fe017896adc3ad66d3524c9c9c54
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7e7abd1b4dbd4f064e3df76d3ce96cc640fdb83fe12440e34c4e94de5dc9ac01
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9623f80738737f22c717a4262aa756fc5e2cd7d18d0d2842f53148de45dc81ec
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:99fd3750e2e46b63e97783e1c2d11f0a6d652ae4d1ee48cc27594356ed0fe238
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2acdb6c434c9b8d7057c6d627425c7ddfd0a206d98e9d43b0c3e164d01587a81
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8715411d67584e8b8ee5de5af78cfc1a263656fc2e3e52390a5b49a21234cc2
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.10909090909090909,
6
  "eval_steps": 1000,
7
- "global_step": 48000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3752,6 +3752,318 @@
3752
  "eval_samples_per_second": 77.404,
3753
  "eval_steps_per_second": 19.351,
3754
  "step": 48000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3755
  }
3756
  ],
3757
  "logging_steps": 100,
@@ -3771,7 +4083,7 @@
3771
  "attributes": {}
3772
  }
3773
  },
3774
- "total_flos": 1.195899887812608e+18,
3775
  "train_batch_size": 22,
3776
  "trial_name": null,
3777
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.14545454545454545,
6
  "eval_steps": 1000,
7
+ "global_step": 52000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3752
  "eval_samples_per_second": 77.404,
3753
  "eval_steps_per_second": 19.351,
3754
  "step": 48000
3755
+ },
3756
+ {
3757
+ "epoch": 0.11,
3758
+ "grad_norm": 0.1616736799478531,
3759
+ "learning_rate": 0.0005990926493458485,
3760
+ "loss": 2.7104031372070314,
3761
+ "step": 48100
3762
+ },
3763
+ {
3764
+ "epoch": 0.11090909090909092,
3765
+ "grad_norm": 0.15068742632865906,
3766
+ "learning_rate": 0.0005976906630331684,
3767
+ "loss": 2.6817190551757815,
3768
+ "step": 48200
3769
+ },
3770
+ {
3771
+ "epoch": 0.11181818181818182,
3772
+ "grad_norm": 0.1564316600561142,
3773
+ "learning_rate": 0.0005962878777101768,
3774
+ "loss": 2.699707336425781,
3775
+ "step": 48300
3776
+ },
3777
+ {
3778
+ "epoch": 0.11272727272727273,
3779
+ "grad_norm": 0.1532757431268692,
3780
+ "learning_rate": 0.0005948843048502316,
3781
+ "loss": 2.7237933349609373,
3782
+ "step": 48400
3783
+ },
3784
+ {
3785
+ "epoch": 0.11363636363636363,
3786
+ "grad_norm": 0.18864984810352325,
3787
+ "learning_rate": 0.0005934799559331318,
3788
+ "loss": 2.7044161987304687,
3789
+ "step": 48500
3790
+ },
3791
+ {
3792
+ "epoch": 0.11454545454545455,
3793
+ "grad_norm": 0.1558079868555069,
3794
+ "learning_rate": 0.0005920748424450241,
3795
+ "loss": 2.7322943115234377,
3796
+ "step": 48600
3797
+ },
3798
+ {
3799
+ "epoch": 0.11545454545454545,
3800
+ "grad_norm": 0.14599354565143585,
3801
+ "learning_rate": 0.0005906689758783081,
3802
+ "loss": 2.6957037353515627,
3803
+ "step": 48700
3804
+ },
3805
+ {
3806
+ "epoch": 0.11636363636363636,
3807
+ "grad_norm": 0.16394606232643127,
3808
+ "learning_rate": 0.0005892623677315435,
3809
+ "loss": 2.735413818359375,
3810
+ "step": 48800
3811
+ },
3812
+ {
3813
+ "epoch": 0.11727272727272728,
3814
+ "grad_norm": 0.16016560792922974,
3815
+ "learning_rate": 0.0005878550295093547,
3816
+ "loss": 2.743086853027344,
3817
+ "step": 48900
3818
+ },
3819
+ {
3820
+ "epoch": 0.11818181818181818,
3821
+ "grad_norm": 0.1616428941488266,
3822
+ "learning_rate": 0.0005864469727223377,
3823
+ "loss": 2.6787353515625,
3824
+ "step": 49000
3825
+ },
3826
+ {
3827
+ "epoch": 0.11818181818181818,
3828
+ "eval_loss": 3.087231159210205,
3829
+ "eval_runtime": 7.451,
3830
+ "eval_samples_per_second": 77.305,
3831
+ "eval_steps_per_second": 19.326,
3832
+ "step": 49000
3833
+ },
3834
+ {
3835
+ "epoch": 0.1190909090909091,
3836
+ "grad_norm": 0.1776418387889862,
3837
+ "learning_rate": 0.0005850382088869656,
3838
+ "loss": 2.6989129638671874,
3839
+ "step": 49100
3840
+ },
3841
+ {
3842
+ "epoch": 0.12,
3843
+ "grad_norm": 0.14982151985168457,
3844
+ "learning_rate": 0.0005836287495254947,
3845
+ "loss": 2.6925369262695313,
3846
+ "step": 49200
3847
+ },
3848
+ {
3849
+ "epoch": 0.12090909090909091,
3850
+ "grad_norm": 0.15882854163646698,
3851
+ "learning_rate": 0.0005822186061658693,
3852
+ "loss": 2.7088381958007814,
3853
+ "step": 49300
3854
+ },
3855
+ {
3856
+ "epoch": 0.12181818181818181,
3857
+ "grad_norm": 0.17874722182750702,
3858
+ "learning_rate": 0.0005808077903416286,
3859
+ "loss": 2.6966705322265625,
3860
+ "step": 49400
3861
+ },
3862
+ {
3863
+ "epoch": 0.12272727272727273,
3864
+ "grad_norm": 0.17074082791805267,
3865
+ "learning_rate": 0.0005793963135918122,
3866
+ "loss": 2.7259332275390626,
3867
+ "step": 49500
3868
+ },
3869
+ {
3870
+ "epoch": 0.12363636363636364,
3871
+ "grad_norm": 0.15236154198646545,
3872
+ "learning_rate": 0.0005779841874608646,
3873
+ "loss": 2.7060787963867186,
3874
+ "step": 49600
3875
+ },
3876
+ {
3877
+ "epoch": 0.12454545454545454,
3878
+ "grad_norm": 0.15823383629322052,
3879
+ "learning_rate": 0.0005765714234985422,
3880
+ "loss": 2.681339111328125,
3881
+ "step": 49700
3882
+ },
3883
+ {
3884
+ "epoch": 0.12545454545454546,
3885
+ "grad_norm": 0.21889136731624603,
3886
+ "learning_rate": 0.0005751580332598178,
3887
+ "loss": 2.693275146484375,
3888
+ "step": 49800
3889
+ },
3890
+ {
3891
+ "epoch": 0.12636363636363637,
3892
+ "grad_norm": 0.1490192860364914,
3893
+ "learning_rate": 0.000573744028304787,
3894
+ "loss": 2.6937310791015623,
3895
+ "step": 49900
3896
+ },
3897
+ {
3898
+ "epoch": 0.12727272727272726,
3899
+ "grad_norm": 0.1623009741306305,
3900
+ "learning_rate": 0.0005723294201985724,
3901
+ "loss": 2.678116760253906,
3902
+ "step": 50000
3903
+ },
3904
+ {
3905
+ "epoch": 0.12727272727272726,
3906
+ "eval_loss": 3.082812786102295,
3907
+ "eval_runtime": 7.4452,
3908
+ "eval_samples_per_second": 77.366,
3909
+ "eval_steps_per_second": 19.341,
3910
+ "step": 50000
3911
+ },
3912
+ {
3913
+ "epoch": 0.12818181818181817,
3914
+ "grad_norm": 0.1424219161272049,
3915
+ "learning_rate": 0.0005709142205112308,
3916
+ "loss": 2.690723876953125,
3917
+ "step": 50100
3918
+ },
3919
+ {
3920
+ "epoch": 0.1290909090909091,
3921
+ "grad_norm": 0.15792372822761536,
3922
+ "learning_rate": 0.0005694984408176566,
3923
+ "loss": 2.7299853515625,
3924
+ "step": 50200
3925
+ },
3926
+ {
3927
+ "epoch": 0.13,
3928
+ "grad_norm": 0.151839479804039,
3929
+ "learning_rate": 0.0005680820926974884,
3930
+ "loss": 2.665604248046875,
3931
+ "step": 50300
3932
+ },
3933
+ {
3934
+ "epoch": 0.13090909090909092,
3935
+ "grad_norm": 0.1519307941198349,
3936
+ "learning_rate": 0.0005666651877350139,
3937
+ "loss": 2.6907989501953127,
3938
+ "step": 50400
3939
+ },
3940
+ {
3941
+ "epoch": 0.1318181818181818,
3942
+ "grad_norm": 0.14843234419822693,
3943
+ "learning_rate": 0.0005652477375190755,
3944
+ "loss": 2.684622802734375,
3945
+ "step": 50500
3946
+ },
3947
+ {
3948
+ "epoch": 0.13272727272727272,
3949
+ "grad_norm": 0.1492377370595932,
3950
+ "learning_rate": 0.000563829753642975,
3951
+ "loss": 2.68843994140625,
3952
+ "step": 50600
3953
+ },
3954
+ {
3955
+ "epoch": 0.13363636363636364,
3956
+ "grad_norm": 0.1397194117307663,
3957
+ "learning_rate": 0.0005624112477043789,
3958
+ "loss": 2.6910018920898438,
3959
+ "step": 50700
3960
+ },
3961
+ {
3962
+ "epoch": 0.13454545454545455,
3963
+ "grad_norm": 0.14176790416240692,
3964
+ "learning_rate": 0.0005609922313052237,
3965
+ "loss": 2.6935055541992186,
3966
+ "step": 50800
3967
+ },
3968
+ {
3969
+ "epoch": 0.13545454545454547,
3970
+ "grad_norm": 0.16330276429653168,
3971
+ "learning_rate": 0.0005595727160516208,
3972
+ "loss": 2.6888656616210938,
3973
+ "step": 50900
3974
+ },
3975
+ {
3976
+ "epoch": 0.13636363636363635,
3977
+ "grad_norm": 0.1529064178466797,
3978
+ "learning_rate": 0.0005581527135537623,
3979
+ "loss": 2.709434814453125,
3980
+ "step": 51000
3981
+ },
3982
+ {
3983
+ "epoch": 0.13636363636363635,
3984
+ "eval_loss": 3.07966685295105,
3985
+ "eval_runtime": 7.3998,
3986
+ "eval_samples_per_second": 77.84,
3987
+ "eval_steps_per_second": 19.46,
3988
+ "step": 51000
3989
+ },
3990
+ {
3991
+ "epoch": 0.13727272727272727,
3992
+ "grad_norm": 0.1499529331922531,
3993
+ "learning_rate": 0.0005567322354258249,
3994
+ "loss": 2.694036560058594,
3995
+ "step": 51100
3996
+ },
3997
+ {
3998
+ "epoch": 0.13818181818181818,
3999
+ "grad_norm": 0.1941758096218109,
4000
+ "learning_rate": 0.0005553112932858754,
4001
+ "loss": 2.679619140625,
4002
+ "step": 51200
4003
+ },
4004
+ {
4005
+ "epoch": 0.1390909090909091,
4006
+ "grad_norm": 0.1703883409500122,
4007
+ "learning_rate": 0.0005538898987557763,
4008
+ "loss": 2.7490829467773437,
4009
+ "step": 51300
4010
+ },
4011
+ {
4012
+ "epoch": 0.14,
4013
+ "grad_norm": 0.1707204133272171,
4014
+ "learning_rate": 0.0005524680634610897,
4015
+ "loss": 2.701058349609375,
4016
+ "step": 51400
4017
+ },
4018
+ {
4019
+ "epoch": 0.1409090909090909,
4020
+ "grad_norm": 0.1748669594526291,
4021
+ "learning_rate": 0.0005510457990309829,
4022
+ "loss": 2.6962744140625,
4023
+ "step": 51500
4024
+ },
4025
+ {
4026
+ "epoch": 0.14181818181818182,
4027
+ "grad_norm": 0.16774185001850128,
4028
+ "learning_rate": 0.0005496231170981329,
4029
+ "loss": 2.6645233154296877,
4030
+ "step": 51600
4031
+ },
4032
+ {
4033
+ "epoch": 0.14272727272727273,
4034
+ "grad_norm": 0.15208636224269867,
4035
+ "learning_rate": 0.000548200029298632,
4036
+ "loss": 2.6871743774414063,
4037
+ "step": 51700
4038
+ },
4039
+ {
4040
+ "epoch": 0.14363636363636365,
4041
+ "grad_norm": 0.17048777639865875,
4042
+ "learning_rate": 0.0005467765472718914,
4043
+ "loss": 2.6941012573242187,
4044
+ "step": 51800
4045
+ },
4046
+ {
4047
+ "epoch": 0.14454545454545453,
4048
+ "grad_norm": 0.1853557974100113,
4049
+ "learning_rate": 0.000545352682660547,
4050
+ "loss": 2.669810791015625,
4051
+ "step": 51900
4052
+ },
4053
+ {
4054
+ "epoch": 0.14545454545454545,
4055
+ "grad_norm": 0.18524755537509918,
4056
+ "learning_rate": 0.0005439284471103641,
4057
+ "loss": 2.714384765625,
4058
+ "step": 52000
4059
+ },
4060
+ {
4061
+ "epoch": 0.14545454545454545,
4062
+ "eval_loss": 3.08223819732666,
4063
+ "eval_runtime": 7.441,
4064
+ "eval_samples_per_second": 77.409,
4065
+ "eval_steps_per_second": 19.352,
4066
+ "step": 52000
4067
  }
4068
  ],
4069
  "logging_steps": 100,
 
4083
  "attributes": {}
4084
  }
4085
  },
4086
+ "total_flos": 1.295558211796992e+18,
4087
  "train_batch_size": 22,
4088
  "trial_name": null,
4089
  "trial_params": null