CodeIsAbstract commited on
Commit
c797c2c
·
verified ·
1 Parent(s): f47d1de

Training in progress, step 40000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:113fb894b062b29cf117cc658db504261451a8ff8c1a0cb06fb7e28935cf14f7
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:effd2d14476df74bc5f8738835a04cd2c53990d3720280723645c691281e29a7
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a46699af39339744e0585d58fd78b70cf3632fc84e8d7d9301d7e78f3c4dfc81
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7c29b0dac895207221ca2ff233a263d1f8bafafacb0961d5bd379e4086db6f23
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cadea5b42974dfb2cd316f1afeedc6d7ac097ea0c6cf0e13d2fffcae3de277b9
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee2b9e8d80dbc1551afe65580bf18f91ce6a2908c043aa9c60a24adad8385acc
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f1f70882941e6b31d82263e4fc807f24daa03205e781becc1cf39b5843e1bb88
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7778da91eeef8b8307a814fd6b609cdd038d764b06717fd843f1619aa9b55603
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -4,7 +4,7 @@
4
  "best_model_checkpoint": null,
5
  "epoch": 0.03636363636363636,
6
  "eval_steps": 1000,
7
- "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2816,6 +2816,318 @@
2816
  "eval_samples_per_second": 77.327,
2817
  "eval_steps_per_second": 19.332,
2818
  "step": 36000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2819
  }
2820
  ],
2821
  "logging_steps": 100,
@@ -2835,7 +3147,7 @@
2835
  "attributes": {}
2836
  }
2837
  },
2838
- "total_flos": 8.96924915859456e+17,
2839
  "train_batch_size": 22,
2840
  "trial_name": null,
2841
  "trial_params": null
 
4
  "best_model_checkpoint": null,
5
  "epoch": 0.03636363636363636,
6
  "eval_steps": 1000,
7
+ "global_step": 40000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2816
  "eval_samples_per_second": 77.327,
2817
  "eval_steps_per_second": 19.332,
2818
  "step": 36000
2819
+ },
2820
+ {
2821
+ "epoch": 0.0009090909090909091,
2822
+ "grad_norm": 0.20997287333011627,
2823
+ "learning_rate": 0.0007582221078899511,
2824
+ "loss": 2.71048828125,
2825
+ "step": 36100
2826
+ },
2827
+ {
2828
+ "epoch": 0.0018181818181818182,
2829
+ "grad_norm": 0.1532030999660492,
2830
+ "learning_rate": 0.0007569965605062139,
2831
+ "loss": 2.7088739013671876,
2832
+ "step": 36200
2833
+ },
2834
+ {
2835
+ "epoch": 0.0027272727272727275,
2836
+ "grad_norm": 0.14671523869037628,
2837
+ "learning_rate": 0.000755768911151874,
2838
+ "loss": 2.730252990722656,
2839
+ "step": 36300
2840
+ },
2841
+ {
2842
+ "epoch": 0.0036363636363636364,
2843
+ "grad_norm": 0.1369636058807373,
2844
+ "learning_rate": 0.0007545391698678547,
2845
+ "loss": 2.7253591918945315,
2846
+ "step": 36400
2847
+ },
2848
+ {
2849
+ "epoch": 0.004545454545454545,
2850
+ "grad_norm": 0.16328130662441254,
2851
+ "learning_rate": 0.00075330734671219,
2852
+ "loss": 2.6905337524414064,
2853
+ "step": 36500
2854
+ },
2855
+ {
2856
+ "epoch": 0.005454545454545455,
2857
+ "grad_norm": 0.13856680691242218,
2858
+ "learning_rate": 0.0007520734517599409,
2859
+ "loss": 2.6842236328125,
2860
+ "step": 36600
2861
+ },
2862
+ {
2863
+ "epoch": 0.006363636363636364,
2864
+ "grad_norm": 0.14775875210762024,
2865
+ "learning_rate": 0.0007508374951031139,
2866
+ "loss": 2.7059420776367187,
2867
+ "step": 36700
2868
+ },
2869
+ {
2870
+ "epoch": 0.007272727272727273,
2871
+ "grad_norm": 0.14245547354221344,
2872
+ "learning_rate": 0.0007495994868505774,
2873
+ "loss": 2.6968829345703127,
2874
+ "step": 36800
2875
+ },
2876
+ {
2877
+ "epoch": 0.008181818181818182,
2878
+ "grad_norm": 0.15311609208583832,
2879
+ "learning_rate": 0.000748359437127981,
2880
+ "loss": 2.7318048095703125,
2881
+ "step": 36900
2882
+ },
2883
+ {
2884
+ "epoch": 0.00909090909090909,
2885
+ "grad_norm": 0.1730499267578125,
2886
+ "learning_rate": 0.0007471173560776705,
2887
+ "loss": 2.6874020385742186,
2888
+ "step": 37000
2889
+ },
2890
+ {
2891
+ "epoch": 0.00909090909090909,
2892
+ "eval_loss": 3.13167405128479,
2893
+ "eval_runtime": 7.5047,
2894
+ "eval_samples_per_second": 76.752,
2895
+ "eval_steps_per_second": 19.188,
2896
+ "step": 37000
2897
+ },
2898
+ {
2899
+ "epoch": 0.01,
2900
+ "grad_norm": 0.15705707669258118,
2901
+ "learning_rate": 0.0007458732538586064,
2902
+ "loss": 2.692235107421875,
2903
+ "step": 37100
2904
+ },
2905
+ {
2906
+ "epoch": 0.01090909090909091,
2907
+ "grad_norm": 0.1563568264245987,
2908
+ "learning_rate": 0.0007446271406462797,
2909
+ "loss": 2.706945495605469,
2910
+ "step": 37200
2911
+ },
2912
+ {
2913
+ "epoch": 0.011818181818181818,
2914
+ "grad_norm": 0.16822531819343567,
2915
+ "learning_rate": 0.00074337902663263,
2916
+ "loss": 2.685476989746094,
2917
+ "step": 37300
2918
+ },
2919
+ {
2920
+ "epoch": 0.012727272727272728,
2921
+ "grad_norm": 0.14851588010787964,
2922
+ "learning_rate": 0.000742128922025961,
2923
+ "loss": 2.692755126953125,
2924
+ "step": 37400
2925
+ },
2926
+ {
2927
+ "epoch": 0.013636363636363636,
2928
+ "grad_norm": 0.1650208979845047,
2929
+ "learning_rate": 0.0007408768370508576,
2930
+ "loss": 2.68127197265625,
2931
+ "step": 37500
2932
+ },
2933
+ {
2934
+ "epoch": 0.014545454545454545,
2935
+ "grad_norm": 0.1679512858390808,
2936
+ "learning_rate": 0.0007396227819481021,
2937
+ "loss": 2.697837219238281,
2938
+ "step": 37600
2939
+ },
2940
+ {
2941
+ "epoch": 0.015454545454545455,
2942
+ "grad_norm": 0.1745394617319107,
2943
+ "learning_rate": 0.00073836676697459,
2944
+ "loss": 2.6750924682617185,
2945
+ "step": 37700
2946
+ },
2947
+ {
2948
+ "epoch": 0.016363636363636365,
2949
+ "grad_norm": 0.14067143201828003,
2950
+ "learning_rate": 0.0007371088024032475,
2951
+ "loss": 2.7095330810546874,
2952
+ "step": 37800
2953
+ },
2954
+ {
2955
+ "epoch": 0.017272727272727273,
2956
+ "grad_norm": 0.16379132866859436,
2957
+ "learning_rate": 0.0007358488985229455,
2958
+ "loss": 2.6968374633789063,
2959
+ "step": 37900
2960
+ },
2961
+ {
2962
+ "epoch": 0.01818181818181818,
2963
+ "grad_norm": 0.15668943524360657,
2964
+ "learning_rate": 0.000734587065638417,
2965
+ "loss": 2.697041320800781,
2966
+ "step": 38000
2967
+ },
2968
+ {
2969
+ "epoch": 0.01818181818181818,
2970
+ "eval_loss": 3.128469467163086,
2971
+ "eval_runtime": 7.47,
2972
+ "eval_samples_per_second": 77.109,
2973
+ "eval_steps_per_second": 19.277,
2974
+ "step": 38000
2975
+ },
2976
+ {
2977
+ "epoch": 0.019090909090909092,
2978
+ "grad_norm": 0.1642797887325287,
2979
+ "learning_rate": 0.0007333233140701722,
2980
+ "loss": 2.6924237060546874,
2981
+ "step": 38100
2982
+ },
2983
+ {
2984
+ "epoch": 0.02,
2985
+ "grad_norm": 0.16816826164722443,
2986
+ "learning_rate": 0.0007320576541544144,
2987
+ "loss": 2.69775634765625,
2988
+ "step": 38200
2989
+ },
2990
+ {
2991
+ "epoch": 0.02090909090909091,
2992
+ "grad_norm": 0.1469792127609253,
2993
+ "learning_rate": 0.000730790096242955,
2994
+ "loss": 2.6987783813476565,
2995
+ "step": 38300
2996
+ },
2997
+ {
2998
+ "epoch": 0.02181818181818182,
2999
+ "grad_norm": 0.15169106423854828,
3000
+ "learning_rate": 0.0007295206507031289,
3001
+ "loss": 2.6766021728515623,
3002
+ "step": 38400
3003
+ },
3004
+ {
3005
+ "epoch": 0.022727272727272728,
3006
+ "grad_norm": 0.14681929349899292,
3007
+ "learning_rate": 0.0007282493279177103,
3008
+ "loss": 2.691963806152344,
3009
+ "step": 38500
3010
+ },
3011
+ {
3012
+ "epoch": 0.023636363636363636,
3013
+ "grad_norm": 0.1452484130859375,
3014
+ "learning_rate": 0.000726976138284827,
3015
+ "loss": 2.6880203247070313,
3016
+ "step": 38600
3017
+ },
3018
+ {
3019
+ "epoch": 0.024545454545454544,
3020
+ "grad_norm": 0.15825141966342926,
3021
+ "learning_rate": 0.0007257010922178762,
3022
+ "loss": 2.65422119140625,
3023
+ "step": 38700
3024
+ },
3025
+ {
3026
+ "epoch": 0.025454545454545455,
3027
+ "grad_norm": 0.15018871426582336,
3028
+ "learning_rate": 0.000724424200145438,
3029
+ "loss": 2.663526611328125,
3030
+ "step": 38800
3031
+ },
3032
+ {
3033
+ "epoch": 0.026363636363636363,
3034
+ "grad_norm": 0.16328509151935577,
3035
+ "learning_rate": 0.0007231454725111919,
3036
+ "loss": 2.6569583129882814,
3037
+ "step": 38900
3038
+ },
3039
+ {
3040
+ "epoch": 0.02727272727272727,
3041
+ "grad_norm": 0.14703254401683807,
3042
+ "learning_rate": 0.0007218649197738298,
3043
+ "loss": 2.69577392578125,
3044
+ "step": 39000
3045
+ },
3046
+ {
3047
+ "epoch": 0.02727272727272727,
3048
+ "eval_loss": 3.1291093826293945,
3049
+ "eval_runtime": 7.4103,
3050
+ "eval_samples_per_second": 77.729,
3051
+ "eval_steps_per_second": 19.432,
3052
+ "step": 39000
3053
+ },
3054
+ {
3055
+ "epoch": 0.028181818181818183,
3056
+ "grad_norm": 0.1588928997516632,
3057
+ "learning_rate": 0.0007205825524069714,
3058
+ "loss": 2.6820574951171876,
3059
+ "step": 39100
3060
+ },
3061
+ {
3062
+ "epoch": 0.02909090909090909,
3063
+ "grad_norm": 0.18345706164836884,
3064
+ "learning_rate": 0.0007192983808990781,
3065
+ "loss": 2.7058285522460936,
3066
+ "step": 39200
3067
+ },
3068
+ {
3069
+ "epoch": 0.03,
3070
+ "grad_norm": 0.1431887000799179,
3071
+ "learning_rate": 0.0007180124157533671,
3072
+ "loss": 2.6863577270507815,
3073
+ "step": 39300
3074
+ },
3075
+ {
3076
+ "epoch": 0.03090909090909091,
3077
+ "grad_norm": 0.15360857546329498,
3078
+ "learning_rate": 0.0007167246674877264,
3079
+ "loss": 2.6997265625,
3080
+ "step": 39400
3081
+ },
3082
+ {
3083
+ "epoch": 0.031818181818181815,
3084
+ "grad_norm": 0.15961426496505737,
3085
+ "learning_rate": 0.0007154351466346273,
3086
+ "loss": 2.66740966796875,
3087
+ "step": 39500
3088
+ },
3089
+ {
3090
+ "epoch": 0.03272727272727273,
3091
+ "grad_norm": 0.15588298439979553,
3092
+ "learning_rate": 0.0007141438637410398,
3093
+ "loss": 2.6652230834960937,
3094
+ "step": 39600
3095
+ },
3096
+ {
3097
+ "epoch": 0.03363636363636364,
3098
+ "grad_norm": 0.1670597940683365,
3099
+ "learning_rate": 0.000712850829368345,
3100
+ "loss": 2.681721496582031,
3101
+ "step": 39700
3102
+ },
3103
+ {
3104
+ "epoch": 0.034545454545454546,
3105
+ "grad_norm": 0.15003304183483124,
3106
+ "learning_rate": 0.0007115560540922497,
3107
+ "loss": 2.6677728271484376,
3108
+ "step": 39800
3109
+ },
3110
+ {
3111
+ "epoch": 0.035454545454545454,
3112
+ "grad_norm": 0.1557546705007553,
3113
+ "learning_rate": 0.0007102595485026993,
3114
+ "loss": 2.6783493041992186,
3115
+ "step": 39900
3116
+ },
3117
+ {
3118
+ "epoch": 0.03636363636363636,
3119
+ "grad_norm": 0.1732822060585022,
3120
+ "learning_rate": 0.0007089613232037915,
3121
+ "loss": 2.6648553466796874,
3122
+ "step": 40000
3123
+ },
3124
+ {
3125
+ "epoch": 0.03636363636363636,
3126
+ "eval_loss": 3.129150867462158,
3127
+ "eval_runtime": 7.5094,
3128
+ "eval_samples_per_second": 76.703,
3129
+ "eval_steps_per_second": 19.176,
3130
+ "step": 40000
3131
  }
3132
  ],
3133
  "logging_steps": 100,
 
3147
  "attributes": {}
3148
  }
3149
  },
3150
+ "total_flos": 9.9658323984384e+17,
3151
  "train_batch_size": 22,
3152
  "trial_name": null,
3153
  "trial_params": null