CodeIsAbstract commited on
Commit
aaf6617
·
verified ·
1 Parent(s): dd17991

Training in progress, step 40000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cb666cb8677bceaafadd9b9b5e659661ea31b2586e5ed485eda4a008d0ef5ef9
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ed761b6beba77fb535cbed31c0aa2e899d00d55d8771441f7ca65e1ed885e7b
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7d2d23f0c5c7e252c38dfbeb64477cc0089f8a30877e6082b7e1a3a74d55684d
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a2850114abd645eb97f3c25c47199cf3e8cf4b9bc81d7c505f6e6dc8f2a5e61
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:51f1ebff0433542b44861a56d7924f69b6e5f3c7271e7aae262253030bc6e5a0
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:319c69ec456b4fcce9018c73539aecc383b0cb34ee142655e66ea42576668bfe
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c07a50c67c728026d8510c54dbf23cc51e59527759deefd728a51b01b02ec546
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be73e97160af6aa989641ee6aa859d208a12c6f78430f9ec70c726fde6de838d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 38000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3010,6 +3010,164 @@
3010
  "eval_samples_per_second": 220.749,
3011
  "eval_steps_per_second": 13.875,
3012
  "step": 38000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3013
  }
3014
  ],
3015
  "logging_steps": 100,
@@ -3029,7 +3187,7 @@
3029
  "attributes": {}
3030
  }
3031
  },
3032
- "total_flos": 1.48958512939008e+18,
3033
  "train_batch_size": 120,
3034
  "trial_name": null,
3035
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 40000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3010
  "eval_samples_per_second": 220.749,
3011
  "eval_steps_per_second": 13.875,
3012
  "step": 38000
3013
+ },
3014
+ {
3015
+ "epoch": 0.122,
3016
+ "grad_norm": 0.14622287452220917,
3017
+ "learning_rate": 0.0001341596278714808,
3018
+ "loss": 2.981756286621094,
3019
+ "step": 38100
3020
+ },
3021
+ {
3022
+ "epoch": 0.124,
3023
+ "grad_norm": 0.14599764347076416,
3024
+ "learning_rate": 0.00013201900456264433,
3025
+ "loss": 2.9829974365234375,
3026
+ "step": 38200
3027
+ },
3028
+ {
3029
+ "epoch": 0.126,
3030
+ "grad_norm": 0.15080520510673523,
3031
+ "learning_rate": 0.00012989299607050176,
3032
+ "loss": 3.010633544921875,
3033
+ "step": 38300
3034
+ },
3035
+ {
3036
+ "epoch": 0.128,
3037
+ "grad_norm": 0.1488070785999298,
3038
+ "learning_rate": 0.00012778168683208935,
3039
+ "loss": 2.9718212890625,
3040
+ "step": 38400
3041
+ },
3042
+ {
3043
+ "epoch": 0.13,
3044
+ "grad_norm": 0.16047491133213043,
3045
+ "learning_rate": 0.00012568516070064328,
3046
+ "loss": 2.979062805175781,
3047
+ "step": 38500
3048
+ },
3049
+ {
3050
+ "epoch": 0.132,
3051
+ "grad_norm": 0.14107033610343933,
3052
+ "learning_rate": 0.00012360350094227102,
3053
+ "loss": 2.9834564208984373,
3054
+ "step": 38600
3055
+ },
3056
+ {
3057
+ "epoch": 0.134,
3058
+ "grad_norm": 0.16287438571453094,
3059
+ "learning_rate": 0.0001215367902326442,
3060
+ "loss": 2.988681640625,
3061
+ "step": 38700
3062
+ },
3063
+ {
3064
+ "epoch": 0.136,
3065
+ "grad_norm": 0.1597314476966858,
3066
+ "learning_rate": 0.00011948511065371376,
3067
+ "loss": 2.9878759765625,
3068
+ "step": 38800
3069
+ },
3070
+ {
3071
+ "epoch": 0.138,
3072
+ "grad_norm": 0.1734859049320221,
3073
+ "learning_rate": 0.0001174485436904515,
3074
+ "loss": 2.9992620849609377,
3075
+ "step": 38900
3076
+ },
3077
+ {
3078
+ "epoch": 0.14,
3079
+ "grad_norm": 0.15174660086631775,
3080
+ "learning_rate": 0.0001154271702276129,
3081
+ "loss": 2.9760171508789064,
3082
+ "step": 39000
3083
+ },
3084
+ {
3085
+ "epoch": 0.14,
3086
+ "eval_accuracy": 0.383085765906371,
3087
+ "eval_loss": 3.2904224395751953,
3088
+ "eval_runtime": 8.8802,
3089
+ "eval_samples_per_second": 218.577,
3090
+ "eval_steps_per_second": 13.738,
3091
+ "step": 39000
3092
+ },
3093
+ {
3094
+ "epoch": 0.142,
3095
+ "grad_norm": 0.14373154938220978,
3096
+ "learning_rate": 0.00011342107054652494,
3097
+ "loss": 2.9714199829101564,
3098
+ "step": 39100
3099
+ },
3100
+ {
3101
+ "epoch": 0.144,
3102
+ "grad_norm": 0.1446230262517929,
3103
+ "learning_rate": 0.00011143032432189776,
3104
+ "loss": 2.985970764160156,
3105
+ "step": 39200
3106
+ },
3107
+ {
3108
+ "epoch": 0.146,
3109
+ "grad_norm": 0.15378804504871368,
3110
+ "learning_rate": 0.00010945501061865998,
3111
+ "loss": 2.9741244506835938,
3112
+ "step": 39300
3113
+ },
3114
+ {
3115
+ "epoch": 0.148,
3116
+ "grad_norm": 0.14110712707042694,
3117
+ "learning_rate": 0.00010749520788881883,
3118
+ "loss": 2.9957791137695313,
3119
+ "step": 39400
3120
+ },
3121
+ {
3122
+ "epoch": 0.15,
3123
+ "grad_norm": 0.14115746319293976,
3124
+ "learning_rate": 0.00010555099396834378,
3125
+ "loss": 2.957854919433594,
3126
+ "step": 39500
3127
+ },
3128
+ {
3129
+ "epoch": 0.152,
3130
+ "grad_norm": 0.13682854175567627,
3131
+ "learning_rate": 0.0001036224460740765,
3132
+ "loss": 2.9818161010742186,
3133
+ "step": 39600
3134
+ },
3135
+ {
3136
+ "epoch": 0.154,
3137
+ "grad_norm": 0.15571151673793793,
3138
+ "learning_rate": 0.00010170964080066225,
3139
+ "loss": 2.9793820190429687,
3140
+ "step": 39700
3141
+ },
3142
+ {
3143
+ "epoch": 0.156,
3144
+ "grad_norm": 0.14156827330589294,
3145
+ "learning_rate": 9.981265411750934e-05,
3146
+ "loss": 2.963619384765625,
3147
+ "step": 39800
3148
+ },
3149
+ {
3150
+ "epoch": 0.158,
3151
+ "grad_norm": 0.1549520343542099,
3152
+ "learning_rate": 9.793156136577125e-05,
3153
+ "loss": 2.9770501708984374,
3154
+ "step": 39900
3155
+ },
3156
+ {
3157
+ "epoch": 0.16,
3158
+ "grad_norm": 0.15499472618103027,
3159
+ "learning_rate": 9.606643725535436e-05,
3160
+ "loss": 2.9700833129882813,
3161
+ "step": 40000
3162
+ },
3163
+ {
3164
+ "epoch": 0.16,
3165
+ "eval_accuracy": 0.38321784219605565,
3166
+ "eval_loss": 3.286919593811035,
3167
+ "eval_runtime": 8.6196,
3168
+ "eval_samples_per_second": 225.184,
3169
+ "eval_steps_per_second": 14.154,
3170
+ "step": 40000
3171
  }
3172
  ],
3173
  "logging_steps": 100,
 
3187
  "attributes": {}
3188
  }
3189
  },
3190
+ "total_flos": 1.5679843467264e+18,
3191
  "train_batch_size": 120,
3192
  "trial_name": null,
3193
  "trial_params": null