CodeIsAbstract commited on
Commit
a73c132
·
verified ·
1 Parent(s): 13ac386

Training in progress, step 16000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2e0ff05e0e5d852455b4d83922c7abbecdec428e19e46c0b823ee661ac50cdb5
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:69c3b821cf12eeaa022a644bc22041eac717a7550c375de853c5ee543353c15e
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32d116e4c03c1c5f0b5164cbe30850c283753e36a9119b42ffefe95697cc08e3
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:099f48c6bd773fc094319bca20e4dc9f9c6178fc27af5dbdbf0e8f59bf625192
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:74ba315e55e014287fc4f548e786e1c64991f07f2dd31f970e2f14eab773946e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:79bfab8e6d704b55615f0aa752b32bcc403a434935bef147322933f0756a46f1
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:238cbdda4406a9b9e4f702cada921a66b6ebd035a4f580dabf57521e3dc0c97d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a3daf8da2259c0a45cc86b801dfa39aec812fc636174a8b32bb6d363e93dc412
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.28,
6
  "eval_steps": 1000,
7
- "global_step": 14000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1114,6 +1114,164 @@
1114
  "eval_samples_per_second": 194.114,
1115
  "eval_steps_per_second": 12.201,
1116
  "step": 14000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1117
  }
1118
  ],
1119
  "logging_steps": 100,
@@ -1133,7 +1291,7 @@
1133
  "attributes": {}
1134
  }
1135
  },
1136
- "total_flos": 4.3897061376e+17,
1137
  "train_batch_size": 120,
1138
  "trial_name": null,
1139
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.32,
6
  "eval_steps": 1000,
7
+ "global_step": 16000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1114
  "eval_samples_per_second": 194.114,
1115
  "eval_steps_per_second": 12.201,
1116
  "step": 14000
1117
+ },
1118
+ {
1119
+ "epoch": 0.282,
1120
+ "grad_norm": 0.22074639797210693,
1121
+ "learning_rate": 0.0004913582739740128,
1122
+ "loss": 3.578910217285156,
1123
+ "step": 14100
1124
+ },
1125
+ {
1126
+ "epoch": 0.284,
1127
+ "grad_norm": 0.20357950031757355,
1128
+ "learning_rate": 0.000489898413862256,
1129
+ "loss": 3.5391220092773437,
1130
+ "step": 14200
1131
+ },
1132
+ {
1133
+ "epoch": 0.286,
1134
+ "grad_norm": 0.2000010460615158,
1135
+ "learning_rate": 0.0004884310117020142,
1136
+ "loss": 3.558287353515625,
1137
+ "step": 14300
1138
+ },
1139
+ {
1140
+ "epoch": 0.288,
1141
+ "grad_norm": 0.20651493966579437,
1142
+ "learning_rate": 0.00048695612577296456,
1143
+ "loss": 3.5417572021484376,
1144
+ "step": 14400
1145
+ },
1146
+ {
1147
+ "epoch": 0.29,
1148
+ "grad_norm": 0.18496066331863403,
1149
+ "learning_rate": 0.0004854738146520113,
1150
+ "loss": 3.5497479248046875,
1151
+ "step": 14500
1152
+ },
1153
+ {
1154
+ "epoch": 0.292,
1155
+ "grad_norm": 0.23755651712417603,
1156
+ "learning_rate": 0.000483984137210959,
1157
+ "loss": 3.5615032958984374,
1158
+ "step": 14600
1159
+ },
1160
+ {
1161
+ "epoch": 0.294,
1162
+ "grad_norm": 0.20592571794986725,
1163
+ "learning_rate": 0.0004824871526141749,
1164
+ "loss": 3.5748583984375,
1165
+ "step": 14700
1166
+ },
1167
+ {
1168
+ "epoch": 0.296,
1169
+ "grad_norm": 0.2018384486436844,
1170
+ "learning_rate": 0.00048098292031623904,
1171
+ "loss": 3.575079040527344,
1172
+ "step": 14800
1173
+ },
1174
+ {
1175
+ "epoch": 0.298,
1176
+ "grad_norm": 0.19982250034809113,
1177
+ "learning_rate": 0.00047947150005958237,
1178
+ "loss": 3.56544921875,
1179
+ "step": 14900
1180
+ },
1181
+ {
1182
+ "epoch": 0.3,
1183
+ "grad_norm": 0.2055491805076599,
1184
+ "learning_rate": 0.00047795295187211473,
1185
+ "loss": 3.5550732421875,
1186
+ "step": 15000
1187
+ },
1188
+ {
1189
+ "epoch": 0.3,
1190
+ "eval_accuracy": 0.3619394445335035,
1191
+ "eval_loss": 3.490281105041504,
1192
+ "eval_runtime": 5.758,
1193
+ "eval_samples_per_second": 337.096,
1194
+ "eval_steps_per_second": 21.188,
1195
+ "step": 15000
1196
+ },
1197
+ {
1198
+ "epoch": 0.302,
1199
+ "grad_norm": 0.18866001069545746,
1200
+ "learning_rate": 0.00047642733606484064,
1201
+ "loss": 3.544368896484375,
1202
+ "step": 15100
1203
+ },
1204
+ {
1205
+ "epoch": 0.304,
1206
+ "grad_norm": 0.19551938772201538,
1207
+ "learning_rate": 0.00047489471322946335,
1208
+ "loss": 3.5462738037109376,
1209
+ "step": 15200
1210
+ },
1211
+ {
1212
+ "epoch": 0.306,
1213
+ "grad_norm": 0.20952892303466797,
1214
+ "learning_rate": 0.00047335514423597913,
1215
+ "loss": 3.547786865234375,
1216
+ "step": 15300
1217
+ },
1218
+ {
1219
+ "epoch": 0.308,
1220
+ "grad_norm": 0.1951807737350464,
1221
+ "learning_rate": 0.0004718086902302595,
1222
+ "loss": 3.5607293701171874,
1223
+ "step": 15400
1224
+ },
1225
+ {
1226
+ "epoch": 0.31,
1227
+ "grad_norm": 0.19585217535495758,
1228
+ "learning_rate": 0.00047025541263162244,
1229
+ "loss": 3.5599630737304686,
1230
+ "step": 15500
1231
+ },
1232
+ {
1233
+ "epoch": 0.312,
1234
+ "grad_norm": 0.2213257998228073,
1235
+ "learning_rate": 0.00046869537313039343,
1236
+ "loss": 3.5485220336914063,
1237
+ "step": 15600
1238
+ },
1239
+ {
1240
+ "epoch": 0.314,
1241
+ "grad_norm": 0.28433483839035034,
1242
+ "learning_rate": 0.0004671286336854554,
1243
+ "loss": 3.558450012207031,
1244
+ "step": 15700
1245
+ },
1246
+ {
1247
+ "epoch": 0.316,
1248
+ "grad_norm": 0.23959818482398987,
1249
+ "learning_rate": 0.00046555525652178736,
1250
+ "loss": 3.5676364135742187,
1251
+ "step": 15800
1252
+ },
1253
+ {
1254
+ "epoch": 0.318,
1255
+ "grad_norm": 0.20079217851161957,
1256
+ "learning_rate": 0.0004639753041279938,
1257
+ "loss": 3.552779541015625,
1258
+ "step": 15900
1259
+ },
1260
+ {
1261
+ "epoch": 0.32,
1262
+ "grad_norm": 0.20217208564281464,
1263
+ "learning_rate": 0.00046238883925382235,
1264
+ "loss": 3.537972717285156,
1265
+ "step": 16000
1266
+ },
1267
+ {
1268
+ "epoch": 0.32,
1269
+ "eval_accuracy": 0.363115024333292,
1270
+ "eval_loss": 3.475611448287964,
1271
+ "eval_runtime": 5.721,
1272
+ "eval_samples_per_second": 339.274,
1273
+ "eval_steps_per_second": 21.325,
1274
+ "step": 16000
1275
  }
1276
  ],
1277
  "logging_steps": 100,
 
1291
  "attributes": {}
1292
  }
1293
  },
1294
+ "total_flos": 5.0168070144e+17,
1295
  "train_batch_size": 120,
1296
  "trial_name": null,
1297
  "trial_params": null