CodeIsAbstract commited on
Commit
487de7a
·
verified ·
1 Parent(s): c88270e

Training in progress, step 56000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4e7ba8d2524bc9f928490251d5557e2326e4ee76b8165e53f83fb0334d18ddfe
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bbc019d20c5bc5141a3e9004bef5a52531dc2a92512d93ad8455f732db0eb48d
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7e7abd1b4dbd4f064e3df76d3ce96cc640fdb83fe12440e34c4e94de5dc9ac01
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd70cbbec6d1891a81fb67c9296a4bf20c7702703681faa48b7e22206072facc
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:99fd3750e2e46b63e97783e1c2d11f0a6d652ae4d1ee48cc27594356ed0fe238
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fcedeb5e64ebf070876b0c1c4c22aebd04ef9acb651910f4b487df10455e11d
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c8715411d67584e8b8ee5de5af78cfc1a263656fc2e3e52390a5b49a21234cc2
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:46e948248514973c2ef1b72e4ce8dd6fba8939559c40be1ef19c251df415a7e3
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.14545454545454545,
6
  "eval_steps": 1000,
7
- "global_step": 52000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -4064,6 +4064,318 @@
4064
  "eval_samples_per_second": 77.409,
4065
  "eval_steps_per_second": 19.352,
4066
  "step": 52000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4067
  }
4068
  ],
4069
  "logging_steps": 100,
@@ -4083,7 +4395,7 @@
4083
  "attributes": {}
4084
  }
4085
  },
4086
- "total_flos": 1.295558211796992e+18,
4087
  "train_batch_size": 22,
4088
  "trial_name": null,
4089
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.18181818181818182,
6
  "eval_steps": 1000,
7
+ "global_step": 56000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
4064
  "eval_samples_per_second": 77.409,
4065
  "eval_steps_per_second": 19.352,
4066
  "step": 52000
4067
+ },
4068
+ {
4069
+ "epoch": 0.14636363636363636,
4070
+ "grad_norm": 0.15897710621356964,
4071
+ "learning_rate": 0.0005425038522701411,
4072
+ "loss": 2.6896728515625,
4073
+ "step": 52100
4074
+ },
4075
+ {
4076
+ "epoch": 0.14727272727272728,
4077
+ "grad_norm": 0.1507830172777176,
4078
+ "learning_rate": 0.0005410789097916161,
4079
+ "loss": 2.661712341308594,
4080
+ "step": 52200
4081
+ },
4082
+ {
4083
+ "epoch": 0.1481818181818182,
4084
+ "grad_norm": 0.1556272655725479,
4085
+ "learning_rate": 0.0005396536313293696,
4086
+ "loss": 2.6831851196289063,
4087
+ "step": 52300
4088
+ },
4089
+ {
4090
+ "epoch": 0.14909090909090908,
4091
+ "grad_norm": 0.15099219977855682,
4092
+ "learning_rate": 0.0005382280285407307,
4093
+ "loss": 2.6684881591796876,
4094
+ "step": 52400
4095
+ },
4096
+ {
4097
+ "epoch": 0.15,
4098
+ "grad_norm": 0.15491120517253876,
4099
+ "learning_rate": 0.0005368021130856809,
4100
+ "loss": 2.673580322265625,
4101
+ "step": 52500
4102
+ },
4103
+ {
4104
+ "epoch": 0.1509090909090909,
4105
+ "grad_norm": 0.1487579345703125,
4106
+ "learning_rate": 0.0005353758966267588,
4107
+ "loss": 2.690845642089844,
4108
+ "step": 52600
4109
+ },
4110
+ {
4111
+ "epoch": 0.15181818181818182,
4112
+ "grad_norm": 0.1554812639951706,
4113
+ "learning_rate": 0.0005339493908289656,
4114
+ "loss": 2.67916748046875,
4115
+ "step": 52700
4116
+ },
4117
+ {
4118
+ "epoch": 0.15272727272727274,
4119
+ "grad_norm": 0.15371479094028473,
4120
+ "learning_rate": 0.0005325226073596681,
4121
+ "loss": 2.6689898681640627,
4122
+ "step": 52800
4123
+ },
4124
+ {
4125
+ "epoch": 0.15363636363636363,
4126
+ "grad_norm": 0.21308736503124237,
4127
+ "learning_rate": 0.0005310955578885049,
4128
+ "loss": 2.696147766113281,
4129
+ "step": 52900
4130
+ },
4131
+ {
4132
+ "epoch": 0.15454545454545454,
4133
+ "grad_norm": 0.15344563126564026,
4134
+ "learning_rate": 0.0005296682540872898,
4135
+ "loss": 2.6800167846679686,
4136
+ "step": 53000
4137
+ },
4138
+ {
4139
+ "epoch": 0.15454545454545454,
4140
+ "eval_loss": 3.073288917541504,
4141
+ "eval_runtime": 7.4514,
4142
+ "eval_samples_per_second": 77.301,
4143
+ "eval_steps_per_second": 19.325,
4144
+ "step": 53000
4145
+ },
4146
+ {
4147
+ "epoch": 0.15545454545454546,
4148
+ "grad_norm": 0.14833194017410278,
4149
+ "learning_rate": 0.000528240707629917,
4150
+ "loss": 2.6591696166992187,
4151
+ "step": 53100
4152
+ },
4153
+ {
4154
+ "epoch": 0.15636363636363637,
4155
+ "grad_norm": 0.200588196516037,
4156
+ "learning_rate": 0.0005268129301922651,
4157
+ "loss": 2.652269287109375,
4158
+ "step": 53200
4159
+ },
4160
+ {
4161
+ "epoch": 0.1572727272727273,
4162
+ "grad_norm": 0.1412370204925537,
4163
+ "learning_rate": 0.0005253849334521023,
4164
+ "loss": 2.674943542480469,
4165
+ "step": 53300
4166
+ },
4167
+ {
4168
+ "epoch": 0.15818181818181817,
4169
+ "grad_norm": 0.1481807678937912,
4170
+ "learning_rate": 0.0005239567290889901,
4171
+ "loss": 2.6497378540039063,
4172
+ "step": 53400
4173
+ },
4174
+ {
4175
+ "epoch": 0.1590909090909091,
4176
+ "grad_norm": 0.1553225964307785,
4177
+ "learning_rate": 0.0005225283287841883,
4178
+ "loss": 2.655411682128906,
4179
+ "step": 53500
4180
+ },
4181
+ {
4182
+ "epoch": 0.16,
4183
+ "grad_norm": 0.1549694985151291,
4184
+ "learning_rate": 0.0005210997442205594,
4185
+ "loss": 2.654962158203125,
4186
+ "step": 53600
4187
+ },
4188
+ {
4189
+ "epoch": 0.16090909090909092,
4190
+ "grad_norm": 0.1860799938440323,
4191
+ "learning_rate": 0.0005196709870824726,
4192
+ "loss": 2.6886834716796875,
4193
+ "step": 53700
4194
+ },
4195
+ {
4196
+ "epoch": 0.1618181818181818,
4197
+ "grad_norm": 0.14446097612380981,
4198
+ "learning_rate": 0.0005182420690557091,
4199
+ "loss": 2.6884750366210937,
4200
+ "step": 53800
4201
+ },
4202
+ {
4203
+ "epoch": 0.16272727272727272,
4204
+ "grad_norm": 0.14465349912643433,
4205
+ "learning_rate": 0.0005168130018273655,
4206
+ "loss": 2.677554931640625,
4207
+ "step": 53900
4208
+ },
4209
+ {
4210
+ "epoch": 0.16363636363636364,
4211
+ "grad_norm": 0.14523738622665405,
4212
+ "learning_rate": 0.0005153837970857591,
4213
+ "loss": 2.66891357421875,
4214
+ "step": 54000
4215
+ },
4216
+ {
4217
+ "epoch": 0.16363636363636364,
4218
+ "eval_loss": 3.070340394973755,
4219
+ "eval_runtime": 7.4203,
4220
+ "eval_samples_per_second": 77.625,
4221
+ "eval_steps_per_second": 19.406,
4222
+ "step": 54000
4223
+ },
4224
+ {
4225
+ "epoch": 0.16454545454545455,
4226
+ "grad_norm": 0.14817142486572266,
4227
+ "learning_rate": 0.0005139544665203315,
4228
+ "loss": 2.69839599609375,
4229
+ "step": 54100
4230
+ },
4231
+ {
4232
+ "epoch": 0.16545454545454547,
4233
+ "grad_norm": 0.18234474956989288,
4234
+ "learning_rate": 0.0005125250218215538,
4235
+ "loss": 2.667855224609375,
4236
+ "step": 54200
4237
+ },
4238
+ {
4239
+ "epoch": 0.16636363636363635,
4240
+ "grad_norm": 0.16814777255058289,
4241
+ "learning_rate": 0.0005110954746808307,
4242
+ "loss": 2.683269348144531,
4243
+ "step": 54300
4244
+ },
4245
+ {
4246
+ "epoch": 0.16727272727272727,
4247
+ "grad_norm": 0.1632496416568756,
4248
+ "learning_rate": 0.0005096658367904042,
4249
+ "loss": 2.670632629394531,
4250
+ "step": 54400
4251
+ },
4252
+ {
4253
+ "epoch": 0.16818181818181818,
4254
+ "grad_norm": 0.15290036797523499,
4255
+ "learning_rate": 0.0005082361198432592,
4256
+ "loss": 2.6895196533203123,
4257
+ "step": 54500
4258
+ },
4259
+ {
4260
+ "epoch": 0.1690909090909091,
4261
+ "grad_norm": 0.15638327598571777,
4262
+ "learning_rate": 0.0005068063355330264,
4263
+ "loss": 2.676280212402344,
4264
+ "step": 54600
4265
+ },
4266
+ {
4267
+ "epoch": 0.17,
4268
+ "grad_norm": 0.14327116310596466,
4269
+ "learning_rate": 0.0005053764955538885,
4270
+ "loss": 2.693046875,
4271
+ "step": 54700
4272
+ },
4273
+ {
4274
+ "epoch": 0.1709090909090909,
4275
+ "grad_norm": 0.1540258675813675,
4276
+ "learning_rate": 0.0005039466116004827,
4277
+ "loss": 2.6647442626953124,
4278
+ "step": 54800
4279
+ },
4280
+ {
4281
+ "epoch": 0.17181818181818181,
4282
+ "grad_norm": 0.14872531592845917,
4283
+ "learning_rate": 0.0005025166953678061,
4284
+ "loss": 2.669239501953125,
4285
+ "step": 54900
4286
+ },
4287
+ {
4288
+ "epoch": 0.17272727272727273,
4289
+ "grad_norm": 0.15456752479076385,
4290
+ "learning_rate": 0.0005010867585511199,
4291
+ "loss": 2.6779644775390623,
4292
+ "step": 55000
4293
+ },
4294
+ {
4295
+ "epoch": 0.17272727272727273,
4296
+ "eval_loss": 3.070594072341919,
4297
+ "eval_runtime": 7.4584,
4298
+ "eval_samples_per_second": 77.229,
4299
+ "eval_steps_per_second": 19.307,
4300
+ "step": 55000
4301
+ },
4302
+ {
4303
+ "epoch": 0.17363636363636364,
4304
+ "grad_norm": 0.14803044497966766,
4305
+ "learning_rate": 0.0004996568128458535,
4306
+ "loss": 2.698114013671875,
4307
+ "step": 55100
4308
+ },
4309
+ {
4310
+ "epoch": 0.17454545454545456,
4311
+ "grad_norm": 0.14924941956996918,
4312
+ "learning_rate": 0.0004982268699475092,
4313
+ "loss": 2.638841247558594,
4314
+ "step": 55200
4315
+ },
4316
+ {
4317
+ "epoch": 0.17545454545454545,
4318
+ "grad_norm": 0.15640807151794434,
4319
+ "learning_rate": 0.0004967969415515659,
4320
+ "loss": 2.6517404174804686,
4321
+ "step": 55300
4322
+ },
4323
+ {
4324
+ "epoch": 0.17636363636363636,
4325
+ "grad_norm": 0.18078777194023132,
4326
+ "learning_rate": 0.0004953670393533846,
4327
+ "loss": 2.6669937133789063,
4328
+ "step": 55400
4329
+ },
4330
+ {
4331
+ "epoch": 0.17727272727272728,
4332
+ "grad_norm": 0.15050143003463745,
4333
+ "learning_rate": 0.0004939371750481115,
4334
+ "loss": 2.6495355224609374,
4335
+ "step": 55500
4336
+ },
4337
+ {
4338
+ "epoch": 0.1781818181818182,
4339
+ "grad_norm": 0.15939530730247498,
4340
+ "learning_rate": 0.0004925073603305833,
4341
+ "loss": 2.673974609375,
4342
+ "step": 55600
4343
+ },
4344
+ {
4345
+ "epoch": 0.17909090909090908,
4346
+ "grad_norm": 0.7551309466362,
4347
+ "learning_rate": 0.0004910776068952304,
4348
+ "loss": 2.6610736083984374,
4349
+ "step": 55700
4350
+ },
4351
+ {
4352
+ "epoch": 0.18,
4353
+ "grad_norm": 0.14220239222049713,
4354
+ "learning_rate": 0.0004896479264359827,
4355
+ "loss": 2.7120089721679688,
4356
+ "step": 55800
4357
+ },
4358
+ {
4359
+ "epoch": 0.1809090909090909,
4360
+ "grad_norm": 0.17650385200977325,
4361
+ "learning_rate": 0.0004882183306461728,
4362
+ "loss": 2.6712158203125,
4363
+ "step": 55900
4364
+ },
4365
+ {
4366
+ "epoch": 0.18181818181818182,
4367
+ "grad_norm": 0.15496277809143066,
4368
+ "learning_rate": 0.00048678883121844096,
4369
+ "loss": 2.6805792236328125,
4370
+ "step": 56000
4371
+ },
4372
+ {
4373
+ "epoch": 0.18181818181818182,
4374
+ "eval_loss": 3.068009614944458,
4375
+ "eval_runtime": 7.4586,
4376
+ "eval_samples_per_second": 77.227,
4377
+ "eval_steps_per_second": 19.307,
4378
+ "step": 56000
4379
  }
4380
  ],
4381
  "logging_steps": 100,
 
4395
  "attributes": {}
4396
  }
4397
  },
4398
+ "total_flos": 1.395216535781376e+18,
4399
  "train_batch_size": 22,
4400
  "trial_name": null,
4401
  "trial_params": null