CodeIsAbstract commited on
Commit
f8d4b62
·
verified ·
1 Parent(s): 4dfbd30

Training in progress, step 44000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:effd2d14476df74bc5f8738835a04cd2c53990d3720280723645c691281e29a7
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c1b4eb62e0fcdc8d3a810aceb36afb64f22f1ea6d26a83aa96829502ed3bddd7
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7c29b0dac895207221ca2ff233a263d1f8bafafacb0961d5bd379e4086db6f23
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a3db83ad78622f7a5388f826ec38184d84eaea94ecb11d8534774dd1bc46ce45
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ee2b9e8d80dbc1551afe65580bf18f91ce6a2908c043aa9c60a24adad8385acc
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:752c23b16f0b286a12046bd803bd32f280dd78420031d337d9d0b8ad8d9c480b
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7778da91eeef8b8307a814fd6b609cdd038d764b06717fd843f1619aa9b55603
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:419a73c41e3de0ebe2db15bfe3b49945cbd97c4de2356752fbadf096577c3d0b
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.03636363636363636,
6
  "eval_steps": 1000,
7
- "global_step": 40000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3128,6 +3128,318 @@
3128
  "eval_samples_per_second": 76.703,
3129
  "eval_steps_per_second": 19.176,
3130
  "step": 40000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3131
  }
3132
  ],
3133
  "logging_steps": 100,
@@ -3147,7 +3459,7 @@
3147
  "attributes": {}
3148
  }
3149
  },
3150
- "total_flos": 9.9658323984384e+17,
3151
  "train_batch_size": 22,
3152
  "trial_name": null,
3153
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.07272727272727272,
6
  "eval_steps": 1000,
7
+ "global_step": 44000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3128
  "eval_samples_per_second": 76.703,
3129
  "eval_steps_per_second": 19.176,
3130
  "step": 40000
3131
+ },
3132
+ {
3133
+ "epoch": 0.03727272727272727,
3134
+ "grad_norm": 0.15226680040359497,
3135
+ "learning_rate": 0.0007076613888136895,
3136
+ "loss": 2.7523992919921874,
3137
+ "step": 40100
3138
+ },
3139
+ {
3140
+ "epoch": 0.038181818181818185,
3141
+ "grad_norm": 0.16509854793548584,
3142
+ "learning_rate": 0.0007063597559645352,
3143
+ "loss": 2.7390933227539063,
3144
+ "step": 40200
3145
+ },
3146
+ {
3147
+ "epoch": 0.03909090909090909,
3148
+ "grad_norm": 0.1810932159423828,
3149
+ "learning_rate": 0.0007050564353023619,
3150
+ "loss": 2.7575619506835936,
3151
+ "step": 40300
3152
+ },
3153
+ {
3154
+ "epoch": 0.04,
3155
+ "grad_norm": 0.14917488396167755,
3156
+ "learning_rate": 0.0007037514374870077,
3157
+ "loss": 2.700941162109375,
3158
+ "step": 40400
3159
+ },
3160
+ {
3161
+ "epoch": 0.04090909090909091,
3162
+ "grad_norm": 0.1438620686531067,
3163
+ "learning_rate": 0.0007024447731920284,
3164
+ "loss": 2.7219302368164064,
3165
+ "step": 40500
3166
+ },
3167
+ {
3168
+ "epoch": 0.04181818181818182,
3169
+ "grad_norm": 0.1588471382856369,
3170
+ "learning_rate": 0.000701136453104609,
3171
+ "loss": 2.7354092407226562,
3172
+ "step": 40600
3173
+ },
3174
+ {
3175
+ "epoch": 0.042727272727272725,
3176
+ "grad_norm": 0.18232181668281555,
3177
+ "learning_rate": 0.0006998264879254782,
3178
+ "loss": 2.7257977294921876,
3179
+ "step": 40700
3180
+ },
3181
+ {
3182
+ "epoch": 0.04363636363636364,
3183
+ "grad_norm": 0.15463081002235413,
3184
+ "learning_rate": 0.0006985148883688194,
3185
+ "loss": 2.7430325317382813,
3186
+ "step": 40800
3187
+ },
3188
+ {
3189
+ "epoch": 0.04454545454545455,
3190
+ "grad_norm": 0.16292454302310944,
3191
+ "learning_rate": 0.0006972016651621834,
3192
+ "loss": 2.725971374511719,
3193
+ "step": 40900
3194
+ },
3195
+ {
3196
+ "epoch": 0.045454545454545456,
3197
+ "grad_norm": 0.14817778766155243,
3198
+ "learning_rate": 0.0006958868290464014,
3199
+ "loss": 2.7598446655273436,
3200
+ "step": 41000
3201
+ },
3202
+ {
3203
+ "epoch": 0.045454545454545456,
3204
+ "eval_loss": 3.1211743354797363,
3205
+ "eval_runtime": 7.424,
3206
+ "eval_samples_per_second": 77.586,
3207
+ "eval_steps_per_second": 19.396,
3208
+ "step": 41000
3209
+ },
3210
+ {
3211
+ "epoch": 0.046363636363636364,
3212
+ "grad_norm": 0.1458057463169098,
3213
+ "learning_rate": 0.0006945703907754957,
3214
+ "loss": 2.7241558837890625,
3215
+ "step": 41100
3216
+ },
3217
+ {
3218
+ "epoch": 0.04727272727272727,
3219
+ "grad_norm": 0.1644594669342041,
3220
+ "learning_rate": 0.0006932523611165934,
3221
+ "loss": 2.737247314453125,
3222
+ "step": 41200
3223
+ },
3224
+ {
3225
+ "epoch": 0.04818181818181818,
3226
+ "grad_norm": 0.1583612859249115,
3227
+ "learning_rate": 0.000691932750849837,
3228
+ "loss": 2.7208633422851562,
3229
+ "step": 41300
3230
+ },
3231
+ {
3232
+ "epoch": 0.04909090909090909,
3233
+ "grad_norm": 0.15082280337810516,
3234
+ "learning_rate": 0.000690611570768297,
3235
+ "loss": 2.7353948974609374,
3236
+ "step": 41400
3237
+ },
3238
+ {
3239
+ "epoch": 0.05,
3240
+ "grad_norm": 0.16890287399291992,
3241
+ "learning_rate": 0.0006892888316778836,
3242
+ "loss": 2.7053826904296874,
3243
+ "step": 41500
3244
+ },
3245
+ {
3246
+ "epoch": 0.05090909090909091,
3247
+ "grad_norm": 0.15229052305221558,
3248
+ "learning_rate": 0.0006879645443972575,
3249
+ "loss": 2.6989901733398436,
3250
+ "step": 41600
3251
+ },
3252
+ {
3253
+ "epoch": 0.05181818181818182,
3254
+ "grad_norm": 0.15202166140079498,
3255
+ "learning_rate": 0.0006866387197577427,
3256
+ "loss": 2.7251095581054687,
3257
+ "step": 41700
3258
+ },
3259
+ {
3260
+ "epoch": 0.05272727272727273,
3261
+ "grad_norm": 0.1463591307401657,
3262
+ "learning_rate": 0.0006853113686032368,
3263
+ "loss": 2.7278387451171877,
3264
+ "step": 41800
3265
+ },
3266
+ {
3267
+ "epoch": 0.053636363636363635,
3268
+ "grad_norm": 0.14916226267814636,
3269
+ "learning_rate": 0.0006839825017901229,
3270
+ "loss": 2.71453369140625,
3271
+ "step": 41900
3272
+ },
3273
+ {
3274
+ "epoch": 0.05454545454545454,
3275
+ "grad_norm": 0.15262916684150696,
3276
+ "learning_rate": 0.0006826521301871804,
3277
+ "loss": 2.712431640625,
3278
+ "step": 42000
3279
+ },
3280
+ {
3281
+ "epoch": 0.05454545454545454,
3282
+ "eval_loss": 3.1164402961730957,
3283
+ "eval_runtime": 7.416,
3284
+ "eval_samples_per_second": 77.67,
3285
+ "eval_steps_per_second": 19.418,
3286
+ "step": 42000
3287
+ },
3288
+ {
3289
+ "epoch": 0.05545454545454546,
3290
+ "grad_norm": 0.14367501437664032,
3291
+ "learning_rate": 0.0006813202646754969,
3292
+ "loss": 2.770457458496094,
3293
+ "step": 42100
3294
+ },
3295
+ {
3296
+ "epoch": 0.056363636363636366,
3297
+ "grad_norm": 0.16685152053833008,
3298
+ "learning_rate": 0.000679986916148378,
3299
+ "loss": 2.7458120727539064,
3300
+ "step": 42200
3301
+ },
3302
+ {
3303
+ "epoch": 0.057272727272727274,
3304
+ "grad_norm": 0.1416223794221878,
3305
+ "learning_rate": 0.0006786520955112592,
3306
+ "loss": 2.737471923828125,
3307
+ "step": 42300
3308
+ },
3309
+ {
3310
+ "epoch": 0.05818181818181818,
3311
+ "grad_norm": 0.19099284708499908,
3312
+ "learning_rate": 0.0006773158136816167,
3313
+ "loss": 2.7410955810546875,
3314
+ "step": 42400
3315
+ },
3316
+ {
3317
+ "epoch": 0.05909090909090909,
3318
+ "grad_norm": 0.14040355384349823,
3319
+ "learning_rate": 0.0006759780815888771,
3320
+ "loss": 2.7524224853515626,
3321
+ "step": 42500
3322
+ },
3323
+ {
3324
+ "epoch": 0.06,
3325
+ "grad_norm": 0.14514034986495972,
3326
+ "learning_rate": 0.0006746389101743291,
3327
+ "loss": 2.7450900268554688,
3328
+ "step": 42600
3329
+ },
3330
+ {
3331
+ "epoch": 0.060909090909090906,
3332
+ "grad_norm": 0.1578412503004074,
3333
+ "learning_rate": 0.0006732983103910333,
3334
+ "loss": 2.7615512084960936,
3335
+ "step": 42700
3336
+ },
3337
+ {
3338
+ "epoch": 0.06181818181818182,
3339
+ "grad_norm": 0.18139956891536713,
3340
+ "learning_rate": 0.0006719562932037332,
3341
+ "loss": 2.709564514160156,
3342
+ "step": 42800
3343
+ },
3344
+ {
3345
+ "epoch": 0.06272727272727273,
3346
+ "grad_norm": 0.15958115458488464,
3347
+ "learning_rate": 0.000670612869588765,
3348
+ "loss": 2.7086691284179687,
3349
+ "step": 42900
3350
+ },
3351
+ {
3352
+ "epoch": 0.06363636363636363,
3353
+ "grad_norm": 0.18487316370010376,
3354
+ "learning_rate": 0.0006692680505339684,
3355
+ "loss": 2.7571722412109376,
3356
+ "step": 43000
3357
+ },
3358
+ {
3359
+ "epoch": 0.06363636363636363,
3360
+ "eval_loss": 3.1164326667785645,
3361
+ "eval_runtime": 7.4455,
3362
+ "eval_samples_per_second": 77.362,
3363
+ "eval_steps_per_second": 19.34,
3364
+ "step": 43000
3365
+ },
3366
+ {
3367
+ "epoch": 0.06454545454545454,
3368
+ "grad_norm": 0.14378763735294342,
3369
+ "learning_rate": 0.0006679218470385959,
3370
+ "loss": 2.713703308105469,
3371
+ "step": 43100
3372
+ },
3373
+ {
3374
+ "epoch": 0.06545454545454546,
3375
+ "grad_norm": 0.14831411838531494,
3376
+ "learning_rate": 0.0006665742701132233,
3377
+ "loss": 2.7057376098632813,
3378
+ "step": 43200
3379
+ },
3380
+ {
3381
+ "epoch": 0.06636363636363636,
3382
+ "grad_norm": 0.14014075696468353,
3383
+ "learning_rate": 0.0006652253307796605,
3384
+ "loss": 2.732189025878906,
3385
+ "step": 43300
3386
+ },
3387
+ {
3388
+ "epoch": 0.06727272727272728,
3389
+ "grad_norm": 0.1462206244468689,
3390
+ "learning_rate": 0.0006638750400708597,
3391
+ "loss": 2.735457763671875,
3392
+ "step": 43400
3393
+ },
3394
+ {
3395
+ "epoch": 0.06818181818181818,
3396
+ "grad_norm": 0.15208710730075836,
3397
+ "learning_rate": 0.0006625234090308261,
3398
+ "loss": 2.7204766845703126,
3399
+ "step": 43500
3400
+ },
3401
+ {
3402
+ "epoch": 0.06909090909090909,
3403
+ "grad_norm": 0.17330685257911682,
3404
+ "learning_rate": 0.0006611704487145273,
3405
+ "loss": 2.7174346923828123,
3406
+ "step": 43600
3407
+ },
3408
+ {
3409
+ "epoch": 0.07,
3410
+ "grad_norm": 0.14701761305332184,
3411
+ "learning_rate": 0.000659816170187804,
3412
+ "loss": 2.7171310424804687,
3413
+ "step": 43700
3414
+ },
3415
+ {
3416
+ "epoch": 0.07090909090909091,
3417
+ "grad_norm": 0.15500852465629578,
3418
+ "learning_rate": 0.0006584605845272769,
3419
+ "loss": 2.7264663696289064,
3420
+ "step": 43800
3421
+ },
3422
+ {
3423
+ "epoch": 0.07181818181818182,
3424
+ "grad_norm": 0.173856720328331,
3425
+ "learning_rate": 0.0006571037028202593,
3426
+ "loss": 2.7247988891601564,
3427
+ "step": 43900
3428
+ },
3429
+ {
3430
+ "epoch": 0.07272727272727272,
3431
+ "grad_norm": 0.152163565158844,
3432
+ "learning_rate": 0.0006557455361646638,
3433
+ "loss": 2.7246688842773437,
3434
+ "step": 44000
3435
+ },
3436
+ {
3437
+ "epoch": 0.07272727272727272,
3438
+ "eval_loss": 3.1064937114715576,
3439
+ "eval_runtime": 7.4422,
3440
+ "eval_samples_per_second": 77.396,
3441
+ "eval_steps_per_second": 19.349,
3442
+ "step": 44000
3443
  }
3444
  ],
3445
  "logging_steps": 100,
 
3459
  "attributes": {}
3460
  }
3461
  },
3462
+ "total_flos": 1.096241563828224e+18,
3463
  "train_batch_size": 22,
3464
  "trial_name": null,
3465
  "trial_params": null