CodeIsAbstract commited on
Commit
4dd25fd
·
verified ·
1 Parent(s): 3f13f0c

Training in progress, step 44000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:90615d1d2178f5b43d414dd7d578f864f00b4f2a8cd51186bc86a62375823fa5
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d51a0a5e453607425c848e0863d5a4ab36a4fc9fed901c4fb5e3d7c1d8bba2eb
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:13df368800beae9d4318c6d8b41062821af4d94ce7343c05db2b88f701598d75
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5b93cb01513134734f79ec44022214e96eaad2625ec4fee9b3ae02d90392b2e2
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1900219b8b9edacee7fac6b19a0563f8d104e8198851f8681ad0fcf475ad606d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad8615a2bc8d5c9830c43502db09bac7159496906784a5e3e8858b493b86a74e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d406c9f3bd6b7c91833c1f6b701e14e2778777027556d7f617683149adeaebed
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0bfb9abbd73e30c47a72af0932c2957ee7105fa4396a7522b57a756dacd1ff9d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2,
6
  "eval_steps": 1000,
7
- "global_step": 42000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3326,6 +3326,164 @@
3326
  "eval_samples_per_second": 222.015,
3327
  "eval_steps_per_second": 13.955,
3328
  "step": 42000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3329
  }
3330
  ],
3331
  "logging_steps": 100,
@@ -3345,7 +3503,7 @@
3345
  "attributes": {}
3346
  }
3347
  },
3348
- "total_flos": 1.64638356406272e+18,
3349
  "train_batch_size": 120,
3350
  "trial_name": null,
3351
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.24,
6
  "eval_steps": 1000,
7
+ "global_step": 44000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3326
  "eval_samples_per_second": 222.015,
3327
  "eval_steps_per_second": 13.955,
3328
  "step": 42000
3329
+ },
3330
+ {
3331
+ "epoch": 0.202,
3332
+ "grad_norm": 0.1419166624546051,
3333
+ "learning_rate": 6.0713037564567694e-05,
3334
+ "loss": 2.9582757568359375,
3335
+ "step": 42100
3336
+ },
3337
+ {
3338
+ "epoch": 0.204,
3339
+ "grad_norm": 0.14514276385307312,
3340
+ "learning_rate": 5.921681134910717e-05,
3341
+ "loss": 2.9580014038085936,
3342
+ "step": 42200
3343
+ },
3344
+ {
3345
+ "epoch": 0.206,
3346
+ "grad_norm": 0.16064871847629547,
3347
+ "learning_rate": 5.773809137876235e-05,
3348
+ "loss": 2.968863525390625,
3349
+ "step": 42300
3350
+ },
3351
+ {
3352
+ "epoch": 0.208,
3353
+ "grad_norm": 0.14985115826129913,
3354
+ "learning_rate": 5.6276936382711144e-05,
3355
+ "loss": 2.9703680419921876,
3356
+ "step": 42400
3357
+ },
3358
+ {
3359
+ "epoch": 0.21,
3360
+ "grad_norm": 0.14414609968662262,
3361
+ "learning_rate": 5.483340439251644e-05,
3362
+ "loss": 2.982804870605469,
3363
+ "step": 42500
3364
+ },
3365
+ {
3366
+ "epoch": 0.212,
3367
+ "grad_norm": 0.16357700526714325,
3368
+ "learning_rate": 5.340755273982273e-05,
3369
+ "loss": 2.9501416015625,
3370
+ "step": 42600
3371
+ },
3372
+ {
3373
+ "epoch": 0.214,
3374
+ "grad_norm": 0.1366511881351471,
3375
+ "learning_rate": 5.199943805407742e-05,
3376
+ "loss": 2.971943664550781,
3377
+ "step": 42700
3378
+ },
3379
+ {
3380
+ "epoch": 0.216,
3381
+ "grad_norm": 0.1515188068151474,
3382
+ "learning_rate": 5.0609116260282864e-05,
3383
+ "loss": 2.978944091796875,
3384
+ "step": 42800
3385
+ },
3386
+ {
3387
+ "epoch": 0.218,
3388
+ "grad_norm": 0.13808998465538025,
3389
+ "learning_rate": 4.9236642576774906e-05,
3390
+ "loss": 2.962769775390625,
3391
+ "step": 42900
3392
+ },
3393
+ {
3394
+ "epoch": 0.22,
3395
+ "grad_norm": 0.17845956981182098,
3396
+ "learning_rate": 4.788207151302964e-05,
3397
+ "loss": 2.94796142578125,
3398
+ "step": 43000
3399
+ },
3400
+ {
3401
+ "epoch": 0.22,
3402
+ "eval_accuracy": 0.38396997129609184,
3403
+ "eval_loss": 3.2809572219848633,
3404
+ "eval_runtime": 8.8116,
3405
+ "eval_samples_per_second": 220.278,
3406
+ "eval_steps_per_second": 13.845,
3407
+ "step": 43000
3408
+ },
3409
+ {
3410
+ "epoch": 0.222,
3411
+ "grad_norm": 0.15073730051517487,
3412
+ "learning_rate": 4.654545686749872e-05,
3413
+ "loss": 2.980477600097656,
3414
+ "step": 43100
3415
+ },
3416
+ {
3417
+ "epoch": 0.224,
3418
+ "grad_norm": 0.14061863720417023,
3419
+ "learning_rate": 4.522685172547269e-05,
3420
+ "loss": 2.951471252441406,
3421
+ "step": 43200
3422
+ },
3423
+ {
3424
+ "epoch": 0.226,
3425
+ "grad_norm": 0.14917391538619995,
3426
+ "learning_rate": 4.3926308456972596e-05,
3427
+ "loss": 2.9555926513671875,
3428
+ "step": 43300
3429
+ },
3430
+ {
3431
+ "epoch": 0.228,
3432
+ "grad_norm": 0.15331073105335236,
3433
+ "learning_rate": 4.264387871466963e-05,
3434
+ "loss": 2.9574981689453126,
3435
+ "step": 43400
3436
+ },
3437
+ {
3438
+ "epoch": 0.23,
3439
+ "grad_norm": 0.15429548919200897,
3440
+ "learning_rate": 4.137961343183472e-05,
3441
+ "loss": 2.9603759765625,
3442
+ "step": 43500
3443
+ },
3444
+ {
3445
+ "epoch": 0.232,
3446
+ "grad_norm": 0.15303488075733185,
3447
+ "learning_rate": 4.0133562820314385e-05,
3448
+ "loss": 2.9570083618164062,
3449
+ "step": 43600
3450
+ },
3451
+ {
3452
+ "epoch": 0.234,
3453
+ "grad_norm": 0.16372403502464294,
3454
+ "learning_rate": 3.8905776368537426e-05,
3455
+ "loss": 2.970541687011719,
3456
+ "step": 43700
3457
+ },
3458
+ {
3459
+ "epoch": 0.236,
3460
+ "grad_norm": 0.1566547304391861,
3461
+ "learning_rate": 3.7696302839549336e-05,
3462
+ "loss": 2.9552093505859376,
3463
+ "step": 43800
3464
+ },
3465
+ {
3466
+ "epoch": 0.238,
3467
+ "grad_norm": 0.15626369416713715,
3468
+ "learning_rate": 3.650519026907484e-05,
3469
+ "loss": 2.954615478515625,
3470
+ "step": 43900
3471
+ },
3472
+ {
3473
+ "epoch": 0.24,
3474
+ "grad_norm": 0.14116933941841125,
3475
+ "learning_rate": 3.5332485963611217e-05,
3476
+ "loss": 2.963729248046875,
3477
+ "step": 44000
3478
+ },
3479
+ {
3480
+ "epoch": 0.24,
3481
+ "eval_accuracy": 0.38423714852331653,
3482
+ "eval_loss": 3.2790355682373047,
3483
+ "eval_runtime": 8.6003,
3484
+ "eval_samples_per_second": 225.69,
3485
+ "eval_steps_per_second": 14.186,
3486
+ "step": 44000
3487
  }
3488
  ],
3489
  "logging_steps": 100,
 
3503
  "attributes": {}
3504
  }
3505
  },
3506
+ "total_flos": 1.72478278139904e+18,
3507
  "train_batch_size": 120,
3508
  "trial_name": null,
3509
  "trial_params": null