CodeIsAbstract commited on
Commit
90134e1
·
verified ·
1 Parent(s): d258182

Training in progress, step 48000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c1b4eb62e0fcdc8d3a810aceb36afb64f22f1ea6d26a83aa96829502ed3bddd7
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6bc9d91d9a60b7700203af96a974794d27be7eea3f420ff1ae5b370210ee2ea
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a3db83ad78622f7a5388f826ec38184d84eaea94ecb11d8534774dd1bc46ce45
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:84e7123b1ce19594fbb22be55c4498bf5de4fe017896adc3ad66d3524c9c9c54
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:752c23b16f0b286a12046bd803bd32f280dd78420031d337d9d0b8ad8d9c480b
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9623f80738737f22c717a4262aa756fc5e2cd7d18d0d2842f53148de45dc81ec
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:419a73c41e3de0ebe2db15bfe3b49945cbd97c4de2356752fbadf096577c3d0b
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2acdb6c434c9b8d7057c6d627425c7ddfd0a206d98e9d43b0c3e164d01587a81
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.07272727272727272,
6
  "eval_steps": 1000,
7
- "global_step": 44000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -3440,6 +3440,318 @@
3440
  "eval_samples_per_second": 77.396,
3441
  "eval_steps_per_second": 19.349,
3442
  "step": 44000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3443
  }
3444
  ],
3445
  "logging_steps": 100,
@@ -3459,7 +3771,7 @@
3459
  "attributes": {}
3460
  }
3461
  },
3462
- "total_flos": 1.096241563828224e+18,
3463
  "train_batch_size": 22,
3464
  "trial_name": null,
3465
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.10909090909090909,
6
  "eval_steps": 1000,
7
+ "global_step": 48000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
3440
  "eval_samples_per_second": 77.396,
3441
  "eval_steps_per_second": 19.349,
3442
  "step": 44000
3443
+ },
3444
+ {
3445
+ "epoch": 0.07363636363636364,
3446
+ "grad_norm": 0.14658956229686737,
3447
+ "learning_rate": 0.0006543860956689131,
3448
+ "loss": 2.7102786254882814,
3449
+ "step": 44100
3450
+ },
3451
+ {
3452
+ "epoch": 0.07454545454545454,
3453
+ "grad_norm": 0.1424274891614914,
3454
+ "learning_rate": 0.0006530253924518484,
3455
+ "loss": 2.682833557128906,
3456
+ "step": 44200
3457
+ },
3458
+ {
3459
+ "epoch": 0.07545454545454545,
3460
+ "grad_norm": 0.1558665931224823,
3461
+ "learning_rate": 0.0006516634376426387,
3462
+ "loss": 2.7235848999023435,
3463
+ "step": 44300
3464
+ },
3465
+ {
3466
+ "epoch": 0.07636363636363637,
3467
+ "grad_norm": 0.1660631000995636,
3468
+ "learning_rate": 0.0006503002423806897,
3469
+ "loss": 2.6941436767578124,
3470
+ "step": 44400
3471
+ },
3472
+ {
3473
+ "epoch": 0.07727272727272727,
3474
+ "grad_norm": 0.15640541911125183,
3475
+ "learning_rate": 0.0006489358178155531,
3476
+ "loss": 2.7176858520507814,
3477
+ "step": 44500
3478
+ },
3479
+ {
3480
+ "epoch": 0.07818181818181819,
3481
+ "grad_norm": 0.1507701575756073,
3482
+ "learning_rate": 0.0006475701751068345,
3483
+ "loss": 2.72189697265625,
3484
+ "step": 44600
3485
+ },
3486
+ {
3487
+ "epoch": 0.07909090909090909,
3488
+ "grad_norm": 0.1533944010734558,
3489
+ "learning_rate": 0.000646203325424103,
3490
+ "loss": 2.6939328002929686,
3491
+ "step": 44700
3492
+ },
3493
+ {
3494
+ "epoch": 0.08,
3495
+ "grad_norm": 0.13101057708263397,
3496
+ "learning_rate": 0.0006448352799467994,
3497
+ "loss": 2.743724365234375,
3498
+ "step": 44800
3499
+ },
3500
+ {
3501
+ "epoch": 0.0809090909090909,
3502
+ "grad_norm": 0.15205194056034088,
3503
+ "learning_rate": 0.0006434660498641453,
3504
+ "loss": 2.6808477783203126,
3505
+ "step": 44900
3506
+ },
3507
+ {
3508
+ "epoch": 0.08181818181818182,
3509
+ "grad_norm": 0.16786165535449982,
3510
+ "learning_rate": 0.0006420956463750505,
3511
+ "loss": 2.7101092529296875,
3512
+ "step": 45000
3513
+ },
3514
+ {
3515
+ "epoch": 0.08181818181818182,
3516
+ "eval_loss": 3.100367307662964,
3517
+ "eval_runtime": 7.4121,
3518
+ "eval_samples_per_second": 77.711,
3519
+ "eval_steps_per_second": 19.428,
3520
+ "step": 45000
3521
+ },
3522
+ {
3523
+ "epoch": 0.08272727272727273,
3524
+ "grad_norm": 0.1478699892759323,
3525
+ "learning_rate": 0.0006407240806880225,
3526
+ "loss": 2.717208251953125,
3527
+ "step": 45100
3528
+ },
3529
+ {
3530
+ "epoch": 0.08363636363636363,
3531
+ "grad_norm": 0.1527082324028015,
3532
+ "learning_rate": 0.0006393513640210743,
3533
+ "loss": 2.7066421508789062,
3534
+ "step": 45200
3535
+ },
3536
+ {
3537
+ "epoch": 0.08454545454545455,
3538
+ "grad_norm": 0.16474194824695587,
3539
+ "learning_rate": 0.0006379775076016327,
3540
+ "loss": 2.6989260864257814,
3541
+ "step": 45300
3542
+ },
3543
+ {
3544
+ "epoch": 0.08545454545454545,
3545
+ "grad_norm": 0.17212653160095215,
3546
+ "learning_rate": 0.0006366025226664465,
3547
+ "loss": 2.72157958984375,
3548
+ "step": 45400
3549
+ },
3550
+ {
3551
+ "epoch": 0.08636363636363636,
3552
+ "grad_norm": 0.1925841122865677,
3553
+ "learning_rate": 0.0006352264204614946,
3554
+ "loss": 2.7147427368164063,
3555
+ "step": 45500
3556
+ },
3557
+ {
3558
+ "epoch": 0.08727272727272728,
3559
+ "grad_norm": 0.19100086390972137,
3560
+ "learning_rate": 0.0006338492122418944,
3561
+ "loss": 2.6837368774414063,
3562
+ "step": 45600
3563
+ },
3564
+ {
3565
+ "epoch": 0.08818181818181818,
3566
+ "grad_norm": 0.15223105251789093,
3567
+ "learning_rate": 0.0006324709092718088,
3568
+ "loss": 2.6951806640625,
3569
+ "step": 45700
3570
+ },
3571
+ {
3572
+ "epoch": 0.0890909090909091,
3573
+ "grad_norm": 0.16720791161060333,
3574
+ "learning_rate": 0.000631091522824355,
3575
+ "loss": 2.6881533813476564,
3576
+ "step": 45800
3577
+ },
3578
+ {
3579
+ "epoch": 0.09,
3580
+ "grad_norm": 0.28359341621398926,
3581
+ "learning_rate": 0.0006297110641815118,
3582
+ "loss": 2.695024108886719,
3583
+ "step": 45900
3584
+ },
3585
+ {
3586
+ "epoch": 0.09090909090909091,
3587
+ "grad_norm": 0.16058634221553802,
3588
+ "learning_rate": 0.0006283295446340276,
3589
+ "loss": 2.694168701171875,
3590
+ "step": 46000
3591
+ },
3592
+ {
3593
+ "epoch": 0.09090909090909091,
3594
+ "eval_loss": 3.092599630355835,
3595
+ "eval_runtime": 7.4402,
3596
+ "eval_samples_per_second": 77.417,
3597
+ "eval_steps_per_second": 19.354,
3598
+ "step": 46000
3599
+ },
3600
+ {
3601
+ "epoch": 0.09181818181818181,
3602
+ "grad_norm": 0.14820711314678192,
3603
+ "learning_rate": 0.0006269469754813277,
3604
+ "loss": 2.678899841308594,
3605
+ "step": 46100
3606
+ },
3607
+ {
3608
+ "epoch": 0.09272727272727273,
3609
+ "grad_norm": 0.14676456153392792,
3610
+ "learning_rate": 0.0006255633680314226,
3611
+ "loss": 2.6948358154296876,
3612
+ "step": 46200
3613
+ },
3614
+ {
3615
+ "epoch": 0.09363636363636364,
3616
+ "grad_norm": 0.14033126831054688,
3617
+ "learning_rate": 0.0006241787336008143,
3618
+ "loss": 2.6942843627929687,
3619
+ "step": 46300
3620
+ },
3621
+ {
3622
+ "epoch": 0.09454545454545454,
3623
+ "grad_norm": 0.15463101863861084,
3624
+ "learning_rate": 0.000622793083514405,
3625
+ "loss": 2.6831298828125,
3626
+ "step": 46400
3627
+ },
3628
+ {
3629
+ "epoch": 0.09545454545454546,
3630
+ "grad_norm": 0.1568794697523117,
3631
+ "learning_rate": 0.0006214064291054039,
3632
+ "loss": 2.6844097900390627,
3633
+ "step": 46500
3634
+ },
3635
+ {
3636
+ "epoch": 0.09636363636363636,
3637
+ "grad_norm": 0.15985362231731415,
3638
+ "learning_rate": 0.0006200187817152341,
3639
+ "loss": 2.6705093383789062,
3640
+ "step": 46600
3641
+ },
3642
+ {
3643
+ "epoch": 0.09727272727272727,
3644
+ "grad_norm": 0.18481206893920898,
3645
+ "learning_rate": 0.0006186301526934407,
3646
+ "loss": 2.7399118041992185,
3647
+ "step": 46700
3648
+ },
3649
+ {
3650
+ "epoch": 0.09818181818181818,
3651
+ "grad_norm": 0.17036649584770203,
3652
+ "learning_rate": 0.0006172405533975973,
3653
+ "loss": 2.7174920654296875,
3654
+ "step": 46800
3655
+ },
3656
+ {
3657
+ "epoch": 0.09909090909090909,
3658
+ "grad_norm": 0.21465139091014862,
3659
+ "learning_rate": 0.0006158499951932138,
3660
+ "loss": 2.7115213012695314,
3661
+ "step": 46900
3662
+ },
3663
+ {
3664
+ "epoch": 0.1,
3665
+ "grad_norm": 0.16568434238433838,
3666
+ "learning_rate": 0.0006144584894536423,
3667
+ "loss": 2.7298922729492188,
3668
+ "step": 47000
3669
+ },
3670
+ {
3671
+ "epoch": 0.1,
3672
+ "eval_loss": 3.0938024520874023,
3673
+ "eval_runtime": 7.4616,
3674
+ "eval_samples_per_second": 77.195,
3675
+ "eval_steps_per_second": 19.299,
3676
+ "step": 47000
3677
+ },
3678
+ {
3679
+ "epoch": 0.1009090909090909,
3680
+ "grad_norm": 0.14605851471424103,
3681
+ "learning_rate": 0.0006130660475599854,
3682
+ "loss": 2.7149703979492186,
3683
+ "step": 47100
3684
+ },
3685
+ {
3686
+ "epoch": 0.10181818181818182,
3687
+ "grad_norm": 0.21608054637908936,
3688
+ "learning_rate": 0.0006116726809010022,
3689
+ "loss": 2.7035641479492187,
3690
+ "step": 47200
3691
+ },
3692
+ {
3693
+ "epoch": 0.10272727272727272,
3694
+ "grad_norm": 0.1545097827911377,
3695
+ "learning_rate": 0.0006102784008730155,
3696
+ "loss": 2.6880255126953125,
3697
+ "step": 47300
3698
+ },
3699
+ {
3700
+ "epoch": 0.10363636363636364,
3701
+ "grad_norm": 0.14439067244529724,
3702
+ "learning_rate": 0.0006088832188798183,
3703
+ "loss": 2.6835076904296873,
3704
+ "step": 47400
3705
+ },
3706
+ {
3707
+ "epoch": 0.10454545454545454,
3708
+ "grad_norm": 0.17214855551719666,
3709
+ "learning_rate": 0.0006074871463325809,
3710
+ "loss": 2.6864669799804686,
3711
+ "step": 47500
3712
+ },
3713
+ {
3714
+ "epoch": 0.10545454545454545,
3715
+ "grad_norm": 0.15308986604213715,
3716
+ "learning_rate": 0.0006060901946497581,
3717
+ "loss": 2.6995504760742186,
3718
+ "step": 47600
3719
+ },
3720
+ {
3721
+ "epoch": 0.10636363636363637,
3722
+ "grad_norm": 0.21875260770320892,
3723
+ "learning_rate": 0.000604692375256994,
3724
+ "loss": 2.6868545532226564,
3725
+ "step": 47700
3726
+ },
3727
+ {
3728
+ "epoch": 0.10727272727272727,
3729
+ "grad_norm": 0.1660732924938202,
3730
+ "learning_rate": 0.0006032936995870303,
3731
+ "loss": 2.718809814453125,
3732
+ "step": 47800
3733
+ },
3734
+ {
3735
+ "epoch": 0.10818181818181818,
3736
+ "grad_norm": 0.1624031513929367,
3737
+ "learning_rate": 0.000601894179079612,
3738
+ "loss": 2.718572692871094,
3739
+ "step": 47900
3740
+ },
3741
+ {
3742
+ "epoch": 0.10909090909090909,
3743
+ "grad_norm": 0.17190197110176086,
3744
+ "learning_rate": 0.0006004938251813943,
3745
+ "loss": 2.7020748901367186,
3746
+ "step": 48000
3747
+ },
3748
+ {
3749
+ "epoch": 0.10909090909090909,
3750
+ "eval_loss": 3.089716911315918,
3751
+ "eval_runtime": 7.4415,
3752
+ "eval_samples_per_second": 77.404,
3753
+ "eval_steps_per_second": 19.351,
3754
+ "step": 48000
3755
  }
3756
  ],
3757
  "logging_steps": 100,
 
3771
  "attributes": {}
3772
  }
3773
  },
3774
+ "total_flos": 1.195899887812608e+18,
3775
  "train_batch_size": 22,
3776
  "trial_name": null,
3777
  "trial_params": null