CodeIsAbstract commited on
Commit
b6f435d
·
verified ·
1 Parent(s): 7620a46

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9a3b1570476040558106ef329ac5c35eaf7b927d872082a820226ad0028b5dd9
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a481e433fa6eac80981a2ff00a0f8f7cf3e3e08ee088dd15e950410638af8d4
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ac1c439f2b333c6ba22b924fd6eb4e28a8524e3ac41694760d6b949858f9f4ad
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:80c48bc7e746e70fff3e665a65cbf1675701d35dc75d28205041543695987838
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:595e0adb3f51b4014c388d14cb8a20e5767cd34a947ffb570078cd186928d2c7
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38818bd9f30dd9f6202fc8de57f60ee4722cb7b211723c490ff280c9aeccae51
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f43de46bf0618be59b12a78ea440d0013f604015e857ea4a03ea845dabb299ad
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1d444c1a49dd1792c344220625a92a47a0fd68dd629b20f674d9dfda9433cfe5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 340.705,
483
  "eval_steps_per_second": 21.415,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
@@ -501,7 +659,7 @@
501
  "attributes": {}
502
  }
503
  },
504
- "total_flos": 1.8813026304e+17,
505
  "train_batch_size": 120,
506
  "trial_name": null,
507
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 340.705,
483
  "eval_steps_per_second": 21.415,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.122,
488
+ "grad_norm": 0.25070950388908386,
489
+ "learning_rate": 0.0005791620179029091,
490
+ "loss": 3.7485830688476565,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.124,
495
+ "grad_norm": 0.23322179913520813,
496
+ "learning_rate": 0.0005784641497700528,
497
+ "loss": 3.7345367431640626,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.126,
502
+ "grad_norm": 0.27075648307800293,
503
+ "learning_rate": 0.0005777552220918525,
504
+ "loss": 3.747908935546875,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.128,
509
+ "grad_norm": 0.22065384685993195,
510
+ "learning_rate": 0.0005770352630242405,
511
+ "loss": 3.7215350341796873,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.13,
516
+ "grad_norm": 0.2387268990278244,
517
+ "learning_rate": 0.0005763043011612747,
518
+ "loss": 3.7065789794921873,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.132,
523
+ "grad_norm": 0.2440839409828186,
524
+ "learning_rate": 0.0005755623655340019,
525
+ "loss": 3.7222140502929686,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.134,
530
+ "grad_norm": 0.7614469528198242,
531
+ "learning_rate": 0.000574809485609306,
532
+ "loss": 3.7069766235351564,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.136,
537
+ "grad_norm": 0.25670167803764343,
538
+ "learning_rate": 0.0005740456912887371,
539
+ "loss": 3.7104458618164062,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.138,
544
+ "grad_norm": 0.25214189291000366,
545
+ "learning_rate": 0.0005732710129073234,
546
+ "loss": 3.7040945434570314,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.14,
551
+ "grad_norm": 0.2262343317270279,
552
+ "learning_rate": 0.0005724854812323674,
553
+ "loss": 3.686309814453125,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.14,
558
+ "eval_accuracy": 0.3409998074307532,
559
+ "eval_loss": 3.705655097961426,
560
+ "eval_runtime": 6.3038,
561
+ "eval_samples_per_second": 307.912,
562
+ "eval_steps_per_second": 19.354,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.142,
567
+ "grad_norm": 0.2256837785243988,
568
+ "learning_rate": 0.0005716891274622232,
569
+ "loss": 3.6851046752929686,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.144,
574
+ "grad_norm": 0.22516745328903198,
575
+ "learning_rate": 0.0005708819832250578,
576
+ "loss": 3.6776641845703124,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.146,
581
+ "grad_norm": 0.28322234749794006,
582
+ "learning_rate": 0.000570064080577595,
583
+ "loss": 3.6858819580078124,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.148,
588
+ "grad_norm": 0.2589821219444275,
589
+ "learning_rate": 0.0005692354520038415,
590
+ "loss": 3.683369140625,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.15,
595
+ "grad_norm": 0.25471949577331543,
596
+ "learning_rate": 0.0005683961304137982,
597
+ "loss": 3.6476519775390623,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.152,
602
+ "grad_norm": 0.24686427414417267,
603
+ "learning_rate": 0.0005675461491421516,
604
+ "loss": 3.66244873046875,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.154,
609
+ "grad_norm": 0.23603662848472595,
610
+ "learning_rate": 0.0005666855419469506,
611
+ "loss": 3.6586270141601562,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.156,
616
+ "grad_norm": 0.23954837024211884,
617
+ "learning_rate": 0.0005658143430082659,
618
+ "loss": 3.6526480102539063,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.158,
623
+ "grad_norm": 0.24939873814582825,
624
+ "learning_rate": 0.0005649325869268321,
625
+ "loss": 3.6602520751953125,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "grad_norm": 0.22606751322746277,
631
+ "learning_rate": 0.0005640403087226739,
632
+ "loss": 3.6500091552734375,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.16,
637
+ "eval_accuracy": 0.34401134847875336,
638
+ "eval_loss": 3.6704835891723633,
639
+ "eval_runtime": 5.9263,
640
+ "eval_samples_per_second": 327.521,
641
+ "eval_steps_per_second": 20.586,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,
 
659
  "attributes": {}
660
  }
661
  },
662
+ "total_flos": 2.5084035072e+17,
663
  "train_batch_size": 120,
664
  "trial_name": null,
665
  "trial_params": null