CodeIsAbstract commited on
Commit
84a1225
·
verified ·
1 Parent(s): 952493f

Training in progress, step 900, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fec3e662918181e81fec2617a5f9f5de6a01b3e7022f265f73654bbc50438347
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ce638fe074432886d0a23f98a8f165e69d1e6fee84a526cf594d00f8534028a1
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:38ffba74dde1d2bade94cf7d477b69f097fefa5eddd6cbfa453ed04630627cc1
3
  size 1386414411
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0caa91a435695f3ccebbf12966391316e72dc5ad6f9cc2e90bacdd64a48fdc1c
3
  size 1386414411
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:af6ae550ab924258b75cba7b10f460b254db57e2f0ee1e6652ad45c4ae51f55f
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2be637d2dd60e3cec1955d0241821cbcccd2af3688716d84100a306b44dad4a4
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e216a9e826d4c344530b7e6d00074e0b36744d381337d9fe3838255770b48b79
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8d05de60c66749991154136b82121e270010e0cf03e7ae880dd38dcbbe6fd4a
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:303505cecb97596e2f53b5d89a1eed07f741f9537510242aaeccaa44a51fbf19
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f617c1b6753801a9d80064eb68997fd29d2b80564fcc110cff0fa781e09b8683
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2857142857142857,
6
  "eval_steps": 150,
7
- "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -464,6 +464,234 @@
464
  "eval_samples_per_second": 19.689,
465
  "eval_steps_per_second": 2.481,
466
  "step": 600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
467
  }
468
  ],
469
  "logging_steps": 10,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.42857142857142855,
6
  "eval_steps": 150,
7
+ "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
464
  "eval_samples_per_second": 19.689,
465
  "eval_steps_per_second": 2.481,
466
  "step": 600
467
+ },
468
+ {
469
+ "epoch": 0.2904761904761905,
470
+ "grad_norm": 20.064794540405273,
471
+ "learning_rate": 3.4024735685449766e-06,
472
+ "loss": 17.788371276855468,
473
+ "step": 610
474
+ },
475
+ {
476
+ "epoch": 0.29523809523809524,
477
+ "grad_norm": 19.219192504882812,
478
+ "learning_rate": 3.379847167149008e-06,
479
+ "loss": 17.86636962890625,
480
+ "step": 620
481
+ },
482
+ {
483
+ "epoch": 0.3,
484
+ "grad_norm": 23.365055084228516,
485
+ "learning_rate": 3.3568786004588363e-06,
486
+ "loss": 17.805162048339845,
487
+ "step": 630
488
+ },
489
+ {
490
+ "epoch": 0.3047619047619048,
491
+ "grad_norm": 16.860742568969727,
492
+ "learning_rate": 3.333573564066385e-06,
493
+ "loss": 17.576748657226563,
494
+ "step": 640
495
+ },
496
+ {
497
+ "epoch": 0.30952380952380953,
498
+ "grad_norm": 23.96249008178711,
499
+ "learning_rate": 3.3099378369990875e-06,
500
+ "loss": 17.694325256347657,
501
+ "step": 650
502
+ },
503
+ {
504
+ "epoch": 0.3142857142857143,
505
+ "grad_norm": 16.125730514526367,
506
+ "learning_rate": 3.285977280286845e-06,
507
+ "loss": 17.539736938476562,
508
+ "step": 660
509
+ },
510
+ {
511
+ "epoch": 0.319047619047619,
512
+ "grad_norm": 16.924381256103516,
513
+ "learning_rate": 3.261697835508648e-06,
514
+ "loss": 17.864358520507814,
515
+ "step": 670
516
+ },
517
+ {
518
+ "epoch": 0.3238095238095238,
519
+ "grad_norm": 24.52809715270996,
520
+ "learning_rate": 3.2371055233192185e-06,
521
+ "loss": 17.447659301757813,
522
+ "step": 680
523
+ },
524
+ {
525
+ "epoch": 0.32857142857142857,
526
+ "grad_norm": 24.08842658996582,
527
+ "learning_rate": 3.2122064419560557e-06,
528
+ "loss": 17.473863220214845,
529
+ "step": 690
530
+ },
531
+ {
532
+ "epoch": 0.3333333333333333,
533
+ "grad_norm": 26.00309944152832,
534
+ "learning_rate": 3.1870067657272295e-06,
535
+ "loss": 17.199142456054688,
536
+ "step": 700
537
+ },
538
+ {
539
+ "epoch": 0.3380952380952381,
540
+ "grad_norm": 29.389789581298828,
541
+ "learning_rate": 3.1615127434803192e-06,
542
+ "loss": 17.315379333496093,
543
+ "step": 710
544
+ },
545
+ {
546
+ "epoch": 0.34285714285714286,
547
+ "grad_norm": 30.414424896240234,
548
+ "learning_rate": 3.1357306970528665e-06,
549
+ "loss": 17.431112670898436,
550
+ "step": 720
551
+ },
552
+ {
553
+ "epoch": 0.3476190476190476,
554
+ "grad_norm": 32.14336395263672,
555
+ "learning_rate": 3.1096670197047267e-06,
556
+ "loss": 17.360943603515626,
557
+ "step": 730
558
+ },
559
+ {
560
+ "epoch": 0.3523809523809524,
561
+ "grad_norm": 14.162923812866211,
562
+ "learning_rate": 3.0833281745327123e-06,
563
+ "loss": 17.168132019042968,
564
+ "step": 740
565
+ },
566
+ {
567
+ "epoch": 0.35714285714285715,
568
+ "grad_norm": 21.957395553588867,
569
+ "learning_rate": 3.056720692867918e-06,
570
+ "loss": 17.510189819335938,
571
+ "step": 750
572
+ },
573
+ {
574
+ "epoch": 0.35714285714285715,
575
+ "eval_accuracy": 0.057960629921259846,
576
+ "eval_loss": 17.25590705871582,
577
+ "eval_runtime": 25.9338,
578
+ "eval_samples_per_second": 19.28,
579
+ "eval_steps_per_second": 2.429,
580
+ "step": 750
581
+ },
582
+ {
583
+ "epoch": 0.3619047619047619,
584
+ "grad_norm": 22.730085372924805,
585
+ "learning_rate": 3.029851172656121e-06,
586
+ "loss": 17.168865966796876,
587
+ "step": 760
588
+ },
589
+ {
590
+ "epoch": 0.36666666666666664,
591
+ "grad_norm": 9.091358184814453,
592
+ "learning_rate": 3.0027262768216716e-06,
593
+ "loss": 17.32322540283203,
594
+ "step": 770
595
+ },
596
+ {
597
+ "epoch": 0.37142857142857144,
598
+ "grad_norm": 46.21049118041992,
599
+ "learning_rate": 2.9753527316152624e-06,
600
+ "loss": 17.2936767578125,
601
+ "step": 780
602
+ },
603
+ {
604
+ "epoch": 0.3761904761904762,
605
+ "grad_norm": 12.04414176940918,
606
+ "learning_rate": 2.9477373249459973e-06,
607
+ "loss": 17.235751342773437,
608
+ "step": 790
609
+ },
610
+ {
611
+ "epoch": 0.38095238095238093,
612
+ "grad_norm": 43.7669563293457,
613
+ "learning_rate": 2.919886904698173e-06,
614
+ "loss": 17.19912109375,
615
+ "step": 800
616
+ },
617
+ {
618
+ "epoch": 0.38571428571428573,
619
+ "grad_norm": 15.64034652709961,
620
+ "learning_rate": 2.8918083770331857e-06,
621
+ "loss": 17.119082641601562,
622
+ "step": 810
623
+ },
624
+ {
625
+ "epoch": 0.3904761904761905,
626
+ "grad_norm": 15.29228401184082,
627
+ "learning_rate": 2.8635087046769856e-06,
628
+ "loss": 17.21935119628906,
629
+ "step": 820
630
+ },
631
+ {
632
+ "epoch": 0.3952380952380952,
633
+ "grad_norm": 40.923484802246094,
634
+ "learning_rate": 2.83499490519351e-06,
635
+ "loss": 17.068731689453124,
636
+ "step": 830
637
+ },
638
+ {
639
+ "epoch": 0.4,
640
+ "grad_norm": 18.847187042236328,
641
+ "learning_rate": 2.8062740492445104e-06,
642
+ "loss": 17.178439331054687,
643
+ "step": 840
644
+ },
645
+ {
646
+ "epoch": 0.40476190476190477,
647
+ "grad_norm": 10.75904369354248,
648
+ "learning_rate": 2.7773532588362205e-06,
649
+ "loss": 17.27076873779297,
650
+ "step": 850
651
+ },
652
+ {
653
+ "epoch": 0.4095238095238095,
654
+ "grad_norm": 22.709970474243164,
655
+ "learning_rate": 2.748239705553287e-06,
656
+ "loss": 17.397381591796876,
657
+ "step": 860
658
+ },
659
+ {
660
+ "epoch": 0.4142857142857143,
661
+ "grad_norm": 22.443639755249023,
662
+ "learning_rate": 2.718940608780408e-06,
663
+ "loss": 17.009933471679688,
664
+ "step": 870
665
+ },
666
+ {
667
+ "epoch": 0.41904761904761906,
668
+ "grad_norm": 9.636713027954102,
669
+ "learning_rate": 2.6894632339121183e-06,
670
+ "loss": 17.1143310546875,
671
+ "step": 880
672
+ },
673
+ {
674
+ "epoch": 0.4238095238095238,
675
+ "grad_norm": 11.698447227478027,
676
+ "learning_rate": 2.6598148905511656e-06,
677
+ "loss": 17.266459655761718,
678
+ "step": 890
679
+ },
680
+ {
681
+ "epoch": 0.42857142857142855,
682
+ "grad_norm": 34.808990478515625,
683
+ "learning_rate": 2.6300029306959236e-06,
684
+ "loss": 17.431172180175782,
685
+ "step": 900
686
+ },
687
+ {
688
+ "epoch": 0.42857142857142855,
689
+ "eval_accuracy": 0.059070866141732285,
690
+ "eval_loss": 16.987903594970703,
691
+ "eval_runtime": 25.5466,
692
+ "eval_samples_per_second": 19.572,
693
+ "eval_steps_per_second": 2.466,
694
+ "step": 900
695
  }
696
  ],
697
  "logging_steps": 10,