CodeIsAbstract commited on
Commit
16b4153
·
verified ·
1 Parent(s): 5a4967f

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:82a943209b657b15ff00004efcaa88da19c9f1fd4603f11587c023fc47eda605
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b5b2537f5142e6fa774f5229435c8fe08d7962419f4ec779873d8e1778b6a5a
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:13ae1bff671c06244b8db21ff756e241dc09da71f9f9dc4f2ec99c071e13b99f
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fde75f324a8d5912df33d74af3539f7d5c0a866216e75d10d52848057c5278fd
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15bd918144b9671ae295d46e463180bdad2e73e51a7670636feb8fc58e14c426
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:231a1d74fdd644709c48076ea6500b1a4f8ca1620a6f210d62a4c539f1ee0b0f
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2666cc7db576490d8041b6acfda665e20af01ea5f7120fdf1d1c04b59674d0ca
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0182b4495f3c647cd51b29236a0c27e55ac1c8000e2e254b23fc4d917b1a56da
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 306.4,
483
  "eval_steps_per_second": 19.259,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
@@ -501,7 +659,7 @@
501
  "attributes": {}
502
  }
503
  },
504
- "total_flos": 2.3519765200896e+17,
505
  "train_batch_size": 120,
506
  "trial_name": null,
507
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 306.4,
483
  "eval_steps_per_second": 19.259,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.122,
488
+ "grad_norm": 0.18666808307170868,
489
+ "learning_rate": 0.000965270029838182,
490
+ "loss": 3.6821221923828125,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.124,
495
+ "grad_norm": 0.15751489996910095,
496
+ "learning_rate": 0.0009641069162834215,
497
+ "loss": 3.671753234863281,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.126,
502
+ "grad_norm": 0.22766178846359253,
503
+ "learning_rate": 0.0009629253701530875,
504
+ "loss": 3.6878753662109376,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.128,
509
+ "grad_norm": 0.1677054762840271,
510
+ "learning_rate": 0.0009617254383737343,
511
+ "loss": 3.6619857788085937,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.13,
516
+ "grad_norm": 0.17945168912410736,
517
+ "learning_rate": 0.0009605071686021245,
518
+ "loss": 3.6484750366210936,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.132,
523
+ "grad_norm": 0.17799174785614014,
524
+ "learning_rate": 0.0009592706092233365,
525
+ "loss": 3.6635867309570314,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.134,
530
+ "grad_norm": 0.5586189031600952,
531
+ "learning_rate": 0.0009580158093488436,
532
+ "loss": 3.6489688110351564,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.136,
537
+ "grad_norm": 0.2055712342262268,
538
+ "learning_rate": 0.000956742818814562,
539
+ "loss": 3.6523666381835938,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.138,
544
+ "grad_norm": 0.1812128871679306,
545
+ "learning_rate": 0.0009554516881788724,
546
+ "loss": 3.6479287719726563,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.14,
551
+ "grad_norm": 0.1664743423461914,
552
+ "learning_rate": 0.0009541424687206124,
553
+ "loss": 3.6319317626953125,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.14,
558
+ "eval_accuracy": 0.3448148965923309,
559
+ "eval_loss": 3.6666343212127686,
560
+ "eval_runtime": 6.38,
561
+ "eval_samples_per_second": 304.231,
562
+ "eval_steps_per_second": 19.122,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.142,
567
+ "grad_norm": 0.15217284858226776,
568
+ "learning_rate": 0.0009528152124370387,
569
+ "loss": 3.6308517456054688,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.144,
574
+ "grad_norm": 0.16507816314697266,
575
+ "learning_rate": 0.0009514699720417631,
576
+ "loss": 3.6217156982421876,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.146,
581
+ "grad_norm": 0.2040642648935318,
582
+ "learning_rate": 0.0009501068009626583,
583
+ "loss": 3.632703857421875,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.148,
588
+ "grad_norm": 0.18437133729457855,
589
+ "learning_rate": 0.0009487257533397361,
590
+ "loss": 3.6346307373046876,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.15,
595
+ "grad_norm": 0.18760915100574493,
596
+ "learning_rate": 0.0009473268840229971,
597
+ "loss": 3.5973056030273436,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.152,
602
+ "grad_norm": 0.17388536036014557,
603
+ "learning_rate": 0.0009459102485702528,
604
+ "loss": 3.611794738769531,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.154,
609
+ "grad_norm": 0.18797264993190765,
610
+ "learning_rate": 0.0009444759032449177,
611
+ "loss": 3.609403381347656,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.156,
616
+ "grad_norm": 0.18140241503715515,
617
+ "learning_rate": 0.0009430239050137766,
618
+ "loss": 3.6027340698242187,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.158,
623
+ "grad_norm": 0.17433640360832214,
624
+ "learning_rate": 0.0009415543115447203,
625
+ "loss": 3.6214892578125,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "grad_norm": 0.16605013608932495,
631
+ "learning_rate": 0.0009400671812044565,
632
+ "loss": 3.6061587524414063,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.16,
637
+ "eval_accuracy": 0.34762277801806923,
638
+ "eval_loss": 3.640489339828491,
639
+ "eval_runtime": 8.0453,
640
+ "eval_samples_per_second": 241.259,
641
+ "eval_steps_per_second": 15.164,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,
 
659
  "attributes": {}
660
  }
661
  },
662
+ "total_flos": 3.1359686934528e+17,
663
  "train_batch_size": 120,
664
  "trial_name": null,
665
  "trial_params": null