CodeIsAbstract commited on
Commit
7008e7a
·
verified ·
1 Parent(s): b3d0f08

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4c06f8bdb3470386515b85743f5f433fba520ab14f1b589e6f665e843ff544c6
3
  size 1600779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebb0d1344d85e1fc40cd8989c7c6db8bce3a550c89c85b825448ecbec3b9d6c1
3
  size 1600779
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:47079a3c6b76024ae75741e62abdf81d9e0ad0f98cded6e05c983e0a05682bad
3
  size 621149155
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2abc969de4ce645f9d76eb977581465612165dd82ab373f63378a5f647630661
3
  size 621149155
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d2a3ee2b9270eba0a2311f907ea53e1bec546e16e72a9c2fda6872be77c59c2f
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a2f17c67fa754158f3c03c5f9856c1aa1a0efb60e8b2ba7ff0933cd7e615deac
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1b4ff89f8eafc39f075c9216838e6c051cd47ba4912a50dce51d8507d2b54554
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:418ff55ed0dbe835354a2c0bf4053deea4eb6d628f7fb25addc88d7dd433e1a0
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:949e92b373ef785d8153838a856e7e94fcdbceccf0df94fac6b1428fba0aa52c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7a35f8395aebcabd2d2ff704dcd0a2516f18acacf15eaa6f6de4d80c3f5a64ac
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:06a6a2db4cb588d8ce4130d4e05d07a9c55b5589d5cb162b1a3d954475ff88d3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:20538af5e8a8c67556fafaef54a8d5e99ae679aaa736d860c04c1a2e37c3c03c
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7088f74e466de020dfa05daf938b6e89670196138b802849a37053644f6497b8
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7766e34119d4055998fe3273163a12ee2907d627abe9c4714c8b112cc18d1fb5
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0e94d95b4f9bc5f7112993c3b1fffc2e8c2ce3186f5ea05083591485e3d86055
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b86ea660b800de8c8f56b525f1c8ec70de64ab61f7f595cc6d7a9eec27f020ed
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:abab3cbb093f08eba5cc70d5ebf9d591bdbafc0cbf9d3444496ddf74c29c9cf8
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:afea3cb817289360580454c91b5d591d50af9204d37e43c3818d916f106c73a9
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5972654a558bb8bded698917db31ad4f8203a51ad38c3abf88508a55e8574b82
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f608549c3734c6d52376589878069a475672b0ca9e5e32be42d15b90c85b8d1
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eb2a89a85e1c8ac7ee9b5678d857a4d99090faa37a24489f44839e0801dfe16a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9567d76a2efaa946d19913fd9fe6f6f70f6f7101d7dcb8866c09743534f2d4d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 37.972,
483
  "eval_steps_per_second": 0.608,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 37.972,
483
  "eval_steps_per_second": 0.608,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.122,
488
+ "grad_norm": 1.9359498023986816,
489
+ "learning_rate": 0.0019718035350990712,
490
+ "loss": 35.0183,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.124,
495
+ "grad_norm": 1.7273503541946411,
496
+ "learning_rate": 0.0019702227914230566,
497
+ "loss": 34.6009,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.126,
502
+ "grad_norm": 4.945074081420898,
503
+ "learning_rate": 0.001968599607059059,
504
+ "loss": 34.7242,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.128,
509
+ "grad_norm": 1.6802929639816284,
510
+ "learning_rate": 0.0019669340530104207,
511
+ "loss": 34.5441,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.13,
516
+ "grad_norm": 1.8048967123031616,
517
+ "learning_rate": 0.001965226202133872,
518
+ "loss": 34.3155,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.132,
523
+ "grad_norm": 1.656825065612793,
524
+ "learning_rate": 0.0019634761291363427,
525
+ "loss": 33.7591,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.134,
530
+ "grad_norm": 1.642227053642273,
531
+ "learning_rate": 0.0019616839105716954,
532
+ "loss": 33.6263,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.136,
537
+ "grad_norm": 1.531063199043274,
538
+ "learning_rate": 0.0019598496248373755,
539
+ "loss": 34.4644,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.138,
544
+ "grad_norm": 2.118232011795044,
545
+ "learning_rate": 0.001957973352170984,
546
+ "loss": 34.597,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.14,
551
+ "grad_norm": 2.264650344848633,
552
+ "learning_rate": 0.001956055174646765,
553
+ "loss": 34.2066,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.14,
558
+ "eval_accuracy": 0.27988845401174167,
559
+ "eval_loss": 34.06083679199219,
560
+ "eval_runtime": 3.2263,
561
+ "eval_samples_per_second": 38.744,
562
+ "eval_steps_per_second": 0.62,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.142,
567
+ "grad_norm": 1.5203992128372192,
568
+ "learning_rate": 0.0019540951761720174,
569
+ "loss": 34.176,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.144,
574
+ "grad_norm": 1.5354284048080444,
575
+ "learning_rate": 0.0019520934424834247,
576
+ "loss": 33.8863,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.146,
581
+ "grad_norm": 1.9357357025146484,
582
+ "learning_rate": 0.0019500500611433025,
583
+ "loss": 33.6482,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.148,
588
+ "grad_norm": 1.6156283617019653,
589
+ "learning_rate": 0.0019479651215357707,
590
+ "loss": 33.4445,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.15,
595
+ "grad_norm": 1.962512731552124,
596
+ "learning_rate": 0.0019458387148628417,
597
+ "loss": 33.8778,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.152,
602
+ "grad_norm": 1.920979380607605,
603
+ "learning_rate": 0.001943670934140432,
604
+ "loss": 34.0846,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.154,
609
+ "grad_norm": 1.5394798517227173,
610
+ "learning_rate": 0.0019414618741942936,
611
+ "loss": 33.9074,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.156,
616
+ "grad_norm": 1.8427844047546387,
617
+ "learning_rate": 0.0019392116316558638,
618
+ "loss": 33.8625,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.158,
623
+ "grad_norm": 1.5528762340545654,
624
+ "learning_rate": 0.001936920304958042,
625
+ "loss": 33.5134,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "grad_norm": 2.177480936050415,
631
+ "learning_rate": 0.0019345879943308804,
632
+ "loss": 33.2939,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.16,
637
+ "eval_accuracy": 0.28487866927592953,
638
+ "eval_loss": 33.50783920288086,
639
+ "eval_runtime": 3.2372,
640
+ "eval_samples_per_second": 38.613,
641
+ "eval_steps_per_second": 0.618,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,