CodeIsAbstract commited on
Commit
092e33a
·
verified ·
1 Parent(s): cbf15a6

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f5c48459fa60b605303b0d27ace00ad2096e8131793c4adcbd8dbba3f7f05b68
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc0c1563393d995412311cba1214a9520dca547b90b519de7f1ce0b2f94384a6
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:495659b2393af3017ab9e8e8f60326b39ab2d3fd0faffcb10f202cad46eb1273
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3aa9b056f2e17085d4900f3271d88de35f6ac5194621790395f2e4f587a69232
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4ea3f206a9236577723633ea7c144cddc74c7390e88e7f9099912f6491fd80e8
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1e47a193d21ecc1ad99cf3d0b11067079c7e2eeed011077dd9d2eec110457c9c
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:867b8afafb899ee4f8d7e6052453cd8a741e4214d85da108553245a0b56d701a
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:768ce36046f98e2d652c3d4e4715ee92969f1d46b33288d5213930731a801cf9
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:faa6fa2287aad321cbb22bbaedd456c7be2ae098ba9ca7cd858cc4c928c56651
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad69461fda5a7151f961131a57f67c28a9bf598350ae5d99ce95c8ec88ff78da
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a6cb874a5313a6c9a2dbde4359bb963ad3cfd50cd3ad51a1fd10c71dd3508a77
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f092d919dcd7a979986c37fe2f4d5718eda679a575e6e71176605e3f28b4da3
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ea3314ca642931da1a5b28e60a20bd875be813361a002674a2f7f8ab48c9856e
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7cfd234ef8e0bce0d28c2272f50bbffc1faa98bd25482e7a5eb6e1a7d18e6aab
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c450d759f6f13b4b263f398164f48782614a20b556bf9dd81d5722276a593f27
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9277cfd5b2385bc7162ccde048b9e10c1f018201304953d47b48ffdd2baae832
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a758a18329b8c1b5690e207a82e38eddc070bca1b9e42028d9bae1a9e005081
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:110f59eb6a8f1b021131a112cdfbbed170d6a8d222fbcb0a48b72e3683fe424f
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:69b890497b4e2dc118225b6da994942d04ab88dde8ff041fe55844b623ed290f
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:28db19b12e448a6216bc82ad0a8dc000f968411ed4a0fc53a441e099e33a6ca8
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:888dd1fbf7803a02c8a29182dd5e65aa537b787d31363ffaf998bf1b2b7561ff
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4dc3ee7ff07df340dc973522b19f5110e955054f7179c9fdb7ad1c0dee99bc27
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.24,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 16.662,
483
  "eval_steps_per_second": 0.533,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.32,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 16.662,
483
  "eval_steps_per_second": 0.533,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.244,
488
+ "grad_norm": 3.177504062652588,
489
+ "learning_rate": 0.0034421101123303897,
490
+ "loss": 46.474375,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.248,
495
+ "grad_norm": 3.0214405059814453,
496
+ "learning_rate": 0.0034245757310132244,
497
+ "loss": 46.3790576171875,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.252,
502
+ "grad_norm": 2.933534860610962,
503
+ "learning_rate": 0.003406816212602642,
504
+ "loss": 46.0579736328125,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.256,
509
+ "grad_norm": 2.5542778968811035,
510
+ "learning_rate": 0.003388834363777341,
511
+ "loss": 46.0082861328125,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.26,
516
+ "grad_norm": 3.1228818893432617,
517
+ "learning_rate": 0.0033706330263526692,
518
+ "loss": 46.218310546875,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.264,
523
+ "grad_norm": 3.096647262573242,
524
+ "learning_rate": 0.003352215076831515,
525
+ "loss": 46.06939453125,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.268,
530
+ "grad_norm": 3.2484421730041504,
531
+ "learning_rate": 0.0033335834259497076,
532
+ "loss": 46.158388671875,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.272,
537
+ "grad_norm": 3.035205841064453,
538
+ "learning_rate": 0.003314741018216011,
539
+ "loss": 45.9556396484375,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.276,
544
+ "grad_norm": 3.4140381813049316,
545
+ "learning_rate": 0.0032956908314467803,
546
+ "loss": 45.8059375,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.28,
551
+ "grad_norm": 3.5898549556732178,
552
+ "learning_rate": 0.003276435876295352,
553
+ "loss": 45.785087890625,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.28,
558
+ "eval_accuracy": 0.2004613880742913,
559
+ "eval_loss": 45.5724983215332,
560
+ "eval_runtime": 7.1637,
561
+ "eval_samples_per_second": 17.449,
562
+ "eval_steps_per_second": 0.558,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.284,
567
+ "grad_norm": 3.6645097732543945,
568
+ "learning_rate": 0.0032569791957762473,
569
+ "loss": 45.450126953125,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.288,
574
+ "grad_norm": 3.449202060699463,
575
+ "learning_rate": 0.003237323864784261,
576
+ "loss": 45.6815869140625,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.292,
581
+ "grad_norm": 3.7189524173736572,
582
+ "learning_rate": 0.0032174729896085105,
583
+ "loss": 45.68208984375,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.296,
588
+ "grad_norm": 3.6789443492889404,
589
+ "learning_rate": 0.0031974297074415237,
590
+ "loss": 45.42337890625,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.3,
595
+ "grad_norm": 4.174764633178711,
596
+ "learning_rate": 0.003177197185883444,
597
+ "loss": 45.540263671875,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.304,
602
+ "grad_norm": 3.5856988430023193,
603
+ "learning_rate": 0.0031567786224414272,
604
+ "loss": 45.4182177734375,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.308,
609
+ "grad_norm": 3.8331663608551025,
610
+ "learning_rate": 0.003136177244024319,
611
+ "loss": 45.5009228515625,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.312,
616
+ "grad_norm": 3.373854160308838,
617
+ "learning_rate": 0.003115396306432675,
618
+ "loss": 45.4340283203125,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.316,
623
+ "grad_norm": 3.256587028503418,
624
+ "learning_rate": 0.003094439093844223,
625
+ "loss": 45.136494140625,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.32,
630
+ "grad_norm": 3.5598511695861816,
631
+ "learning_rate": 0.0030733089182948376,
632
+ "loss": 45.0095849609375,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.32,
637
+ "eval_accuracy": 0.2007086999022483,
638
+ "eval_loss": 45.0313720703125,
639
+ "eval_runtime": 6.9806,
640
+ "eval_samples_per_second": 17.907,
641
+ "eval_steps_per_second": 0.573,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,