anhdai312 commited on
Commit
a3ed760
·
verified ·
1 Parent(s): bc08165

Training in progress, step 90, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:380f08ab32010d6fe11a53ab351050291953fedb116a7277e2843bd4f6582f55
3
  size 83945296
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:989083296434928f43ed9a70e5209904853c84f2fe79dd46eaaf75a623c1c612
3
  size 83945296
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6a32edd2489574518e0200e545f1d6666a3dfda3c078dd03ba171eaca2f004a5
3
  size 43127525
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb29b665aa2a08cbcaf98b95af1daca25364564535eb5b22d2904b2f0cd20b8d
3
  size 43127525
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cd3c5be0654e1f24601c2bc652b3ec070468a284c4b667eeda9c51ced25c95e6
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14771e8b5dca1bdad12c90b67c90f4f7fb7649a9f36b36bc35e42fde4ce0f3eb
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6479e3863717204c87adedee1441b5a8610be4e4328e7c73664485cf28fb3e20
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5725349df08941ce3ad094aef227575807c35b00023b923686388db7b0fecb5b
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a047a4f72eccdbb2b5fa25ef9898708ca9cc71f79b88f4e3ecd8e10e64db0481
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:46d0e3fbb61b4936b37d5244c65578d229675919411e13774b1f2e5dd30f5838
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": 50,
3
  "best_metric": 1.952015995979309,
4
  "best_model_checkpoint": "/tmp/checkpoints_/job_770d3d41-3bcd-47ca-8265-a07b315f88da/checkpoint-50",
5
- "epoch": 0.8,
6
  "eval_steps": 10,
7
- "global_step": 80,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -632,6 +632,84 @@
632
  "eval_samples_per_second": 3.08,
633
  "eval_steps_per_second": 0.801,
634
  "step": 80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
635
  }
636
  ],
637
  "logging_steps": 1,
@@ -651,7 +729,7 @@
651
  "attributes": {}
652
  }
653
  },
654
- "total_flos": 4625532775219200.0,
655
  "train_batch_size": 1,
656
  "trial_name": null,
657
  "trial_params": null
 
2
  "best_global_step": 50,
3
  "best_metric": 1.952015995979309,
4
  "best_model_checkpoint": "/tmp/checkpoints_/job_770d3d41-3bcd-47ca-8265-a07b315f88da/checkpoint-50",
5
+ "epoch": 0.9,
6
  "eval_steps": 10,
7
+ "global_step": 90,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
632
  "eval_samples_per_second": 3.08,
633
  "eval_steps_per_second": 0.801,
634
  "step": 80
635
+ },
636
+ {
637
+ "epoch": 0.81,
638
+ "grad_norm": 3.591055154800415,
639
+ "learning_rate": 1.1578947368421053e-05,
640
+ "loss": 0.2746422588825226,
641
+ "step": 81
642
+ },
643
+ {
644
+ "epoch": 0.82,
645
+ "grad_norm": 4.242494583129883,
646
+ "learning_rate": 1.1052631578947368e-05,
647
+ "loss": 0.2333741933107376,
648
+ "step": 82
649
+ },
650
+ {
651
+ "epoch": 0.83,
652
+ "grad_norm": 2.425588607788086,
653
+ "learning_rate": 1.0526315789473684e-05,
654
+ "loss": 0.10867268592119217,
655
+ "step": 83
656
+ },
657
+ {
658
+ "epoch": 0.84,
659
+ "grad_norm": 4.30889892578125,
660
+ "learning_rate": 1e-05,
661
+ "loss": 0.21931734681129456,
662
+ "step": 84
663
+ },
664
+ {
665
+ "epoch": 0.85,
666
+ "grad_norm": 4.7018842697143555,
667
+ "learning_rate": 9.473684210526317e-06,
668
+ "loss": 0.18337883055210114,
669
+ "step": 85
670
+ },
671
+ {
672
+ "epoch": 0.86,
673
+ "grad_norm": 3.754896640777588,
674
+ "learning_rate": 8.947368421052632e-06,
675
+ "loss": 0.11415284126996994,
676
+ "step": 86
677
+ },
678
+ {
679
+ "epoch": 0.87,
680
+ "grad_norm": 3.8266732692718506,
681
+ "learning_rate": 8.421052631578948e-06,
682
+ "loss": 0.13100245594978333,
683
+ "step": 87
684
+ },
685
+ {
686
+ "epoch": 0.88,
687
+ "grad_norm": 4.878761291503906,
688
+ "learning_rate": 7.894736842105263e-06,
689
+ "loss": 0.18353883922100067,
690
+ "step": 88
691
+ },
692
+ {
693
+ "epoch": 0.89,
694
+ "grad_norm": 3.0917410850524902,
695
+ "learning_rate": 7.3684210526315784e-06,
696
+ "loss": 0.13703593611717224,
697
+ "step": 89
698
+ },
699
+ {
700
+ "epoch": 0.9,
701
+ "grad_norm": 8.035305976867676,
702
+ "learning_rate": 6.842105263157896e-06,
703
+ "loss": 0.3001468777656555,
704
+ "step": 90
705
+ },
706
+ {
707
+ "epoch": 0.9,
708
+ "eval_loss": 2.0694642066955566,
709
+ "eval_runtime": 16.0628,
710
+ "eval_samples_per_second": 3.113,
711
+ "eval_steps_per_second": 0.809,
712
+ "step": 90
713
  }
714
  ],
715
  "logging_steps": 1,
 
729
  "attributes": {}
730
  }
731
  },
732
+ "total_flos": 5199896781864960.0,
733
  "train_batch_size": 1,
734
  "trial_name": null,
735
  "trial_params": null