CodeIsAbstract commited on
Commit
b103958
·
verified ·
1 Parent(s): eb3106b

Training in progress, step 36000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:730a84ab942781715590be7c04cad100c356ef27ced617f53c7a7afc6f79f76e
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e61997d322b558147c11cb3a841b6c879522f4e0401570393ee65009491b13e
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b801cdd780ceaa960f46aaa0c82910a2ee0f77af3430455271bf70ac88460e56
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:22edacf53f862c0abfa4c75213a7ab9aff843ef54f75553c9a6b82800045718b
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3184be6e7aa9630d2d04cfd8a5eadab72c53048d1d80c9edf3ff32dd527e46ff
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:154df71e85c6510d75f95f00a4893e9755fc5bddd41a29f2c39fc290a6432b34
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6825ef0f25e086c9dfb4332e3b7e10288e902088c249c87e47f6e24a13965ac
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b5a447733f53c516a2d17d1be010668b5887ffec8eded0377081439494bab7ef
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.04,
6
  "eval_steps": 1000,
7
- "global_step": 34000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2694,6 +2694,164 @@
2694
  "eval_samples_per_second": 147.895,
2695
  "eval_steps_per_second": 9.296,
2696
  "step": 34000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2697
  }
2698
  ],
2699
  "logging_steps": 100,
@@ -2713,7 +2871,7 @@
2713
  "attributes": {}
2714
  }
2715
  },
2716
- "total_flos": 1.33278669471744e+18,
2717
  "train_batch_size": 120,
2718
  "trial_name": null,
2719
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.08,
6
  "eval_steps": 1000,
7
+ "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2694
  "eval_samples_per_second": 147.895,
2695
  "eval_steps_per_second": 9.296,
2696
  "step": 34000
2697
+ },
2698
+ {
2699
+ "epoch": 0.042,
2700
+ "grad_norm": 0.15125946700572968,
2701
+ "learning_rate": 0.00023073112551075587,
2702
+ "loss": 3.0514120483398437,
2703
+ "step": 34100
2704
+ },
2705
+ {
2706
+ "epoch": 0.044,
2707
+ "grad_norm": 0.1368522197008133,
2708
+ "learning_rate": 0.00022808141471191557,
2709
+ "loss": 3.0052923583984374,
2710
+ "step": 34200
2711
+ },
2712
+ {
2713
+ "epoch": 0.046,
2714
+ "grad_norm": 0.153466135263443,
2715
+ "learning_rate": 0.00022544250349329565,
2716
+ "loss": 3.0354010009765626,
2717
+ "step": 34300
2718
+ },
2719
+ {
2720
+ "epoch": 0.048,
2721
+ "grad_norm": 0.14528165757656097,
2722
+ "learning_rate": 0.00022281449666249222,
2723
+ "loss": 3.0583941650390627,
2724
+ "step": 34400
2725
+ },
2726
+ {
2727
+ "epoch": 0.05,
2728
+ "grad_norm": 0.16713589429855347,
2729
+ "learning_rate": 0.0002201974985940215,
2730
+ "loss": 3.046011962890625,
2731
+ "step": 34500
2732
+ },
2733
+ {
2734
+ "epoch": 0.052,
2735
+ "grad_norm": 0.1621810495853424,
2736
+ "learning_rate": 0.00021759161322517184,
2737
+ "loss": 3.021026916503906,
2738
+ "step": 34600
2739
+ },
2740
+ {
2741
+ "epoch": 0.054,
2742
+ "grad_norm": 0.1539982110261917,
2743
+ "learning_rate": 0.00021499694405187797,
2744
+ "loss": 3.0086825561523436,
2745
+ "step": 34700
2746
+ },
2747
+ {
2748
+ "epoch": 0.056,
2749
+ "grad_norm": 0.1489582359790802,
2750
+ "learning_rate": 0.00021241359412460937,
2751
+ "loss": 3.038551025390625,
2752
+ "step": 34800
2753
+ },
2754
+ {
2755
+ "epoch": 0.058,
2756
+ "grad_norm": 0.15666431188583374,
2757
+ "learning_rate": 0.00020984166604427778,
2758
+ "loss": 3.0483087158203124,
2759
+ "step": 34900
2760
+ },
2761
+ {
2762
+ "epoch": 0.06,
2763
+ "grad_norm": 0.5827662348747253,
2764
+ "learning_rate": 0.00020728126195816238,
2765
+ "loss": 3.0230258178710936,
2766
+ "step": 35000
2767
+ },
2768
+ {
2769
+ "epoch": 0.06,
2770
+ "eval_accuracy": 0.38149883399825174,
2771
+ "eval_loss": 3.3024532794952393,
2772
+ "eval_runtime": 8.9187,
2773
+ "eval_samples_per_second": 217.633,
2774
+ "eval_steps_per_second": 13.679,
2775
+ "step": 35000
2776
+ },
2777
+ {
2778
+ "epoch": 0.062,
2779
+ "grad_norm": 0.14230984449386597,
2780
+ "learning_rate": 0.00020473248355585267,
2781
+ "loss": 3.0248248291015627,
2782
+ "step": 35100
2783
+ },
2784
+ {
2785
+ "epoch": 0.064,
2786
+ "grad_norm": 0.13632775843143463,
2787
+ "learning_rate": 0.00020219543206521,
2788
+ "loss": 3.0088482666015626,
2789
+ "step": 35200
2790
+ },
2791
+ {
2792
+ "epoch": 0.066,
2793
+ "grad_norm": 0.14240941405296326,
2794
+ "learning_rate": 0.00019967020824834648,
2795
+ "loss": 3.0288742065429686,
2796
+ "step": 35300
2797
+ },
2798
+ {
2799
+ "epoch": 0.068,
2800
+ "grad_norm": 0.16221162676811218,
2801
+ "learning_rate": 0.00019715691239762466,
2802
+ "loss": 3.007751770019531,
2803
+ "step": 35400
2804
+ },
2805
+ {
2806
+ "epoch": 0.07,
2807
+ "grad_norm": 0.15388672053813934,
2808
+ "learning_rate": 0.00019465564433167237,
2809
+ "loss": 3.0218731689453127,
2810
+ "step": 35500
2811
+ },
2812
+ {
2813
+ "epoch": 0.072,
2814
+ "grad_norm": 0.1594669222831726,
2815
+ "learning_rate": 0.0001921665033914196,
2816
+ "loss": 3.030206604003906,
2817
+ "step": 35600
2818
+ },
2819
+ {
2820
+ "epoch": 0.074,
2821
+ "grad_norm": 0.15590718388557434,
2822
+ "learning_rate": 0.00018968958843615263,
2823
+ "loss": 3.0212777709960936,
2824
+ "step": 35700
2825
+ },
2826
+ {
2827
+ "epoch": 0.076,
2828
+ "grad_norm": 0.19116833806037903,
2829
+ "learning_rate": 0.0001872249978395878,
2830
+ "loss": 3.014992370605469,
2831
+ "step": 35800
2832
+ },
2833
+ {
2834
+ "epoch": 0.078,
2835
+ "grad_norm": 0.1420874446630478,
2836
+ "learning_rate": 0.00018477282948596425,
2837
+ "loss": 3.003076171875,
2838
+ "step": 35900
2839
+ },
2840
+ {
2841
+ "epoch": 0.08,
2842
+ "grad_norm": 0.14843791723251343,
2843
+ "learning_rate": 0.00018233318076615667,
2844
+ "loss": 3.0334661865234374,
2845
+ "step": 36000
2846
+ },
2847
+ {
2848
+ "epoch": 0.08,
2849
+ "eval_accuracy": 0.3820614184993512,
2850
+ "eval_loss": 3.2988061904907227,
2851
+ "eval_runtime": 8.5815,
2852
+ "eval_samples_per_second": 226.185,
2853
+ "eval_steps_per_second": 14.217,
2854
+ "step": 36000
2855
  }
2856
  ],
2857
  "logging_steps": 100,
 
2871
  "attributes": {}
2872
  }
2873
  },
2874
+ "total_flos": 1.41118591205376e+18,
2875
  "train_batch_size": 120,
2876
  "trial_name": null,
2877
  "trial_params": null