CodeIsAbstract commited on
Commit
3ab3b26
·
verified ·
1 Parent(s): b627104

Training in progress, step 36000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eacd16009db792813323d990cfe5dae9dfaf2c1b2b6746f6ce4ee5c5abd05fa3
3
  size 529337896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d197cd9a3404a8c8f67910780423aecd5fdc3add3f182e0d82ab6682395c0006
3
  size 529337896
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:86b59222e117c9663552ff919bf9d06c38349de96bfd51727eb1c2c51c1eef72
3
  size 871247243
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:419fd2c3e0ba5d1952d3f93465eeb172ba6dac85f67c82c9eeb58a6bc214cded
3
  size 871247243
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:efa0171c7cd70527d3a922da6317b15994f35a839e2d8505c68b0fc08f030110
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6a202a832e7dde4f4f295f57293a1d0bb6e0f939cebe50d0b66fc7a8608bef7a
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dcc336e7edab566b72f19c8cdbf557762e0e53e457ed4985a87b74bd22371a59
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d525574f54acfd6c1c0d9e8a984a9fe4782c94abdc8722296a2554b91af7e0e9
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.68,
6
  "eval_steps": 1000,
7
- "global_step": 34000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2694,6 +2694,164 @@
2694
  "eval_samples_per_second": 88.642,
2695
  "eval_steps_per_second": 0.709,
2696
  "step": 34000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2697
  }
2698
  ],
2699
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.72,
6
  "eval_steps": 1000,
7
+ "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2694
  "eval_samples_per_second": 88.642,
2695
  "eval_steps_per_second": 0.709,
2696
  "step": 34000
2697
+ },
2698
+ {
2699
+ "epoch": 0.682,
2700
+ "grad_norm": 0.1495920717716217,
2701
+ "learning_rate": 0.000503880797188073,
2702
+ "loss": 3.2605,
2703
+ "step": 34100
2704
+ },
2705
+ {
2706
+ "epoch": 0.684,
2707
+ "grad_norm": 0.16431903839111328,
2708
+ "learning_rate": 0.0004981491600794544,
2709
+ "loss": 3.2526,
2710
+ "step": 34200
2711
+ },
2712
+ {
2713
+ "epoch": 0.686,
2714
+ "grad_norm": 0.1864980161190033,
2715
+ "learning_rate": 0.0004924394755523447,
2716
+ "loss": 3.2562,
2717
+ "step": 34300
2718
+ },
2719
+ {
2720
+ "epoch": 0.688,
2721
+ "grad_norm": 0.1656457781791687,
2722
+ "learning_rate": 0.0004867519933668425,
2723
+ "loss": 3.2649,
2724
+ "step": 34400
2725
+ },
2726
+ {
2727
+ "epoch": 0.69,
2728
+ "grad_norm": 0.165181502699852,
2729
+ "learning_rate": 0.0004810869623118437,
2730
+ "loss": 3.257,
2731
+ "step": 34500
2732
+ },
2733
+ {
2734
+ "epoch": 0.692,
2735
+ "grad_norm": 0.13930366933345795,
2736
+ "learning_rate": 0.0004754446301941582,
2737
+ "loss": 3.2525,
2738
+ "step": 34600
2739
+ },
2740
+ {
2741
+ "epoch": 0.694,
2742
+ "grad_norm": 0.1519293636083603,
2743
+ "learning_rate": 0.00046982524382767213,
2744
+ "loss": 3.2596,
2745
+ "step": 34700
2746
+ },
2747
+ {
2748
+ "epoch": 0.696,
2749
+ "grad_norm": 0.14039067924022675,
2750
+ "learning_rate": 0.0004642290490225486,
2751
+ "loss": 3.2582,
2752
+ "step": 34800
2753
+ },
2754
+ {
2755
+ "epoch": 0.698,
2756
+ "grad_norm": 0.14533616602420807,
2757
+ "learning_rate": 0.0004586562905744784,
2758
+ "loss": 3.237,
2759
+ "step": 34900
2760
+ },
2761
+ {
2762
+ "epoch": 0.7,
2763
+ "grad_norm": 0.21432434022426605,
2764
+ "learning_rate": 0.00045310721225396854,
2765
+ "loss": 3.2596,
2766
+ "step": 35000
2767
+ },
2768
+ {
2769
+ "epoch": 0.7,
2770
+ "eval_accuracy": 0.38947553816046965,
2771
+ "eval_loss": 3.2606313228607178,
2772
+ "eval_runtime": 10.6118,
2773
+ "eval_samples_per_second": 94.235,
2774
+ "eval_steps_per_second": 0.754,
2775
+ "step": 35000
2776
+ },
2777
+ {
2778
+ "epoch": 0.702,
2779
+ "grad_norm": 0.15394216775894165,
2780
+ "learning_rate": 0.00044758205679568163,
2781
+ "loss": 3.2522,
2782
+ "step": 35100
2783
+ },
2784
+ {
2785
+ "epoch": 0.704,
2786
+ "grad_norm": 0.14649616181850433,
2787
+ "learning_rate": 0.00044208106588781694,
2788
+ "loss": 3.2523,
2789
+ "step": 35200
2790
+ },
2791
+ {
2792
+ "epoch": 0.706,
2793
+ "grad_norm": 0.13892528414726257,
2794
+ "learning_rate": 0.00043660448016153656,
2795
+ "loss": 3.2486,
2796
+ "step": 35300
2797
+ },
2798
+ {
2799
+ "epoch": 0.708,
2800
+ "grad_norm": 0.16032464802265167,
2801
+ "learning_rate": 0.0004311525391804426,
2802
+ "loss": 3.2464,
2803
+ "step": 35400
2804
+ },
2805
+ {
2806
+ "epoch": 0.71,
2807
+ "grad_norm": 0.14114409685134888,
2808
+ "learning_rate": 0.0004257254814300947,
2809
+ "loss": 3.2546,
2810
+ "step": 35500
2811
+ },
2812
+ {
2813
+ "epoch": 0.712,
2814
+ "grad_norm": 0.15264350175857544,
2815
+ "learning_rate": 0.000420323544307581,
2816
+ "loss": 3.2436,
2817
+ "step": 35600
2818
+ },
2819
+ {
2820
+ "epoch": 0.714,
2821
+ "grad_norm": 0.21544255316257477,
2822
+ "learning_rate": 0.0004149469641111302,
2823
+ "loss": 3.2318,
2824
+ "step": 35700
2825
+ },
2826
+ {
2827
+ "epoch": 0.716,
2828
+ "grad_norm": 0.16077803075313568,
2829
+ "learning_rate": 0.00040959597602977806,
2830
+ "loss": 3.2545,
2831
+ "step": 35800
2832
+ },
2833
+ {
2834
+ "epoch": 0.718,
2835
+ "grad_norm": 0.14488866925239563,
2836
+ "learning_rate": 0.00040427081413307864,
2837
+ "loss": 3.2526,
2838
+ "step": 35900
2839
+ },
2840
+ {
2841
+ "epoch": 0.72,
2842
+ "grad_norm": 0.1400013417005539,
2843
+ "learning_rate": 0.00039897171136086363,
2844
+ "loss": 3.2217,
2845
+ "step": 36000
2846
+ },
2847
+ {
2848
+ "epoch": 0.72,
2849
+ "eval_accuracy": 0.3897240704500978,
2850
+ "eval_loss": 3.257254123687744,
2851
+ "eval_runtime": 11.7287,
2852
+ "eval_samples_per_second": 85.261,
2853
+ "eval_steps_per_second": 0.682,
2854
+ "step": 36000
2855
  }
2856
  ],
2857
  "logging_steps": 100,