CodeIsAbstract commited on
Commit
1262efc
·
verified ·
1 Parent(s): b41fae3

Training in progress, step 38000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0e61997d322b558147c11cb3a841b6c879522f4e0401570393ee65009491b13e
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb666cb8677bceaafadd9b9b5e659661ea31b2586e5ed485eda4a008d0ef5ef9
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:22edacf53f862c0abfa4c75213a7ab9aff843ef54f75553c9a6b82800045718b
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7d2d23f0c5c7e252c38dfbeb64477cc0089f8a30877e6082b7e1a3a74d55684d
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:154df71e85c6510d75f95f00a4893e9755fc5bddd41a29f2c39fc290a6432b34
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51f1ebff0433542b44861a56d7924f69b6e5f3c7271e7aae262253030bc6e5a0
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b5a447733f53c516a2d17d1be010668b5887ffec8eded0377081439494bab7ef
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c07a50c67c728026d8510c54dbf23cc51e59527759deefd728a51b01b02ec546
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 36000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2852,6 +2852,164 @@
2852
  "eval_samples_per_second": 226.185,
2853
  "eval_steps_per_second": 14.217,
2854
  "step": 36000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2855
  }
2856
  ],
2857
  "logging_steps": 100,
@@ -2871,7 +3029,7 @@
2871
  "attributes": {}
2872
  }
2873
  },
2874
- "total_flos": 1.41118591205376e+18,
2875
  "train_batch_size": 120,
2876
  "trial_name": null,
2877
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 1000,
7
+ "global_step": 38000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2852
  "eval_samples_per_second": 226.185,
2853
  "eval_steps_per_second": 14.217,
2854
  "step": 36000
2855
+ },
2856
+ {
2857
+ "epoch": 0.082,
2858
+ "grad_norm": 0.17176257073879242,
2859
+ "learning_rate": 0.00017990614857380728,
2860
+ "loss": 3.0163888549804687,
2861
+ "step": 36100
2862
+ },
2863
+ {
2864
+ "epoch": 0.084,
2865
+ "grad_norm": 0.13782846927642822,
2866
+ "learning_rate": 0.00017749182930147656,
2867
+ "loss": 3.010736999511719,
2868
+ "step": 36200
2869
+ },
2870
+ {
2871
+ "epoch": 0.086,
2872
+ "grad_norm": 0.14697974920272827,
2873
+ "learning_rate": 0.00017509031883681714,
2874
+ "loss": 3.0050363159179687,
2875
+ "step": 36300
2876
+ },
2877
+ {
2878
+ "epoch": 0.088,
2879
+ "grad_norm": 0.1496117264032364,
2880
+ "learning_rate": 0.00017270171255876328,
2881
+ "loss": 2.994004211425781,
2882
+ "step": 36400
2883
+ },
2884
+ {
2885
+ "epoch": 0.09,
2886
+ "grad_norm": 0.14235389232635498,
2887
+ "learning_rate": 0.00017032610533374338,
2888
+ "loss": 3.0134628295898436,
2889
+ "step": 36500
2890
+ },
2891
+ {
2892
+ "epoch": 0.092,
2893
+ "grad_norm": 0.16272151470184326,
2894
+ "learning_rate": 0.0001679635915119136,
2895
+ "loss": 2.9897280883789064,
2896
+ "step": 36600
2897
+ },
2898
+ {
2899
+ "epoch": 0.094,
2900
+ "grad_norm": 0.15706537663936615,
2901
+ "learning_rate": 0.00016561426492340826,
2902
+ "loss": 3.014703369140625,
2903
+ "step": 36700
2904
+ },
2905
+ {
2906
+ "epoch": 0.096,
2907
+ "grad_norm": 0.149558886885643,
2908
+ "learning_rate": 0.0001632782188746153,
2909
+ "loss": 2.9928765869140626,
2910
+ "step": 36800
2911
+ },
2912
+ {
2913
+ "epoch": 0.098,
2914
+ "grad_norm": 0.1528598964214325,
2915
+ "learning_rate": 0.00016095554614446955,
2916
+ "loss": 2.9844891357421877,
2917
+ "step": 36900
2918
+ },
2919
+ {
2920
+ "epoch": 0.1,
2921
+ "grad_norm": 0.15897603332996368,
2922
+ "learning_rate": 0.00015864633898076812,
2923
+ "loss": 3.0097698974609375,
2924
+ "step": 37000
2925
+ },
2926
+ {
2927
+ "epoch": 0.1,
2928
+ "eval_accuracy": 0.38257762506666826,
2929
+ "eval_loss": 3.2958788871765137,
2930
+ "eval_runtime": 9.1504,
2931
+ "eval_samples_per_second": 212.121,
2932
+ "eval_steps_per_second": 13.333,
2933
+ "step": 37000
2934
+ },
2935
+ {
2936
+ "epoch": 0.102,
2937
+ "grad_norm": 0.14843960106372833,
2938
+ "learning_rate": 0.00015635068909650656,
2939
+ "loss": 2.975020446777344,
2940
+ "step": 37100
2941
+ },
2942
+ {
2943
+ "epoch": 0.104,
2944
+ "grad_norm": 0.15033408999443054,
2945
+ "learning_rate": 0.0001540686876662365,
2946
+ "loss": 3.0080502319335936,
2947
+ "step": 37200
2948
+ },
2949
+ {
2950
+ "epoch": 0.106,
2951
+ "grad_norm": 0.15051966905593872,
2952
+ "learning_rate": 0.00015180042532244448,
2953
+ "loss": 2.989608154296875,
2954
+ "step": 37300
2955
+ },
2956
+ {
2957
+ "epoch": 0.108,
2958
+ "grad_norm": 0.13960620760917664,
2959
+ "learning_rate": 0.00014954599215195196,
2960
+ "loss": 3.0019635009765624,
2961
+ "step": 37400
2962
+ },
2963
+ {
2964
+ "epoch": 0.11,
2965
+ "grad_norm": 0.14864717423915863,
2966
+ "learning_rate": 0.00014730547769233876,
2967
+ "loss": 2.99054931640625,
2968
+ "step": 37500
2969
+ },
2970
+ {
2971
+ "epoch": 0.112,
2972
+ "grad_norm": 0.13497526943683624,
2973
+ "learning_rate": 0.00014507897092838496,
2974
+ "loss": 2.9777395629882815,
2975
+ "step": 37600
2976
+ },
2977
+ {
2978
+ "epoch": 0.114,
2979
+ "grad_norm": 0.14896409213542938,
2980
+ "learning_rate": 0.00014286656028853845,
2981
+ "loss": 2.983952941894531,
2982
+ "step": 37700
2983
+ },
2984
+ {
2985
+ "epoch": 0.116,
2986
+ "grad_norm": 0.14890077710151672,
2987
+ "learning_rate": 0.00014066833364140215,
2988
+ "loss": 2.9580343627929686,
2989
+ "step": 37800
2990
+ },
2991
+ {
2992
+ "epoch": 0.118,
2993
+ "grad_norm": 0.1506078690290451,
2994
+ "learning_rate": 0.0001384843782922442,
2995
+ "loss": 3.0047003173828126,
2996
+ "step": 37900
2997
+ },
2998
+ {
2999
+ "epoch": 0.12,
3000
+ "grad_norm": 0.13839152455329895,
3001
+ "learning_rate": 0.0001363147809795307,
3002
+ "loss": 3.0096331787109376,
3003
+ "step": 38000
3004
+ },
3005
+ {
3006
+ "epoch": 0.12,
3007
+ "eval_accuracy": 0.38263912623972757,
3008
+ "eval_loss": 3.293242931365967,
3009
+ "eval_runtime": 8.7928,
3010
+ "eval_samples_per_second": 220.749,
3011
+ "eval_steps_per_second": 13.875,
3012
+ "step": 38000
3013
  }
3014
  ],
3015
  "logging_steps": 100,
 
3029
  "attributes": {}
3030
  }
3031
  },
3032
+ "total_flos": 1.48958512939008e+18,
3033
  "train_batch_size": 120,
3034
  "trial_name": null,
3035
  "trial_params": null