CodeIsAbstract commited on
Commit
a5b9691
·
verified ·
1 Parent(s): 7670e9d

Training in progress, step 14000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c622298879cc833d1bdcc0a29bd5f4d753b0f9b392bcfe41b3a3720e37227629
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2e0ff05e0e5d852455b4d83922c7abbecdec428e19e46c0b823ee661ac50cdb5
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f770627df18e9419658687702329dfc460bbc5339dfe15ca4d75788aa1849e5a
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:32d116e4c03c1c5f0b5164cbe30850c283753e36a9119b42ffefe95697cc08e3
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e52021f1e1b36f10ce9c865a52c47d6a63c888218cae64f20627598071e5030f
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:74ba315e55e014287fc4f548e786e1c64991f07f2dd31f970e2f14eab773946e
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:382f69ae5e5420217ac9e6a070a414359d614c97fe4b4d34a11abdd073f1e7c2
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:238cbdda4406a9b9e4f702cada921a66b6ebd035a4f580dabf57521e3dc0c97d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.24,
6
  "eval_steps": 1000,
7
- "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -956,6 +956,164 @@
956
  "eval_samples_per_second": 342.004,
957
  "eval_steps_per_second": 21.496,
958
  "step": 12000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
959
  }
960
  ],
961
  "logging_steps": 100,
@@ -975,7 +1133,7 @@
975
  "attributes": {}
976
  }
977
  },
978
- "total_flos": 3.7626052608e+17,
979
  "train_batch_size": 120,
980
  "trial_name": null,
981
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.28,
6
  "eval_steps": 1000,
7
+ "global_step": 14000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
956
  "eval_samples_per_second": 342.004,
957
  "eval_steps_per_second": 21.496,
958
  "step": 12000
959
+ },
960
+ {
961
+ "epoch": 0.242,
962
+ "grad_norm": 0.19794468581676483,
963
+ "learning_rate": 0.0005188846240195041,
964
+ "loss": 3.563157958984375,
965
+ "step": 12100
966
+ },
967
+ {
968
+ "epoch": 0.244,
969
+ "grad_norm": 0.22670620679855347,
970
+ "learning_rate": 0.0005175873657931715,
971
+ "loss": 3.573887939453125,
972
+ "step": 12200
973
+ },
974
+ {
975
+ "epoch": 0.246,
976
+ "grad_norm": 0.2241670936346054,
977
+ "learning_rate": 0.0005162814658176576,
978
+ "loss": 3.614027099609375,
979
+ "step": 12300
980
+ },
981
+ {
982
+ "epoch": 0.248,
983
+ "grad_norm": 0.23707276582717896,
984
+ "learning_rate": 0.000514966975958382,
985
+ "loss": 3.628121643066406,
986
+ "step": 12400
987
+ },
988
+ {
989
+ "epoch": 0.25,
990
+ "grad_norm": 0.22935494780540466,
991
+ "learning_rate": 0.0005136439484219223,
992
+ "loss": 3.6195343017578123,
993
+ "step": 12500
994
+ },
995
+ {
996
+ "epoch": 0.252,
997
+ "grad_norm": 0.21692432463169098,
998
+ "learning_rate": 0.0005123124357539403,
999
+ "loss": 3.62494140625,
1000
+ "step": 12600
1001
+ },
1002
+ {
1003
+ "epoch": 0.254,
1004
+ "grad_norm": 0.1959449201822281,
1005
+ "learning_rate": 0.0005109724908370951,
1006
+ "loss": 3.619110412597656,
1007
+ "step": 12700
1008
+ },
1009
+ {
1010
+ "epoch": 0.256,
1011
+ "grad_norm": 0.22109989821910858,
1012
+ "learning_rate": 0.000509624166888943,
1013
+ "loss": 3.609862060546875,
1014
+ "step": 12800
1015
+ },
1016
+ {
1017
+ "epoch": 0.258,
1018
+ "grad_norm": 0.20637010037899017,
1019
+ "learning_rate": 0.0005082675174598238,
1020
+ "loss": 3.6038763427734377,
1021
+ "step": 12900
1022
+ },
1023
+ {
1024
+ "epoch": 0.26,
1025
+ "grad_norm": 0.21297694742679596,
1026
+ "learning_rate": 0.000506902596430734,
1027
+ "loss": 3.627516174316406,
1028
+ "step": 13000
1029
+ },
1030
+ {
1031
+ "epoch": 0.26,
1032
+ "eval_accuracy": 0.35723510890244603,
1033
+ "eval_loss": 3.5337636470794678,
1034
+ "eval_runtime": 5.6987,
1035
+ "eval_samples_per_second": 340.603,
1036
+ "eval_steps_per_second": 21.408,
1037
+ "step": 13000
1038
+ },
1039
+ {
1040
+ "epoch": 0.262,
1041
+ "grad_norm": 0.2433704286813736,
1042
+ "learning_rate": 0.0005055294580111867,
1043
+ "loss": 3.600766296386719,
1044
+ "step": 13100
1045
+ },
1046
+ {
1047
+ "epoch": 0.264,
1048
+ "grad_norm": 0.20984166860580444,
1049
+ "learning_rate": 0.0005041481567370593,
1050
+ "loss": 3.57820556640625,
1051
+ "step": 13200
1052
+ },
1053
+ {
1054
+ "epoch": 0.266,
1055
+ "grad_norm": 0.22343853116035461,
1056
+ "learning_rate": 0.0005027587474684261,
1057
+ "loss": 3.603807373046875,
1058
+ "step": 13300
1059
+ },
1060
+ {
1061
+ "epoch": 0.268,
1062
+ "grad_norm": 0.2158922404050827,
1063
+ "learning_rate": 0.0005013612853873809,
1064
+ "loss": 3.5946688842773438,
1065
+ "step": 13400
1066
+ },
1067
+ {
1068
+ "epoch": 0.27,
1069
+ "grad_norm": 0.21097783744335175,
1070
+ "learning_rate": 0.0004999558259958449,
1071
+ "loss": 3.589884033203125,
1072
+ "step": 13500
1073
+ },
1074
+ {
1075
+ "epoch": 0.272,
1076
+ "grad_norm": 0.28342482447624207,
1077
+ "learning_rate": 0.0004985424251133622,
1078
+ "loss": 3.5845135498046874,
1079
+ "step": 13600
1080
+ },
1081
+ {
1082
+ "epoch": 0.274,
1083
+ "grad_norm": 0.22716180980205536,
1084
+ "learning_rate": 0.0004971211388748826,
1085
+ "loss": 3.601893310546875,
1086
+ "step": 13700
1087
+ },
1088
+ {
1089
+ "epoch": 0.276,
1090
+ "grad_norm": 0.2315467894077301,
1091
+ "learning_rate": 0.000495692023728533,
1092
+ "loss": 3.583456726074219,
1093
+ "step": 13800
1094
+ },
1095
+ {
1096
+ "epoch": 0.278,
1097
+ "grad_norm": 0.20143155753612518,
1098
+ "learning_rate": 0.0004942551364333746,
1099
+ "loss": 3.5840383911132814,
1100
+ "step": 13900
1101
+ },
1102
+ {
1103
+ "epoch": 0.28,
1104
+ "grad_norm": 0.2158096581697464,
1105
+ "learning_rate": 0.0004928105340571493,
1106
+ "loss": 3.5770263671875,
1107
+ "step": 14000
1108
+ },
1109
+ {
1110
+ "epoch": 0.28,
1111
+ "eval_accuracy": 0.3599905630986912,
1112
+ "eval_loss": 3.508211851119995,
1113
+ "eval_runtime": 9.9993,
1114
+ "eval_samples_per_second": 194.114,
1115
+ "eval_steps_per_second": 12.201,
1116
+ "step": 14000
1117
  }
1118
  ],
1119
  "logging_steps": 100,
 
1133
  "attributes": {}
1134
  }
1135
  },
1136
+ "total_flos": 4.3897061376e+17,
1137
  "train_batch_size": 120,
1138
  "trial_name": null,
1139
  "trial_params": null