CodeIsAbstract commited on
Commit
bbb2e09
·
verified ·
1 Parent(s): 996a83f

Training in progress, step 14000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:38983cc6baa199b78ad222dae4c477c4bd0c63e2458372366630348ee28383b5
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54139b974ddfae98ee7fb4da2695b05278ea502ebc86824b3e1c4f49c129b297
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7fd8341a68b63315868b34bd0e54e5f6c54b4a66537f640753aed597e0d3b48d
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff8e73939ebcb2d4e5ba75a94606f6611749ef65cbf2555cdefc0014ed49429b
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4d31404d4c654929d8c8604a09997c74ec2ccd4b35e0d5562c9b8d160dc81e06
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4379e0a27d0683dd964c1d70a99193598c03adb1d5cf2808558afd18cfc5f420
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:51b5657716e2096af61b1cdab76b9fd3dc4ab393f10b4ad41400692973ae8e3f
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9fe9a6301d1d1aeac29e6021bcb10e2a37481f63ed32dd3e17e4ef1f6ab45e17
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.24,
6
  "eval_steps": 1000,
7
- "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -956,6 +956,164 @@
956
  "eval_samples_per_second": 290.897,
957
  "eval_steps_per_second": 18.284,
958
  "step": 12000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
959
  }
960
  ],
961
  "logging_steps": 100,
@@ -975,7 +1133,7 @@
975
  "attributes": {}
976
  }
977
  },
978
- "total_flos": 4.7039530401792e+17,
979
  "train_batch_size": 120,
980
  "trial_name": null,
981
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.28,
6
  "eval_steps": 1000,
7
+ "global_step": 14000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
956
  "eval_samples_per_second": 290.897,
957
  "eval_steps_per_second": 18.284,
958
  "step": 12000
959
+ },
960
+ {
961
+ "epoch": 0.242,
962
+ "grad_norm": 0.1405150443315506,
963
+ "learning_rate": 0.0008648077066991736,
964
+ "loss": 3.527177734375,
965
+ "step": 12100
966
+ },
967
+ {
968
+ "epoch": 0.244,
969
+ "grad_norm": 0.16479545831680298,
970
+ "learning_rate": 0.0008626456096552858,
971
+ "loss": 3.5393338012695312,
972
+ "step": 12200
973
+ },
974
+ {
975
+ "epoch": 0.246,
976
+ "grad_norm": 0.1583888679742813,
977
+ "learning_rate": 0.0008604691096960961,
978
+ "loss": 3.5796047973632814,
979
+ "step": 12300
980
+ },
981
+ {
982
+ "epoch": 0.248,
983
+ "grad_norm": 0.17304493486881256,
984
+ "learning_rate": 0.0008582782932639702,
985
+ "loss": 3.591195068359375,
986
+ "step": 12400
987
+ },
988
+ {
989
+ "epoch": 0.25,
990
+ "grad_norm": 0.15857121348381042,
991
+ "learning_rate": 0.0008560732473698707,
992
+ "loss": 3.584400634765625,
993
+ "step": 12500
994
+ },
995
+ {
996
+ "epoch": 0.252,
997
+ "grad_norm": 0.14658208191394806,
998
+ "learning_rate": 0.0008538540595899008,
999
+ "loss": 3.589564208984375,
1000
+ "step": 12600
1001
+ },
1002
+ {
1003
+ "epoch": 0.254,
1004
+ "grad_norm": 0.143401101231575,
1005
+ "learning_rate": 0.0008516208180618252,
1006
+ "loss": 3.5824600219726563,
1007
+ "step": 12700
1008
+ },
1009
+ {
1010
+ "epoch": 0.256,
1011
+ "grad_norm": 0.16244202852249146,
1012
+ "learning_rate": 0.0008493736114815718,
1013
+ "loss": 3.574875793457031,
1014
+ "step": 12800
1015
+ },
1016
+ {
1017
+ "epoch": 0.258,
1018
+ "grad_norm": 0.1520756632089615,
1019
+ "learning_rate": 0.0008471125290997063,
1020
+ "loss": 3.568282165527344,
1021
+ "step": 12900
1022
+ },
1023
+ {
1024
+ "epoch": 0.26,
1025
+ "grad_norm": 0.1582706868648529,
1026
+ "learning_rate": 0.00084483766071789,
1027
+ "loss": 3.5930340576171873,
1028
+ "step": 13000
1029
+ },
1030
+ {
1031
+ "epoch": 0.26,
1032
+ "eval_accuracy": 0.35929186944409996,
1033
+ "eval_loss": 3.5190227031707764,
1034
+ "eval_runtime": 6.3791,
1035
+ "eval_samples_per_second": 304.275,
1036
+ "eval_steps_per_second": 19.125,
1037
+ "step": 13000
1038
+ },
1039
+ {
1040
+ "epoch": 0.262,
1041
+ "grad_norm": 0.17478469014167786,
1042
+ "learning_rate": 0.0008425490966853113,
1043
+ "loss": 3.5665899658203126,
1044
+ "step": 13100
1045
+ },
1046
+ {
1047
+ "epoch": 0.264,
1048
+ "grad_norm": 0.15228988230228424,
1049
+ "learning_rate": 0.0008402469278950989,
1050
+ "loss": 3.5444100952148436,
1051
+ "step": 13200
1052
+ },
1053
+ {
1054
+ "epoch": 0.266,
1055
+ "grad_norm": 0.15244163572788239,
1056
+ "learning_rate": 0.0008379312457807102,
1057
+ "loss": 3.5695867919921875,
1058
+ "step": 13300
1059
+ },
1060
+ {
1061
+ "epoch": 0.268,
1062
+ "grad_norm": 0.15935847163200378,
1063
+ "learning_rate": 0.0008356021423123017,
1064
+ "loss": 3.5604925537109375,
1065
+ "step": 13400
1066
+ },
1067
+ {
1068
+ "epoch": 0.27,
1069
+ "grad_norm": 0.15347428619861603,
1070
+ "learning_rate": 0.000833259709993075,
1071
+ "loss": 3.5551889038085935,
1072
+ "step": 13500
1073
+ },
1074
+ {
1075
+ "epoch": 0.272,
1076
+ "grad_norm": 0.21459928154945374,
1077
+ "learning_rate": 0.0008309040418556038,
1078
+ "loss": 3.550736389160156,
1079
+ "step": 13600
1080
+ },
1081
+ {
1082
+ "epoch": 0.274,
1083
+ "grad_norm": 0.17246893048286438,
1084
+ "learning_rate": 0.0008285352314581378,
1085
+ "loss": 3.5650180053710936,
1086
+ "step": 13700
1087
+ },
1088
+ {
1089
+ "epoch": 0.276,
1090
+ "grad_norm": 0.16454122960567474,
1091
+ "learning_rate": 0.0008261533728808883,
1092
+ "loss": 3.549154968261719,
1093
+ "step": 13800
1094
+ },
1095
+ {
1096
+ "epoch": 0.278,
1097
+ "grad_norm": 0.14885519444942474,
1098
+ "learning_rate": 0.000823758560722291,
1099
+ "loss": 3.5430950927734375,
1100
+ "step": 13900
1101
+ },
1102
+ {
1103
+ "epoch": 0.28,
1104
+ "grad_norm": 0.1521039605140686,
1105
+ "learning_rate": 0.0008213508900952489,
1106
+ "loss": 3.547527160644531,
1107
+ "step": 14000
1108
+ },
1109
+ {
1110
+ "epoch": 0.28,
1111
+ "eval_accuracy": 0.36159261824608735,
1112
+ "eval_loss": 3.4961202144622803,
1113
+ "eval_runtime": 6.3742,
1114
+ "eval_samples_per_second": 304.507,
1115
+ "eval_steps_per_second": 19.14,
1116
+ "step": 14000
1117
  }
1118
  ],
1119
  "logging_steps": 100,
 
1133
  "attributes": {}
1134
  }
1135
  },
1136
+ "total_flos": 5.4879452135424e+17,
1137
  "train_batch_size": 120,
1138
  "trial_name": null,
1139
  "trial_params": null