CodeIsAbstract commited on
Commit
eea9797
·
verified ·
1 Parent(s): 25e5f75

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6974752fe7a3b14089ab2e7053fc764710e6d0ddbd1d7314081904519abc3ca2
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f892f103723425ad14b38db6edb9aa4c6ee7b516549322cb183cac27a585cd69
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3da8b0c6ec3a28101585919798344dd1a4d2e616af339b16dae843b9a7667d51
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67301da5c57c64e98efb14ecf294eab870a5a0a4e3004242e9e40668fa0a2e5b
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43badff6712b33504459f872ac58be74afdf5ffa731e863439dbed8d3f3b0658
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b4b76f2c1c39522b8c5498459c7e9113970762ae114e0e7f40cb836dc8984719
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12eac59d8227e48f0e6fc2526645da1f69a0138bb2904859d6e0d18216ba63d7
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f50e3306f3c9c6adbe0dedc2fd97abc1244264dc24b6104e59c24eb9f7554450
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c12b9a050992357585c15ce67795abc69719e584223ec4ccbae89612840d277
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:34913556519b4f5ff6ee3a99f05254dcc26ba55b30a512aa1a424a8163b3da18
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:da83cd5eef961d44e4becc44e4089edf261063947ef5a08e30404cfbd611e2c3
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a622b9945abec665a29f44d02e606edae0902d7c858f03471fc439ffb78c8bd
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bfd920f545c0711cea231c43d68bcba64df0cc8fd9e536bc95aea4ceced55b64
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c2db92a63182f2a8e74f6f8ba8a49330276140c1b1e68be053bb7958ff88aa05
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e5e310e83ca23aa2c5186dec200b69809f8eb565d4f17f262d28f343cd531776
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c1dfcf1def0fab315df7efa0e82058c17d657f7d1e24a0863e66bae75531cfbf
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:67aa5039be38b0a461bcbf305aaf4eb5ca5764f1670f75af2089e433b412f3af
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7d41bf7a36b12290d0668a17db5481f4d9bac1fb69402aa3433a16ac534f0d78
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b2c15db3462043815a7b702c6ec5d03f93688f7392547928696d4f1b5bc14356
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e3f52f2981f6a10064b9d54b23539047132389d938a031ae78cc1e296c5a1748
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e31938cdcda9330861d9a6c5fa22f1c1b3736037ca4758e7ec127110a6ee0d1a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d5d1546bfb69154c942dddcbbbdf46b8cffa1900144b2ba63a8323dfd94157b
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.16666666666666666,
6
  "eval_steps": 250,
7
- "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -96,6 +96,94 @@
96
  "eval_samples_per_second": 16.827,
97
  "eval_steps_per_second": 0.538,
98
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
99
  }
100
  ],
101
  "logging_steps": 50,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.3333333333333333,
6
  "eval_steps": 250,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
96
  "eval_samples_per_second": 16.827,
97
  "eval_steps_per_second": 0.538,
98
  "step": 500
99
+ },
100
+ {
101
+ "epoch": 0.18333333333333332,
102
+ "grad_norm": 330.5375061035156,
103
+ "learning_rate": 0.0003809654104932039,
104
+ "loss": 1856.43828125,
105
+ "step": 550
106
+ },
107
+ {
108
+ "epoch": 0.2,
109
+ "grad_norm": 462.14501953125,
110
+ "learning_rate": 0.00037599957197558643,
111
+ "loss": 1845.416875,
112
+ "step": 600
113
+ },
114
+ {
115
+ "epoch": 0.21666666666666667,
116
+ "grad_norm": 348.27215576171875,
117
+ "learning_rate": 0.0003704992285423923,
118
+ "loss": 1836.4821875,
119
+ "step": 650
120
+ },
121
+ {
122
+ "epoch": 0.23333333333333334,
123
+ "grad_norm": 348.34326171875,
124
+ "learning_rate": 0.0003644810845558505,
125
+ "loss": 1828.6059375,
126
+ "step": 700
127
+ },
128
+ {
129
+ "epoch": 0.25,
130
+ "grad_norm": 478.7774353027344,
131
+ "learning_rate": 0.00035796341692145183,
132
+ "loss": 1821.568125,
133
+ "step": 750
134
+ },
135
+ {
136
+ "epoch": 0.25,
137
+ "eval_accuracy": 0.1680097751710655,
138
+ "eval_loss": 60.21141815185547,
139
+ "eval_runtime": 7.4388,
140
+ "eval_samples_per_second": 16.804,
141
+ "eval_steps_per_second": 0.538,
142
+ "step": 750
143
+ },
144
+ {
145
+ "epoch": 0.26666666666666666,
146
+ "grad_norm": 353.61517333984375,
147
+ "learning_rate": 0.0003509660195815883,
148
+ "loss": 1817.4553125,
149
+ "step": 800
150
+ },
151
+ {
152
+ "epoch": 0.2833333333333333,
153
+ "grad_norm": 335.4817810058594,
154
+ "learning_rate": 0.0003435101434020002,
155
+ "loss": 1810.23265625,
156
+ "step": 850
157
+ },
158
+ {
159
+ "epoch": 0.3,
160
+ "grad_norm": 580.8027954101562,
161
+ "learning_rate": 0.0003356184316335953,
162
+ "loss": 1802.5071875,
163
+ "step": 900
164
+ },
165
+ {
166
+ "epoch": 0.31666666666666665,
167
+ "grad_norm": 463.22943115234375,
168
+ "learning_rate": 0.00032731485114564034,
169
+ "loss": 1796.794375,
170
+ "step": 950
171
+ },
172
+ {
173
+ "epoch": 0.3333333333333333,
174
+ "grad_norm": 435.9424133300781,
175
+ "learning_rate": 0.0003186246196391665,
176
+ "loss": 1792.77296875,
177
+ "step": 1000
178
+ },
179
+ {
180
+ "epoch": 0.3333333333333333,
181
+ "eval_accuracy": 0.1721837732160313,
182
+ "eval_loss": 59.25102233886719,
183
+ "eval_runtime": 7.0046,
184
+ "eval_samples_per_second": 17.845,
185
+ "eval_steps_per_second": 0.571,
186
+ "step": 1000
187
  }
188
  ],
189
  "logging_steps": 50,