CodeIsAbstract commited on
Commit
8b37841
·
verified ·
1 Parent(s): 3c3543e

Training in progress, step 800, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:28169bb183740976fb1e1012281faf4daaa901f0809070b5431b167a9aed9330
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73f8c1791b5ee7e58d663de47ee5fe8c35df4f1f6931e02c5c1c05ca2c095b6a
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5bcc09d2987fc871b0f3d765698df5a7ddb67b242d4bf83d5e4d4194e97c62f1
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be87b0db9147168393b32986a7c8b1d815025dfeb0ce1175bbb3f22f44725ce1
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ecde0592a54835c0791d413df0b84cfe61c8375c6c53371a3a262f4a720ce7ee
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:44dfb10fef84f1136d3eee8160822b0d721bb87ad12f498bd11540cf57bcb837
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bff1f3af0c59bc6ebaa29224579d5dfcd2f5ed0cc9dfc92a0a059a14be64f7bc
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eed2e2028bc81a04ee6b31f6b531dfb1c72a67f736dcd67eadafbd77d40b8d1c
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6,
6
  "eval_steps": 100,
7
- "global_step": 600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -146,6 +146,52 @@
146
  "eval_samples_per_second": 78.385,
147
  "eval_steps_per_second": 4.929,
148
  "step": 600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
149
  }
150
  ],
151
  "logging_steps": 50,
@@ -165,7 +211,7 @@
165
  "attributes": {}
166
  }
167
  },
168
- "total_flos": 1.003509981904896e+17,
169
  "train_batch_size": 64,
170
  "trial_name": null,
171
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.8,
6
  "eval_steps": 100,
7
+ "global_step": 800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
146
  "eval_samples_per_second": 78.385,
147
  "eval_steps_per_second": 4.929,
148
  "step": 600
149
+ },
150
+ {
151
+ "epoch": 0.65,
152
+ "grad_norm": 2.2103075981140137,
153
+ "learning_rate": 3.649468851314669e-05,
154
+ "loss": 21.8233,
155
+ "step": 650
156
+ },
157
+ {
158
+ "epoch": 0.7,
159
+ "grad_norm": 1.6479830741882324,
160
+ "learning_rate": 2.7878659847348175e-05,
161
+ "loss": 21.7227,
162
+ "step": 700
163
+ },
164
+ {
165
+ "epoch": 0.7,
166
+ "eval_accuracy": 0.21617785614706886,
167
+ "eval_loss": 5.362000942230225,
168
+ "eval_runtime": 12.8392,
169
+ "eval_samples_per_second": 75.55,
170
+ "eval_steps_per_second": 4.751,
171
+ "step": 700
172
+ },
173
+ {
174
+ "epoch": 0.75,
175
+ "grad_norm": 1.3954864740371704,
176
+ "learning_rate": 2.0015946888825077e-05,
177
+ "loss": 21.6333,
178
+ "step": 750
179
+ },
180
+ {
181
+ "epoch": 0.8,
182
+ "grad_norm": 1.4790278673171997,
183
+ "learning_rate": 1.3174304897095113e-05,
184
+ "loss": 21.4768,
185
+ "step": 800
186
+ },
187
+ {
188
+ "epoch": 0.8,
189
+ "eval_accuracy": 0.21887124010660555,
190
+ "eval_loss": 5.3149943351745605,
191
+ "eval_runtime": 12.5786,
192
+ "eval_samples_per_second": 77.115,
193
+ "eval_steps_per_second": 4.85,
194
+ "step": 800
195
  }
196
  ],
197
  "logging_steps": 50,
 
211
  "attributes": {}
212
  }
213
  },
214
+ "total_flos": 1.338013309206528e+17,
215
  "train_batch_size": 64,
216
  "trial_name": null,
217
  "trial_params": null