rovdetection commited on
Commit
84efa46
·
verified ·
1 Parent(s): fee6211

Training in progress, step 1500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6b3bd15ad5f6e246b6e337225788c69c49a21df6eeb262d2b1b1897cf5cf6146
3
  size 4523108832
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f48d4e3b236d8f6efbf06705c5921fc52712f6cbe83333940faee84b86dc9b95
3
  size 4523108832
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9ecf8f056195b0098a8274b3035e24672255ff4c10ed92a0717c3ca84f2c7426
3
  size 2911851147
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:890a01f7b0ed19d87aa41af9bd25032b9358906ea2c970f07da2b1cfdf970382
3
  size 2911851147
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a8e2011629d8bed3ef560fa11175cac55684c4e12a72634bb24abf767b6c7399
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01f9a0f7843a37be87edd23f4e88aa93b38b95cc2c07503eeb1cf2e4632453a2
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:14ae2a2128444abab378aa06c09a61a84665f758fcc19fc46f5789b0bc1b5665
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca372268f4fa9335030c0cb7aedb6cdba75f457da50e7a4034abb1a2d0843689
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:92235ff997c3a5622fba1c344122ccffab641b57d7443401a19efb448a747750
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d87adb2ba2eff80e48f285a6c3b50b12e8fa43fde329e8b0d491370e73a397d0
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": 1000,
3
  "best_metric": 0.988082230091095,
4
  "best_model_checkpoint": "./sft-out/checkpoint-1000",
5
- "epoch": 1.7673352118901597,
6
  "eval_steps": 500,
7
- "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -164,6 +164,84 @@
164
  "eval_samples_per_second": 18.28,
165
  "eval_steps_per_second": 2.303,
166
  "step": 1000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
167
  }
168
  ],
169
  "logging_steps": 50,
@@ -183,7 +261,7 @@
183
  "attributes": {}
184
  }
185
  },
186
- "total_flos": 3.876769443016704e+16,
187
  "train_batch_size": 1,
188
  "trial_name": null,
189
  "trial_params": null
 
2
  "best_global_step": 1000,
3
  "best_metric": 0.988082230091095,
4
  "best_model_checkpoint": "./sft-out/checkpoint-1000",
5
+ "epoch": 2.6506436819713795,
6
  "eval_steps": 500,
7
+ "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
164
  "eval_samples_per_second": 18.28,
165
  "eval_steps_per_second": 2.303,
166
  "step": 1000
167
+ },
168
+ {
169
+ "epoch": 1.8557378860710536,
170
+ "grad_norm": 2.1198668479919434,
171
+ "learning_rate": 1.5164817945522343e-05,
172
+ "loss": 0.7200093078613281,
173
+ "step": 1050
174
+ },
175
+ {
176
+ "epoch": 1.9441405602519475,
177
+ "grad_norm": 2.416775703430176,
178
+ "learning_rate": 1.4693652811602635e-05,
179
+ "loss": 0.7036862182617187,
180
+ "step": 1100
181
+ },
182
+ {
183
+ "epoch": 2.0318249627051217,
184
+ "grad_norm": 2.5851995944976807,
185
+ "learning_rate": 1.4208720385280673e-05,
186
+ "loss": 0.5860164260864258,
187
+ "step": 1150
188
+ },
189
+ {
190
+ "epoch": 2.120227636886016,
191
+ "grad_norm": 1.8049288988113403,
192
+ "learning_rate": 1.3711443056915611e-05,
193
+ "loss": 0.354575080871582,
194
+ "step": 1200
195
+ },
196
+ {
197
+ "epoch": 2.20863031106691,
198
+ "grad_norm": 2.2025763988494873,
199
+ "learning_rate": 1.3203279426591329e-05,
200
+ "loss": 0.3392633819580078,
201
+ "step": 1250
202
+ },
203
+ {
204
+ "epoch": 2.297032985247804,
205
+ "grad_norm": 2.0318801403045654,
206
+ "learning_rate": 1.2685720025791004e-05,
207
+ "loss": 0.33905509948730467,
208
+ "step": 1300
209
+ },
210
+ {
211
+ "epoch": 2.3854356594286976,
212
+ "grad_norm": 2.0977327823638916,
213
+ "learning_rate": 1.2160282945411513e-05,
214
+ "loss": 0.34515163421630857,
215
+ "step": 1350
216
+ },
217
+ {
218
+ "epoch": 2.4738383336095917,
219
+ "grad_norm": 2.108346462249756,
220
+ "learning_rate": 1.1628509382941232e-05,
221
+ "loss": 0.35274150848388675,
222
+ "step": 1400
223
+ },
224
+ {
225
+ "epoch": 2.562241007790486,
226
+ "grad_norm": 2.778698682785034,
227
+ "learning_rate": 1.1091959121862279e-05,
228
+ "loss": 0.3534046173095703,
229
+ "step": 1450
230
+ },
231
+ {
232
+ "epoch": 2.6506436819713795,
233
+ "grad_norm": 2.0644469261169434,
234
+ "learning_rate": 1.0552205956536803e-05,
235
+ "loss": 0.34923103332519534,
236
+ "step": 1500
237
+ },
238
+ {
239
+ "epoch": 2.6506436819713795,
240
+ "eval_loss": 1.0509159564971924,
241
+ "eval_runtime": 27.3535,
242
+ "eval_samples_per_second": 18.279,
243
+ "eval_steps_per_second": 2.303,
244
+ "step": 1500
245
  }
246
  ],
247
  "logging_steps": 50,
 
261
  "attributes": {}
262
  }
263
  },
264
+ "total_flos": 5.822502544962355e+16,
265
  "train_batch_size": 1,
266
  "trial_name": null,
267
  "trial_params": null