rovdetection commited on
Commit
2c1b9ed
·
verified ·
1 Parent(s): 8ea7944

Training in progress, step 2000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f48d4e3b236d8f6efbf06705c5921fc52712f6cbe83333940faee84b86dc9b95
3
  size 4523108832
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9fa58f85c1881fbdc1256be3df4827baa99f4d296a0d35ff985043a310ce6161
3
  size 4523108832
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:890a01f7b0ed19d87aa41af9bd25032b9358906ea2c970f07da2b1cfdf970382
3
  size 2911851147
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b20c0e007837bb2be3064494728543e31c81ffcdd62114afa5da005a51dd81cb
3
  size 2911851147
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:01f9a0f7843a37be87edd23f4e88aa93b38b95cc2c07503eeb1cf2e4632453a2
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fdc37bbd2e979f041dfbbb004a5c74bab6cdda159cb18116df728588515a9ef6
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ca372268f4fa9335030c0cb7aedb6cdba75f457da50e7a4034abb1a2d0843689
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4aa03f6e0cd07cf67ce1fbe3101d545f5771ef9148b9debf02b11cf6948da5c
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d87adb2ba2eff80e48f285a6c3b50b12e8fa43fde329e8b0d491370e73a397d0
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14c891429705c32a3827fce3cd46c59925610cf516324635d98d8a689f74ce32
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": 1000,
3
  "best_metric": 0.988082230091095,
4
  "best_model_checkpoint": "./sft-out/checkpoint-1000",
5
- "epoch": 2.6506436819713795,
6
  "eval_steps": 500,
7
- "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -242,6 +242,84 @@
242
  "eval_samples_per_second": 18.279,
243
  "eval_steps_per_second": 2.303,
244
  "step": 1500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
245
  }
246
  ],
247
  "logging_steps": 50,
@@ -261,7 +339,7 @@
261
  "attributes": {}
262
  }
263
  },
264
- "total_flos": 5.822502544962355e+16,
265
  "train_batch_size": 1,
266
  "trial_name": null,
267
  "trial_params": null
 
2
  "best_global_step": 1000,
3
  "best_metric": 0.988082230091095,
4
  "best_model_checkpoint": "./sft-out/checkpoint-1000",
5
+ "epoch": 3.5339521520525996,
6
  "eval_steps": 500,
7
+ "global_step": 2000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
242
  "eval_samples_per_second": 18.279,
243
  "eval_steps_per_second": 2.303,
244
  "step": 1500
245
+ },
246
+ {
247
+ "epoch": 2.7390463561522735,
248
+ "grad_norm": 1.9865392446517944,
249
+ "learning_rate": 1.001083307599696e-05,
250
+ "loss": 0.34075843811035156,
251
+ "step": 1550
252
+ },
253
+ {
254
+ "epoch": 2.8274490303331676,
255
+ "grad_norm": 2.692328691482544,
256
+ "learning_rate": 9.469428420178699e-06,
257
+ "loss": 0.3529329299926758,
258
+ "step": 1600
259
+ },
260
+ {
261
+ "epoch": 2.9158517045140613,
262
+ "grad_norm": 2.0326497554779053,
263
+ "learning_rate": 8.92958002222035e-06,
264
+ "loss": 0.35016448974609377,
265
+ "step": 1650
266
+ },
267
+ {
268
+ "epoch": 3.003536106967236,
269
+ "grad_norm": 1.2595773935317993,
270
+ "learning_rate": 8.392871350487781e-06,
271
+ "loss": 0.3463246154785156,
272
+ "step": 1700
273
+ },
274
+ {
275
+ "epoch": 3.0919387811481296,
276
+ "grad_norm": 1.5270416736602783,
277
+ "learning_rate": 7.860876663988886e-06,
278
+ "loss": 0.14206992149353026,
279
+ "step": 1750
280
+ },
281
+ {
282
+ "epoch": 3.1803414553290237,
283
+ "grad_norm": 1.808855652809143,
284
+ "learning_rate": 7.335156394800651e-06,
285
+ "loss": 0.14106608390808106,
286
+ "step": 1800
287
+ },
288
+ {
289
+ "epoch": 3.2687441295099178,
290
+ "grad_norm": 1.834566593170166,
291
+ "learning_rate": 6.817252571053034e-06,
292
+ "loss": 0.138691987991333,
293
+ "step": 1850
294
+ },
295
+ {
296
+ "epoch": 3.357146803690812,
297
+ "grad_norm": 1.779943823814392,
298
+ "learning_rate": 6.308684293894747e-06,
299
+ "loss": 0.13664389610290528,
300
+ "step": 1900
301
+ },
302
+ {
303
+ "epoch": 3.4455494778717055,
304
+ "grad_norm": 1.5618979930877686,
305
+ "learning_rate": 5.810943281707862e-06,
306
+ "loss": 0.13606584548950196,
307
+ "step": 1950
308
+ },
309
+ {
310
+ "epoch": 3.5339521520525996,
311
+ "grad_norm": 1.8645941019058228,
312
+ "learning_rate": 5.325489494640788e-06,
313
+ "loss": 0.13294898033142089,
314
+ "step": 2000
315
+ },
316
+ {
317
+ "epoch": 3.5339521520525996,
318
+ "eval_loss": 1.1485551595687866,
319
+ "eval_runtime": 27.413,
320
+ "eval_samples_per_second": 18.24,
321
+ "eval_steps_per_second": 2.298,
322
+ "step": 2000
323
  }
324
  ],
325
  "logging_steps": 50,
 
339
  "attributes": {}
340
  }
341
  },
342
+ "total_flos": 7.762188828042854e+16,
343
  "train_batch_size": 1,
344
  "trial_name": null,
345
  "trial_params": null