CodeIsAbstract commited on
Commit
73c4007
·
verified ·
1 Parent(s): 02c302c

Training in progress, step 1200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dddca729ab09142c6d4a86292fcb09c8240e120212ff2dea7784702f1c454342
3
  size 579748776
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b7da8882c4eba065851f3d33946d222f14c2f723a8e4b466de1f227567d55531
3
  size 579748776
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b9ae4821caa875af2772c8c1c8a8f25674e533cf460c88cc3b72e2ecf795355b
3
  size 1159627083
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3dc6db80c72368f3c3b7f1bca017a3f8af377e6e79bfb95dfa6de9e3819b27d4
3
  size 1159627083
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5387f6e99d51202c7ef90fef723e457487d08949715af1e6a0aff706c078f5bd
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4d8242b37901ec7cc0992b7a6173ea32c89632c26198ca917c21035a4c247c4
3
  size 14917
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a6418bc04f5e424107d7c94f43f8158a47397e1e0f7c2b0b675aeed07c6fcfe3
3
  size 14917
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6bbac1cf1ce1610978037481a5035e6b523662562dd00a236dbb2507431643e9
3
  size 14917
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4fdb738e61f6acc5e69da32f32a665c3f109e78a2b1b3b8004b382d7b5d9fa18
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d78ed09f8ef41008377f42bff0e363b7ba82ecd1c922f2e831caad0254e1ac5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.45,
6
  "eval_steps": 100,
7
- "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -341,6 +341,117 @@
341
  "eval_samples_per_second": 64.921,
342
  "eval_steps_per_second": 2.142,
343
  "step": 900
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
344
  }
345
  ],
346
  "logging_steps": 25,
@@ -360,7 +471,7 @@
360
  "attributes": {}
361
  }
362
  },
363
- "total_flos": 6.020016214376448e+17,
364
  "train_batch_size": 64,
365
  "trial_name": null,
366
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6,
6
  "eval_steps": 100,
7
+ "global_step": 1200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
341
  "eval_samples_per_second": 64.921,
342
  "eval_steps_per_second": 2.142,
343
  "step": 900
344
+ },
345
+ {
346
+ "epoch": 0.4625,
347
+ "grad_norm": 13.373994827270508,
348
+ "learning_rate": 6.268105304531354e-05,
349
+ "loss": 92.7674,
350
+ "step": 925
351
+ },
352
+ {
353
+ "epoch": 0.475,
354
+ "grad_norm": 20.701358795166016,
355
+ "learning_rate": 6.061695044309973e-05,
356
+ "loss": 92.4968,
357
+ "step": 950
358
+ },
359
+ {
360
+ "epoch": 0.4875,
361
+ "grad_norm": 13.155655860900879,
362
+ "learning_rate": 5.8533715378581e-05,
363
+ "loss": 92.2234,
364
+ "step": 975
365
+ },
366
+ {
367
+ "epoch": 0.5,
368
+ "grad_norm": 46.88108444213867,
369
+ "learning_rate": 5.643510198215082e-05,
370
+ "loss": 92.0125,
371
+ "step": 1000
372
+ },
373
+ {
374
+ "epoch": 0.5,
375
+ "eval_accuracy": 0.17633533442055147,
376
+ "eval_loss": 5.7336883544921875,
377
+ "eval_runtime": 7.3739,
378
+ "eval_samples_per_second": 65.773,
379
+ "eval_steps_per_second": 2.17,
380
+ "step": 1000
381
+ },
382
+ {
383
+ "epoch": 0.5125,
384
+ "grad_norm": 482.2004699707031,
385
+ "learning_rate": 5.432489209699614e-05,
386
+ "loss": 93.2148,
387
+ "step": 1025
388
+ },
389
+ {
390
+ "epoch": 0.525,
391
+ "grad_norm": 297.8095703125,
392
+ "learning_rate": 5.220688846396047e-05,
393
+ "loss": 95.0004,
394
+ "step": 1050
395
+ },
396
+ {
397
+ "epoch": 0.5375,
398
+ "grad_norm": 332.8581848144531,
399
+ "learning_rate": 5.0084907868747746e-05,
400
+ "loss": 95.3143,
401
+ "step": 1075
402
+ },
403
+ {
404
+ "epoch": 0.55,
405
+ "grad_norm": 9733.95703125,
406
+ "learning_rate": 4.796277426381657e-05,
407
+ "loss": 95.8613,
408
+ "step": 1100
409
+ },
410
+ {
411
+ "epoch": 0.55,
412
+ "eval_accuracy": 0.1554284329416992,
413
+ "eval_loss": 6.006833553314209,
414
+ "eval_runtime": 7.2985,
415
+ "eval_samples_per_second": 66.452,
416
+ "eval_steps_per_second": 2.192,
417
+ "step": 1100
418
+ },
419
+ {
420
+ "epoch": 0.5625,
421
+ "grad_norm": 1343.75537109375,
422
+ "learning_rate": 4.5844311877359384e-05,
423
+ "loss": 96.7118,
424
+ "step": 1125
425
+ },
426
+ {
427
+ "epoch": 0.575,
428
+ "grad_norm": 234.01663208007812,
429
+ "learning_rate": 4.373333832178478e-05,
430
+ "loss": 97.3733,
431
+ "step": 1150
432
+ },
433
+ {
434
+ "epoch": 0.5875,
435
+ "grad_norm": 868.69921875,
436
+ "learning_rate": 4.1633657714122e-05,
437
+ "loss": 97.8172,
438
+ "step": 1175
439
+ },
440
+ {
441
+ "epoch": 0.6,
442
+ "grad_norm": 370.631591796875,
443
+ "learning_rate": 3.954905382074478e-05,
444
+ "loss": 98.1012,
445
+ "step": 1200
446
+ },
447
+ {
448
+ "epoch": 0.6,
449
+ "eval_accuracy": 0.14907376084903992,
450
+ "eval_loss": 6.105465888977051,
451
+ "eval_runtime": 7.4782,
452
+ "eval_samples_per_second": 64.855,
453
+ "eval_steps_per_second": 2.14,
454
+ "step": 1200
455
  }
456
  ],
457
  "logging_steps": 25,
 
471
  "attributes": {}
472
  }
473
  },
474
+ "total_flos": 8.026688285835264e+17,
475
  "train_batch_size": 64,
476
  "trial_name": null,
477
  "trial_params": null