CodeIsAbstract commited on
Commit
124e5ea
·
verified ·
1 Parent(s): 911628b

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:615b2afd3bc142e08bf251a928f84b4e1cea9ae26762fc89ba0a4736dba7eb66
3
  size 529337896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5957954f25406e9df797421489d9b120ac7feff52692d5e5c70038c879f86412
3
  size 529337896
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8dfeca88de39ac397647f995ea8cde7b1fb6e6871fa362fb780526f82a902479
3
  size 871247243
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ee0d0e9122e67a2d2d94315cf19cbd6a916fcb06556d38abe0ccc77fd7feae8
3
  size 871247243
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9bb1461e40c7e08171c25a0bc2bf64964397ea8cc712bb88016a59954d83db2d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dbcbab03d5bb70bb2a3066017b02040dc9b35585ddf112a4be38eb15410f67d9
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eb2a89a85e1c8ac7ee9b5678d857a4d99090faa37a24489f44839e0801dfe16a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9567d76a2efaa946d19913fd9fe6f6f70f6f7101d7dcb8866c09743534f2d4d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -482,6 +482,164 @@
482
  "eval_samples_per_second": 83.4,
483
  "eval_steps_per_second": 0.667,
484
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
485
  }
486
  ],
487
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
482
  "eval_samples_per_second": 83.4,
483
  "eval_steps_per_second": 0.667,
484
  "step": 6000
485
+ },
486
+ {
487
+ "epoch": 0.122,
488
+ "grad_norm": 0.14931881427764893,
489
+ "learning_rate": 0.0019718035350990712,
490
+ "loss": 3.7169,
491
+ "step": 6100
492
+ },
493
+ {
494
+ "epoch": 0.124,
495
+ "grad_norm": 0.14251381158828735,
496
+ "learning_rate": 0.0019702227914230566,
497
+ "loss": 3.6929,
498
+ "step": 6200
499
+ },
500
+ {
501
+ "epoch": 0.126,
502
+ "grad_norm": 0.14251849055290222,
503
+ "learning_rate": 0.001968599607059059,
504
+ "loss": 3.7018,
505
+ "step": 6300
506
+ },
507
+ {
508
+ "epoch": 0.128,
509
+ "grad_norm": 0.15696577727794647,
510
+ "learning_rate": 0.0019669340530104207,
511
+ "loss": 3.7034,
512
+ "step": 6400
513
+ },
514
+ {
515
+ "epoch": 0.13,
516
+ "grad_norm": 0.14737001061439514,
517
+ "learning_rate": 0.001965226202133872,
518
+ "loss": 3.6889,
519
+ "step": 6500
520
+ },
521
+ {
522
+ "epoch": 0.132,
523
+ "grad_norm": 0.13667502999305725,
524
+ "learning_rate": 0.0019634761291363427,
525
+ "loss": 3.6935,
526
+ "step": 6600
527
+ },
528
+ {
529
+ "epoch": 0.134,
530
+ "grad_norm": 0.13390734791755676,
531
+ "learning_rate": 0.0019616839105716954,
532
+ "loss": 3.7032,
533
+ "step": 6700
534
+ },
535
+ {
536
+ "epoch": 0.136,
537
+ "grad_norm": 0.14834214746952057,
538
+ "learning_rate": 0.0019598496248373755,
539
+ "loss": 3.6657,
540
+ "step": 6800
541
+ },
542
+ {
543
+ "epoch": 0.138,
544
+ "grad_norm": 0.1314767599105835,
545
+ "learning_rate": 0.001957973352170984,
546
+ "loss": 3.673,
547
+ "step": 6900
548
+ },
549
+ {
550
+ "epoch": 0.14,
551
+ "grad_norm": 0.15161661803722382,
552
+ "learning_rate": 0.001956055174646765,
553
+ "loss": 3.6713,
554
+ "step": 7000
555
+ },
556
+ {
557
+ "epoch": 0.14,
558
+ "eval_accuracy": 0.3484422700587084,
559
+ "eval_loss": 3.6349546909332275,
560
+ "eval_runtime": 12.1265,
561
+ "eval_samples_per_second": 82.464,
562
+ "eval_steps_per_second": 0.66,
563
+ "step": 7000
564
+ },
565
+ {
566
+ "epoch": 0.142,
567
+ "grad_norm": 0.12710200250148773,
568
+ "learning_rate": 0.0019540951761720174,
569
+ "loss": 3.67,
570
+ "step": 7100
571
+ },
572
+ {
573
+ "epoch": 0.144,
574
+ "grad_norm": 0.18498213589191437,
575
+ "learning_rate": 0.0019520934424834247,
576
+ "loss": 3.675,
577
+ "step": 7200
578
+ },
579
+ {
580
+ "epoch": 0.146,
581
+ "grad_norm": 0.13697299361228943,
582
+ "learning_rate": 0.0019500500611433025,
583
+ "loss": 3.6497,
584
+ "step": 7300
585
+ },
586
+ {
587
+ "epoch": 0.148,
588
+ "grad_norm": 0.13290269672870636,
589
+ "learning_rate": 0.0019479651215357707,
590
+ "loss": 3.662,
591
+ "step": 7400
592
+ },
593
+ {
594
+ "epoch": 0.15,
595
+ "grad_norm": 0.12744054198265076,
596
+ "learning_rate": 0.0019458387148628417,
597
+ "loss": 3.6645,
598
+ "step": 7500
599
+ },
600
+ {
601
+ "epoch": 0.152,
602
+ "grad_norm": 0.13697493076324463,
603
+ "learning_rate": 0.001943670934140432,
604
+ "loss": 3.6374,
605
+ "step": 7600
606
+ },
607
+ {
608
+ "epoch": 0.154,
609
+ "grad_norm": 0.14067181944847107,
610
+ "learning_rate": 0.0019414618741942936,
611
+ "loss": 3.6453,
612
+ "step": 7700
613
+ },
614
+ {
615
+ "epoch": 0.156,
616
+ "grad_norm": 0.1316055953502655,
617
+ "learning_rate": 0.0019392116316558638,
618
+ "loss": 3.6679,
619
+ "step": 7800
620
+ },
621
+ {
622
+ "epoch": 0.158,
623
+ "grad_norm": 0.1261526495218277,
624
+ "learning_rate": 0.001936920304958042,
625
+ "loss": 3.6343,
626
+ "step": 7900
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "grad_norm": 0.13644501566886902,
631
+ "learning_rate": 0.0019345879943308804,
632
+ "loss": 3.6365,
633
+ "step": 8000
634
+ },
635
+ {
636
+ "epoch": 0.16,
637
+ "eval_accuracy": 0.35229158512720155,
638
+ "eval_loss": 3.5980424880981445,
639
+ "eval_runtime": 17.6098,
640
+ "eval_samples_per_second": 56.787,
641
+ "eval_steps_per_second": 0.454,
642
+ "step": 8000
643
  }
644
  ],
645
  "logging_steps": 100,