CodeIsAbstract commited on
Commit
f7e18f6
·
verified ·
1 Parent(s): 2fca95c

Training in progress, step 8000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9bb532cf97353e876d3c07dafda9641b1330c0ef72b36326c6da98516454dc96
3
  size 734275920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1697d53984b439e7327a0da7030659bcc191c89124216fe8240d43fb004110ba
3
  size 734275920
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3471614dd25efffa34ced8adea663f06082eaa9ce06ff0ae502a2a2ce49dce6a
3
  size 1468687243
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e24effe110ebba34f42d63b80eaccb2a97fbe0e91f4e1d0d25d0823af27f34a
3
  size 1468687243
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15c2f1e87a9fcaf57351b335abc031c00c3adda5167bcf7ec0b18e183f3ee6ae
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:48e42e78ee2689141db2aa36168809ae145b2ea351a4978872644c5587b84ff8
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:da416040f48730a4d4754962dfc1cf7e84c988b051ed7995fad8903f5b75ce23
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8f78aa7de166ae1b78898a061c80c7d73724dc9fcfb5269ef4e7807084cda31d
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:96b41d044a0b95f3d68b68c393e43b00c5ca90a1a9bc2f1272a5782f85171f5a
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b4fa473a13a8055ecd91a792ceedace9b305b095bc0470e8680559c4564e704
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7909009ee6bb6a7388a27f2195f9d99a87ca4bfa7948fb59ebcf32152929a165
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:db461a9dee610efe78caa588c54fcc74909ce5d166ca7a4133f3f5f5280e91e7
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8793849385d70eb0a4454019ee1dea79e0782bfac3827e1750c91bff2a4e5501
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ea3c4dae1a9a132985c2fe8a16f00f00e6a75b0a89bc641b65e6029f630cc7a8
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:aa6203f4502122aa111ee64a6591a834d190fa04fcbd197aea65cf4f041f3e16
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dc262f4ad8cdaf2d5605f4ed5f2d19a6c8cdcfe8bb557c95eb12f35788aa1a91
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b20e94508e1cbf7f79ffc938a6389a03f2f263a9d7c056df1a5c8567f12db613
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d6e02967c943c7285054ff2d6e85583a0d50e230c4d165759e0c7efb0cb6cc2
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f8c8b444dac54dc5ad78a7292762380772bc71cffe63d34cd4898fae9fc500a4
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c48aca1aa4552b505c8ba0582648eccb7e876162d1d989a4ff2948e15d0afefb
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d36ffe8038dcfcbb1af52f768d5ded0e7902497baa765ea8664469f5c3a22aec
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:091bf8841b544ca817b2c69c4d32e29435f0e7a64d1c3b7d04899e50753a7dbf
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.12,
6
  "eval_steps": 1000,
7
- "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -476,6 +476,162 @@
476
  "eval_samples_per_second": 10.807,
477
  "eval_steps_per_second": 0.54,
478
  "step": 6000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
479
  }
480
  ],
481
  "logging_steps": 100,
@@ -495,7 +651,7 @@
495
  "attributes": {}
496
  }
497
  },
498
- "total_flos": 4.274704613376e+16,
499
  "train_batch_size": 4,
500
  "trial_name": null,
501
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.16,
6
  "eval_steps": 1000,
7
+ "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
476
  "eval_samples_per_second": 10.807,
477
  "eval_steps_per_second": 0.54,
478
  "step": 6000
479
+ },
480
+ {
481
+ "epoch": 0.122,
482
+ "grad_norm": 462631.5,
483
+ "learning_rate": 0.000482635014919091,
484
+ "loss": 59.005234375,
485
+ "step": 6100
486
+ },
487
+ {
488
+ "epoch": 0.124,
489
+ "grad_norm": 273326.59375,
490
+ "learning_rate": 0.00048205345814171074,
491
+ "loss": 58.9213232421875,
492
+ "step": 6200
493
+ },
494
+ {
495
+ "epoch": 0.126,
496
+ "grad_norm": 36951.60546875,
497
+ "learning_rate": 0.00048146268507654377,
498
+ "loss": 58.82025390625,
499
+ "step": 6300
500
+ },
501
+ {
502
+ "epoch": 0.128,
503
+ "grad_norm": 103255.4765625,
504
+ "learning_rate": 0.00048086271918686716,
505
+ "loss": 58.84619140625,
506
+ "step": 6400
507
+ },
508
+ {
509
+ "epoch": 0.13,
510
+ "grad_norm": 102904.96875,
511
+ "learning_rate": 0.00048025358430106227,
512
+ "loss": 58.514482421875,
513
+ "step": 6500
514
+ },
515
+ {
516
+ "epoch": 0.132,
517
+ "grad_norm": 368401.625,
518
+ "learning_rate": 0.00047963530461166826,
519
+ "loss": 58.3533984375,
520
+ "step": 6600
521
+ },
522
+ {
523
+ "epoch": 0.134,
524
+ "grad_norm": 307524.84375,
525
+ "learning_rate": 0.0004790079046744218,
526
+ "loss": 58.44912109375,
527
+ "step": 6700
528
+ },
529
+ {
530
+ "epoch": 0.136,
531
+ "grad_norm": 218012.390625,
532
+ "learning_rate": 0.000478371409407281,
533
+ "loss": 58.45431640625,
534
+ "step": 6800
535
+ },
536
+ {
537
+ "epoch": 0.138,
538
+ "grad_norm": 1243047.5,
539
+ "learning_rate": 0.0004777258440894362,
540
+ "loss": 58.24828125,
541
+ "step": 6900
542
+ },
543
+ {
544
+ "epoch": 0.14,
545
+ "grad_norm": 158677.6875,
546
+ "learning_rate": 0.0004770712343603062,
547
+ "loss": 58.133662109375,
548
+ "step": 7000
549
+ },
550
+ {
551
+ "epoch": 0.14,
552
+ "eval_loss": 58.2567138671875,
553
+ "eval_runtime": 1.8382,
554
+ "eval_samples_per_second": 10.88,
555
+ "eval_steps_per_second": 0.544,
556
+ "step": 7000
557
+ },
558
+ {
559
+ "epoch": 0.142,
560
+ "grad_norm": 237161.1875,
561
+ "learning_rate": 0.0004764076062185194,
562
+ "loss": 58.3987744140625,
563
+ "step": 7100
564
+ },
565
+ {
566
+ "epoch": 0.144,
567
+ "grad_norm": 60140.4921875,
568
+ "learning_rate": 0.00047573498602088154,
569
+ "loss": 58.801220703125,
570
+ "step": 7200
571
+ },
572
+ {
573
+ "epoch": 0.146,
574
+ "grad_norm": 494529.25,
575
+ "learning_rate": 0.00047505340048132916,
576
+ "loss": 58.503544921875,
577
+ "step": 7300
578
+ },
579
+ {
580
+ "epoch": 0.148,
581
+ "grad_norm": 561465.625,
582
+ "learning_rate": 0.00047436287666986803,
583
+ "loss": 58.614169921875,
584
+ "step": 7400
585
+ },
586
+ {
587
+ "epoch": 0.15,
588
+ "grad_norm": 572518.4375,
589
+ "learning_rate": 0.00047366344201149856,
590
+ "loss": 58.28416015625,
591
+ "step": 7500
592
+ },
593
+ {
594
+ "epoch": 0.152,
595
+ "grad_norm": 1078630.125,
596
+ "learning_rate": 0.0004729551242851264,
597
+ "loss": 58.4676513671875,
598
+ "step": 7600
599
+ },
600
+ {
601
+ "epoch": 0.154,
602
+ "grad_norm": 70299.53125,
603
+ "learning_rate": 0.00047223795162245886,
604
+ "loss": 58.200947265625,
605
+ "step": 7700
606
+ },
607
+ {
608
+ "epoch": 0.156,
609
+ "grad_norm": 68182.265625,
610
+ "learning_rate": 0.0004715119525068883,
611
+ "loss": 58.55353515625,
612
+ "step": 7800
613
+ },
614
+ {
615
+ "epoch": 0.158,
616
+ "grad_norm": 82698.640625,
617
+ "learning_rate": 0.00047077715577236015,
618
+ "loss": 58.704990234375,
619
+ "step": 7900
620
+ },
621
+ {
622
+ "epoch": 0.16,
623
+ "grad_norm": 375052.03125,
624
+ "learning_rate": 0.0004700335906022283,
625
+ "loss": 58.521640625,
626
+ "step": 8000
627
+ },
628
+ {
629
+ "epoch": 0.16,
630
+ "eval_loss": 57.7498779296875,
631
+ "eval_runtime": 1.8363,
632
+ "eval_samples_per_second": 10.892,
633
+ "eval_steps_per_second": 0.545,
634
+ "step": 8000
635
  }
636
  ],
637
  "logging_steps": 100,
 
651
  "attributes": {}
652
  }
653
  },
654
+ "total_flos": 5.699606151168e+16,
655
  "train_batch_size": 4,
656
  "trial_name": null,
657
  "trial_params": null