CodeIsAbstract commited on
Commit
084887d
·
verified ·
1 Parent(s): 5cd854a

Training in progress, step 22000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0ba3c2b02c2181522825a50de924743eb11a5c8ec14f96218442ef3bf9c669ad
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e851d31aa87e3f44cdb6b12860dfca56948ed93c4c3e52e3a3bc7173fa578a8b
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:683d6ef85df4768d6187df8f33dd7d019dd22439aeb22f78e278b0cdd0d17a92
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df630ecbd687f1359a191cc034b3009e41a878d4ab7fdb04cf787f0bc23ecb0c
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5451e921407cd3b87f821062b7ea647b65f6d0ce5a09f45a2e5fe63f79c5afdc
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:43b610cc94e504b24cb2bb3bb992068bab81d217de2c8ad61ad01dfb75e98503
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fdc6d60c7c4fe4f9e140eb1a01d69a125cb28ae84f348f2ce7bfdb028834e194
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f23bcbe880608cbeb7ee4f55362f4ac1ed7848d0f02874b239279ca55eb68fab
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 1000,
7
- "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1588,6 +1588,164 @@
1588
  "eval_samples_per_second": 289.838,
1589
  "eval_steps_per_second": 18.218,
1590
  "step": 20000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1591
  }
1592
  ],
1593
  "logging_steps": 100,
@@ -1607,7 +1765,7 @@
1607
  "attributes": {}
1608
  }
1609
  },
1610
- "total_flos": 7.839921733632e+17,
1611
  "train_batch_size": 120,
1612
  "trial_name": null,
1613
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.44,
6
  "eval_steps": 1000,
7
+ "global_step": 22000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1588
  "eval_samples_per_second": 289.838,
1589
  "eval_steps_per_second": 18.218,
1590
  "step": 20000
1591
+ },
1592
+ {
1593
+ "epoch": 0.402,
1594
+ "grad_norm": 0.1547059416770935,
1595
+ "learning_rate": 0.0006542387583121131,
1596
+ "loss": 3.441893310546875,
1597
+ "step": 20100
1598
+ },
1599
+ {
1600
+ "epoch": 0.404,
1601
+ "grad_norm": 0.15452757477760315,
1602
+ "learning_rate": 0.000651238340825213,
1603
+ "loss": 3.434819641113281,
1604
+ "step": 20200
1605
+ },
1606
+ {
1607
+ "epoch": 0.406,
1608
+ "grad_norm": 0.13765129446983337,
1609
+ "learning_rate": 0.0006482319167220591,
1610
+ "loss": 3.4564175415039062,
1611
+ "step": 20300
1612
+ },
1613
+ {
1614
+ "epoch": 0.408,
1615
+ "grad_norm": 0.15758726000785828,
1616
+ "learning_rate": 0.0006452196054064732,
1617
+ "loss": 3.4555868530273437,
1618
+ "step": 20400
1619
+ },
1620
+ {
1621
+ "epoch": 0.41,
1622
+ "grad_norm": 0.3239542543888092,
1623
+ "learning_rate": 0.0006422015265160946,
1624
+ "loss": 3.4481591796875,
1625
+ "step": 20500
1626
+ },
1627
+ {
1628
+ "epoch": 0.412,
1629
+ "grad_norm": 0.15911568701267242,
1630
+ "learning_rate": 0.0006391777999176294,
1631
+ "loss": 3.44850341796875,
1632
+ "step": 20600
1633
+ },
1634
+ {
1635
+ "epoch": 0.414,
1636
+ "grad_norm": 0.12856774032115936,
1637
+ "learning_rate": 0.000636148545702089,
1638
+ "loss": 3.438498840332031,
1639
+ "step": 20700
1640
+ },
1641
+ {
1642
+ "epoch": 0.416,
1643
+ "grad_norm": 0.17129512131214142,
1644
+ "learning_rate": 0.000633113884180021,
1645
+ "loss": 3.438189697265625,
1646
+ "step": 20800
1647
+ },
1648
+ {
1649
+ "epoch": 0.418,
1650
+ "grad_norm": 0.14792455732822418,
1651
+ "learning_rate": 0.0006300739358767311,
1652
+ "loss": 3.425160217285156,
1653
+ "step": 20900
1654
+ },
1655
+ {
1656
+ "epoch": 0.42,
1657
+ "grad_norm": 0.13070790469646454,
1658
+ "learning_rate": 0.0006270288215274957,
1659
+ "loss": 3.4182763671875,
1660
+ "step": 21000
1661
+ },
1662
+ {
1663
+ "epoch": 0.42,
1664
+ "eval_accuracy": 0.3719621193102593,
1665
+ "eval_loss": 3.395620584487915,
1666
+ "eval_runtime": 6.7127,
1667
+ "eval_samples_per_second": 289.155,
1668
+ "eval_steps_per_second": 18.175,
1669
+ "step": 21000
1670
+ },
1671
+ {
1672
+ "epoch": 0.422,
1673
+ "grad_norm": 0.1333521008491516,
1674
+ "learning_rate": 0.0006239786620727669,
1675
+ "loss": 3.440789794921875,
1676
+ "step": 21100
1677
+ },
1678
+ {
1679
+ "epoch": 0.424,
1680
+ "grad_norm": 0.14673130214214325,
1681
+ "learning_rate": 0.0006209235786533696,
1682
+ "loss": 3.41100341796875,
1683
+ "step": 21200
1684
+ },
1685
+ {
1686
+ "epoch": 0.426,
1687
+ "grad_norm": 0.13843247294425964,
1688
+ "learning_rate": 0.00061786369260569,
1689
+ "loss": 3.4241146850585937,
1690
+ "step": 21300
1691
+ },
1692
+ {
1693
+ "epoch": 0.428,
1694
+ "grad_norm": 0.15212669968605042,
1695
+ "learning_rate": 0.0006147991254568566,
1696
+ "loss": 3.4158154296875,
1697
+ "step": 21400
1698
+ },
1699
+ {
1700
+ "epoch": 0.43,
1701
+ "grad_norm": 0.1734033226966858,
1702
+ "learning_rate": 0.0006117299989199135,
1703
+ "loss": 3.4223089599609375,
1704
+ "step": 21500
1705
+ },
1706
+ {
1707
+ "epoch": 0.432,
1708
+ "grad_norm": 0.15912015736103058,
1709
+ "learning_rate": 0.0006086564348889863,
1710
+ "loss": 3.405228271484375,
1711
+ "step": 21600
1712
+ },
1713
+ {
1714
+ "epoch": 0.434,
1715
+ "grad_norm": 0.1569402515888214,
1716
+ "learning_rate": 0.0006055785554344417,
1717
+ "loss": 3.4033087158203124,
1718
+ "step": 21700
1719
+ },
1720
+ {
1721
+ "epoch": 0.436,
1722
+ "grad_norm": 0.15008965134620667,
1723
+ "learning_rate": 0.0006024964827980378,
1724
+ "loss": 3.4061663818359373,
1725
+ "step": 21800
1726
+ },
1727
+ {
1728
+ "epoch": 0.438,
1729
+ "grad_norm": 0.12902535498142242,
1730
+ "learning_rate": 0.0005994103393880712,
1731
+ "loss": 3.3769439697265624,
1732
+ "step": 21900
1733
+ },
1734
+ {
1735
+ "epoch": 0.44,
1736
+ "grad_norm": 0.14630058407783508,
1737
+ "learning_rate": 0.0005963202477745134,
1738
+ "loss": 3.3858111572265623,
1739
+ "step": 22000
1740
+ },
1741
+ {
1742
+ "epoch": 0.44,
1743
+ "eval_accuracy": 0.372494457332805,
1744
+ "eval_loss": 3.3869516849517822,
1745
+ "eval_runtime": 6.5171,
1746
+ "eval_samples_per_second": 297.832,
1747
+ "eval_steps_per_second": 18.72,
1748
+ "step": 22000
1749
  }
1750
  ],
1751
  "logging_steps": 100,
 
1765
  "attributes": {}
1766
  }
1767
  },
1768
+ "total_flos": 8.6239139069952e+17,
1769
  "train_batch_size": 120,
1770
  "trial_name": null,
1771
  "trial_params": null