CodeIsAbstract commited on
Commit
240cea1
·
verified ·
1 Parent(s): 8ad6a71

Training in progress, step 22000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:73cb875f1568155797c9b13f85384966dd718f46953534c4b08f677ecf9a6a82
3
  size 386379
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da60d3804c57e29e90c568ae5208c61903451533ab428233f16399bcbfdc9166
3
  size 386379
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:82e61a71536ab895c8bd16825f7157395f1f1e328b8bc9b70f310b0e401ae1c9
3
  size 1540661735
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae80c44868efaca16e3497e99d1fc33821e44d73899ed13a47e498dc9e7e5f4f
3
  size 1540661735
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8d96b0f1732638e94f0dd38858b482c7b94b7f3411e8a578ab9a90fdbb757ebb
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:61863070aab8e3fd6186dff8c80983e46f5b6001d914499a18aecd5a02fce32e
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fd4b6599a59298e453c4d085cd48234a781aec741f5259b079ae63537e07c1c7
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1c56eb422b2956686f9277e34bc8cc34789a6a9f96cf48c8d1b348ee4c2ad32c
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c2e7187571674580cade9fd9dfacd95fa924ade19964d58e7bb3223bdcdbc044
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:840cbc90b77dbf5757761c33cef483dae80038c8699a537b601bec429e4b01e9
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2cd1fa900a34ad14997cdf150e9de3df071aa3fe4c05fe4db1c61b32b2a4740c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ff0e24afa3409357bf39531523e499a8986d839d13150d8699023666917a09d
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:82990efadc5b10e2a504af1bdd4a3367edbdc97a4dc60703abf25017a2c875ac
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:905af1194cdc14999c18dad2fabad60fafc93ae7a1c0e79c4823b9eb13773481
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6c9abb1ba3e1b7ce8e555b49390f991a3b537fbfb6d6b26e1bd0c291d5a66e7a
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c67c497a04a0d73e36401384f5befb87de2150301460bc892a7b91a194cddf9c
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9d0f333ee4fb3004963191fa2d8aad29bd882472216b178dad7bbc0e452c4074
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1005ca97ab493a99758751af1ed27f7d6861f63a7b13198af207a82d4466553b
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fc906b60e410824a7dd383e3b2d2e5d2a4a21124b5941e8d66cd1dff093e4f33
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:35f85e297907fbcb53b678fa9ba91187159d898c1cb97d3c43297e528ef7f91d
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:aab6b99bf19a139d59a98baf0538093fa1ac1d789fb675ce2d1672f97dd7cb0e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c4367024a5082b707bb4697c2bda35f75c55af7bbf7b2a4d5e72a490087ab5da
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.8,
6
  "eval_steps": 1000,
7
- "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -1588,6 +1588,164 @@
1588
  "eval_samples_per_second": 18.231,
1589
  "eval_steps_per_second": 0.583,
1590
  "step": 20000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1591
  }
1592
  ],
1593
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.88,
6
  "eval_steps": 1000,
7
+ "global_step": 22000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
1588
  "eval_samples_per_second": 18.231,
1589
  "eval_steps_per_second": 0.583,
1590
  "step": 20000
1591
+ },
1592
+ {
1593
+ "epoch": 0.804,
1594
+ "grad_norm": 1.055684208869934,
1595
+ "learning_rate": 0.00036775145669569497,
1596
+ "loss": 42.2956591796875,
1597
+ "step": 20100
1598
+ },
1599
+ {
1600
+ "epoch": 0.808,
1601
+ "grad_norm": 1.0924086570739746,
1602
+ "learning_rate": 0.0003533513873867442,
1603
+ "loss": 42.5350732421875,
1604
+ "step": 20200
1605
+ },
1606
+ {
1607
+ "epoch": 0.812,
1608
+ "grad_norm": 1.0316708087921143,
1609
+ "learning_rate": 0.0003392115511243421,
1610
+ "loss": 42.4560107421875,
1611
+ "step": 20300
1612
+ },
1613
+ {
1614
+ "epoch": 0.816,
1615
+ "grad_norm": 1.1133615970611572,
1616
+ "learning_rate": 0.00032533418253987323,
1617
+ "loss": 42.3101513671875,
1618
+ "step": 20400
1619
+ },
1620
+ {
1621
+ "epoch": 0.82,
1622
+ "grad_norm": 1.103115200996399,
1623
+ "learning_rate": 0.00031172147478485514,
1624
+ "loss": 42.077275390625,
1625
+ "step": 20500
1626
+ },
1627
+ {
1628
+ "epoch": 0.824,
1629
+ "grad_norm": 1.1964024305343628,
1630
+ "learning_rate": 0.00029837557918433946,
1631
+ "loss": 42.031953125,
1632
+ "step": 20600
1633
+ },
1634
+ {
1635
+ "epoch": 0.828,
1636
+ "grad_norm": 1.4636120796203613,
1637
+ "learning_rate": 0.00028529860489691903,
1638
+ "loss": 42.072587890625,
1639
+ "step": 20700
1640
+ },
1641
+ {
1642
+ "epoch": 0.832,
1643
+ "grad_norm": 1.1224356889724731,
1644
+ "learning_rate": 0.00027249261858140186,
1645
+ "loss": 42.538427734375,
1646
+ "step": 20800
1647
+ },
1648
+ {
1649
+ "epoch": 0.836,
1650
+ "grad_norm": 0.7813518643379211,
1651
+ "learning_rate": 0.00025995964407019923,
1652
+ "loss": 42.4304736328125,
1653
+ "step": 20900
1654
+ },
1655
+ {
1656
+ "epoch": 0.84,
1657
+ "grad_norm": 0.8456737399101257,
1658
+ "learning_rate": 0.0002477016620494856,
1659
+ "loss": 42.2731982421875,
1660
+ "step": 21000
1661
+ },
1662
+ {
1663
+ "epoch": 0.84,
1664
+ "eval_accuracy": 0.22656598240469208,
1665
+ "eval_loss": 42.09526824951172,
1666
+ "eval_runtime": 7.2222,
1667
+ "eval_samples_per_second": 17.308,
1668
+ "eval_steps_per_second": 0.554,
1669
+ "step": 21000
1670
+ },
1671
+ {
1672
+ "epoch": 0.844,
1673
+ "grad_norm": 1.4627004861831665,
1674
+ "learning_rate": 0.00023572060974617082,
1675
+ "loss": 42.2644970703125,
1676
+ "step": 21100
1677
+ },
1678
+ {
1679
+ "epoch": 0.848,
1680
+ "grad_norm": 1.2561938762664795,
1681
+ "learning_rate": 0.0002240183806217493,
1682
+ "loss": 42.0703955078125,
1683
+ "step": 21200
1684
+ },
1685
+ {
1686
+ "epoch": 0.852,
1687
+ "grad_norm": 1.2216907739639282,
1688
+ "learning_rate": 0.00021259682407305847,
1689
+ "loss": 42.2020556640625,
1690
+ "step": 21300
1691
+ },
1692
+ {
1693
+ "epoch": 0.856,
1694
+ "grad_norm": 0.8914040923118591,
1695
+ "learning_rate": 0.00020145774514000438,
1696
+ "loss": 42.357060546875,
1697
+ "step": 21400
1698
+ },
1699
+ {
1700
+ "epoch": 0.86,
1701
+ "grad_norm": 1.1476701498031616,
1702
+ "learning_rate": 0.00019060290422029657,
1703
+ "loss": 42.13251953125,
1704
+ "step": 21500
1705
+ },
1706
+ {
1707
+ "epoch": 0.864,
1708
+ "grad_norm": 1.119214653968811,
1709
+ "learning_rate": 0.00018003401679123887,
1710
+ "loss": 42.2200732421875,
1711
+ "step": 21600
1712
+ },
1713
+ {
1714
+ "epoch": 0.868,
1715
+ "grad_norm": 0.9182925820350647,
1716
+ "learning_rate": 0.0001697527531386196,
1717
+ "loss": 42.22140625,
1718
+ "step": 21700
1719
+ },
1720
+ {
1721
+ "epoch": 0.872,
1722
+ "grad_norm": 1.2138001918792725,
1723
+ "learning_rate": 0.00015976073809273973,
1724
+ "loss": 42.254619140625,
1725
+ "step": 21800
1726
+ },
1727
+ {
1728
+ "epoch": 0.876,
1729
+ "grad_norm": 1.321160078048706,
1730
+ "learning_rate": 0.00015005955077163137,
1731
+ "loss": 42.439716796875,
1732
+ "step": 21900
1733
+ },
1734
+ {
1735
+ "epoch": 0.88,
1736
+ "grad_norm": 0.8333455920219421,
1737
+ "learning_rate": 0.00014065072433149607,
1738
+ "loss": 42.4203271484375,
1739
+ "step": 22000
1740
+ },
1741
+ {
1742
+ "epoch": 0.88,
1743
+ "eval_accuracy": 0.2271329423264907,
1744
+ "eval_loss": 42.06480407714844,
1745
+ "eval_runtime": 6.8562,
1746
+ "eval_samples_per_second": 18.232,
1747
+ "eval_steps_per_second": 0.583,
1748
+ "step": 22000
1749
  }
1750
  ],
1751
  "logging_steps": 100,