CodeIsAbstract commited on
Commit
ccc1f20
·
verified ·
1 Parent(s): 2b29d96

Training in progress, step 22000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:80f6cba2576f1ee2628aa99fb4980b692c7f23dc4a1cc36b06da1ee070ed4750
3
  size 496262784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09db5475ee35a12c8b710afb6e180ab5bfcd8ce48967fc58bd5457a4e3977985
3
  size 496262784
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:238c57d596e4b481c83fa003f41489533d0e0d6fa65bc29cfff61620056f3f43
3
  size 992621963
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff60aa4ce7613d8a03c55e84221468eaecbd9f2572db7e051cb5a7fa0b4c3a04
3
  size 992621963
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f2d3bdd032f891e90f3fae19dbb6eb0bfaf6ef591769711ba856b31e40586a5e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4281625b7592c11b921e209e5b1131cddcd67ffe5764c2033c0ef0adb2d550bb
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c5b7f9f4f6c99c7a7ebbb7780b490f0a9830b3a79c1c5a07331e7e6edccccd28
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a0231b5cc777a2809e149f15d91a081c8817fce35708f3c12c17c3012604a372
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.4,
6
  "eval_steps": 1000,
7
- "global_step": 20000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1588,6 +1588,164 @@
1588
  "eval_samples_per_second": 329.872,
1589
  "eval_steps_per_second": 20.734,
1590
  "step": 20000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1591
  }
1592
  ],
1593
  "logging_steps": 100,
@@ -1607,7 +1765,7 @@
1607
  "attributes": {}
1608
  }
1609
  },
1610
- "total_flos": 6.271008768e+17,
1611
  "train_batch_size": 120,
1612
  "trial_name": null,
1613
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.44,
6
  "eval_steps": 1000,
7
+ "global_step": 22000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1588
  "eval_samples_per_second": 329.872,
1589
  "eval_steps_per_second": 20.734,
1590
  "step": 20000
1591
+ },
1592
+ {
1593
+ "epoch": 0.402,
1594
+ "grad_norm": 0.20392677187919617,
1595
+ "learning_rate": 0.0003925432549872678,
1596
+ "loss": 3.467202453613281,
1597
+ "step": 20100
1598
+ },
1599
+ {
1600
+ "epoch": 0.404,
1601
+ "grad_norm": 0.21144644916057587,
1602
+ "learning_rate": 0.0003907430044951278,
1603
+ "loss": 3.463043212890625,
1604
+ "step": 20200
1605
+ },
1606
+ {
1607
+ "epoch": 0.406,
1608
+ "grad_norm": 0.1932135373353958,
1609
+ "learning_rate": 0.00038893915003323544,
1610
+ "loss": 3.484923095703125,
1611
+ "step": 20300
1612
+ },
1613
+ {
1614
+ "epoch": 0.408,
1615
+ "grad_norm": 0.20958249270915985,
1616
+ "learning_rate": 0.00038713176324388387,
1617
+ "loss": 3.483151550292969,
1618
+ "step": 20400
1619
+ },
1620
+ {
1621
+ "epoch": 0.41,
1622
+ "grad_norm": 0.4303552210330963,
1623
+ "learning_rate": 0.00038532091590965674,
1624
+ "loss": 3.477305603027344,
1625
+ "step": 20500
1626
+ },
1627
+ {
1628
+ "epoch": 0.412,
1629
+ "grad_norm": 0.20084762573242188,
1630
+ "learning_rate": 0.0003835066799505776,
1631
+ "loss": 3.47593994140625,
1632
+ "step": 20600
1633
+ },
1634
+ {
1635
+ "epoch": 0.414,
1636
+ "grad_norm": 0.18663835525512695,
1637
+ "learning_rate": 0.0003816891274212534,
1638
+ "loss": 3.465316162109375,
1639
+ "step": 20700
1640
+ },
1641
+ {
1642
+ "epoch": 0.416,
1643
+ "grad_norm": 0.1979358196258545,
1644
+ "learning_rate": 0.00037986833050801256,
1645
+ "loss": 3.465248718261719,
1646
+ "step": 20800
1647
+ },
1648
+ {
1649
+ "epoch": 0.418,
1650
+ "grad_norm": 0.20419807732105255,
1651
+ "learning_rate": 0.0003780443615260386,
1652
+ "loss": 3.4529428100585937,
1653
+ "step": 20900
1654
+ },
1655
+ {
1656
+ "epoch": 0.42,
1657
+ "grad_norm": 0.18685045838356018,
1658
+ "learning_rate": 0.0003762172929164974,
1659
+ "loss": 3.4456027221679686,
1660
+ "step": 21000
1661
+ },
1662
+ {
1663
+ "epoch": 0.42,
1664
+ "eval_accuracy": 0.3708611474909034,
1665
+ "eval_loss": 3.402836799621582,
1666
+ "eval_runtime": 5.7802,
1667
+ "eval_samples_per_second": 335.801,
1668
+ "eval_steps_per_second": 21.106,
1669
+ "step": 21000
1670
+ },
1671
+ {
1672
+ "epoch": 0.422,
1673
+ "grad_norm": 0.18441364169120789,
1674
+ "learning_rate": 0.0003743871972436601,
1675
+ "loss": 3.4683151245117188,
1676
+ "step": 21100
1677
+ },
1678
+ {
1679
+ "epoch": 0.424,
1680
+ "grad_norm": 0.18499158322811127,
1681
+ "learning_rate": 0.0003725541471920217,
1682
+ "loss": 3.440291748046875,
1683
+ "step": 21200
1684
+ },
1685
+ {
1686
+ "epoch": 0.426,
1687
+ "grad_norm": 0.19171582162380219,
1688
+ "learning_rate": 0.00037071821556341393,
1689
+ "loss": 3.4521221923828125,
1690
+ "step": 21300
1691
+ },
1692
+ {
1693
+ "epoch": 0.428,
1694
+ "grad_norm": 0.2012597918510437,
1695
+ "learning_rate": 0.0003688794752741139,
1696
+ "loss": 3.4446597290039063,
1697
+ "step": 21400
1698
+ },
1699
+ {
1700
+ "epoch": 0.43,
1701
+ "grad_norm": 0.20872637629508972,
1702
+ "learning_rate": 0.000367037999351948,
1703
+ "loss": 3.4504824829101564,
1704
+ "step": 21500
1705
+ },
1706
+ {
1707
+ "epoch": 0.432,
1708
+ "grad_norm": 0.19605720043182373,
1709
+ "learning_rate": 0.0003651938609333918,
1710
+ "loss": 3.432515869140625,
1711
+ "step": 21600
1712
+ },
1713
+ {
1714
+ "epoch": 0.434,
1715
+ "grad_norm": 0.21077193319797516,
1716
+ "learning_rate": 0.00036334713326066496,
1717
+ "loss": 3.430672607421875,
1718
+ "step": 21700
1719
+ },
1720
+ {
1721
+ "epoch": 0.436,
1722
+ "grad_norm": 0.20861658453941345,
1723
+ "learning_rate": 0.00036149788967882265,
1724
+ "loss": 3.434753112792969,
1725
+ "step": 21800
1726
+ },
1727
+ {
1728
+ "epoch": 0.438,
1729
+ "grad_norm": 0.18320897221565247,
1730
+ "learning_rate": 0.00035964620363284266,
1731
+ "loss": 3.4054229736328123,
1732
+ "step": 21900
1733
+ },
1734
+ {
1735
+ "epoch": 0.44,
1736
+ "grad_norm": 0.20277726650238037,
1737
+ "learning_rate": 0.00035779214866470794,
1738
+ "loss": 3.412484130859375,
1739
+ "step": 22000
1740
+ },
1741
+ {
1742
+ "epoch": 0.44,
1743
+ "eval_accuracy": 0.37140054302511166,
1744
+ "eval_loss": 3.3950016498565674,
1745
+ "eval_runtime": 6.2447,
1746
+ "eval_samples_per_second": 310.823,
1747
+ "eval_steps_per_second": 19.537,
1748
+ "step": 22000
1749
  }
1750
  ],
1751
  "logging_steps": 100,
 
1765
  "attributes": {}
1766
  }
1767
  },
1768
+ "total_flos": 6.8981096448e+17,
1769
  "train_batch_size": 120,
1770
  "trial_name": null,
1771
  "trial_params": null