Training in progress, step 22000, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +160 -2
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 386379
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:da60d3804c57e29e90c568ae5208c61903451533ab428233f16399bcbfdc9166
|
| 3 |
size 386379
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1540661735
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ae80c44868efaca16e3497e99d1fc33821e44d73899ed13a47e498dc9e7e5f4f
|
| 3 |
size 1540661735
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:61863070aab8e3fd6186dff8c80983e46f5b6001d914499a18aecd5a02fce32e
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1c56eb422b2956686f9277e34bc8cc34789a6a9f96cf48c8d1b348ee4c2ad32c
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:840cbc90b77dbf5757761c33cef483dae80038c8699a537b601bec429e4b01e9
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5ff0e24afa3409357bf39531523e499a8986d839d13150d8699023666917a09d
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:905af1194cdc14999c18dad2fabad60fafc93ae7a1c0e79c4823b9eb13773481
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c67c497a04a0d73e36401384f5befb87de2150301460bc892a7b91a194cddf9c
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1005ca97ab493a99758751af1ed27f7d6861f63a7b13198af207a82d4466553b
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:35f85e297907fbcb53b678fa9ba91187159d898c1cb97d3c43297e528ef7f91d
|
| 3 |
size 14469
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c4367024a5082b707bb4697c2bda35f75c55af7bbf7b2a4d5e72a490087ab5da
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 1000,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -1588,6 +1588,164 @@
|
|
| 1588 |
"eval_samples_per_second": 18.231,
|
| 1589 |
"eval_steps_per_second": 0.583,
|
| 1590 |
"step": 20000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1591 |
}
|
| 1592 |
],
|
| 1593 |
"logging_steps": 100,
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.88,
|
| 6 |
"eval_steps": 1000,
|
| 7 |
+
"global_step": 22000,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 1588 |
"eval_samples_per_second": 18.231,
|
| 1589 |
"eval_steps_per_second": 0.583,
|
| 1590 |
"step": 20000
|
| 1591 |
+
},
|
| 1592 |
+
{
|
| 1593 |
+
"epoch": 0.804,
|
| 1594 |
+
"grad_norm": 1.055684208869934,
|
| 1595 |
+
"learning_rate": 0.00036775145669569497,
|
| 1596 |
+
"loss": 42.2956591796875,
|
| 1597 |
+
"step": 20100
|
| 1598 |
+
},
|
| 1599 |
+
{
|
| 1600 |
+
"epoch": 0.808,
|
| 1601 |
+
"grad_norm": 1.0924086570739746,
|
| 1602 |
+
"learning_rate": 0.0003533513873867442,
|
| 1603 |
+
"loss": 42.5350732421875,
|
| 1604 |
+
"step": 20200
|
| 1605 |
+
},
|
| 1606 |
+
{
|
| 1607 |
+
"epoch": 0.812,
|
| 1608 |
+
"grad_norm": 1.0316708087921143,
|
| 1609 |
+
"learning_rate": 0.0003392115511243421,
|
| 1610 |
+
"loss": 42.4560107421875,
|
| 1611 |
+
"step": 20300
|
| 1612 |
+
},
|
| 1613 |
+
{
|
| 1614 |
+
"epoch": 0.816,
|
| 1615 |
+
"grad_norm": 1.1133615970611572,
|
| 1616 |
+
"learning_rate": 0.00032533418253987323,
|
| 1617 |
+
"loss": 42.3101513671875,
|
| 1618 |
+
"step": 20400
|
| 1619 |
+
},
|
| 1620 |
+
{
|
| 1621 |
+
"epoch": 0.82,
|
| 1622 |
+
"grad_norm": 1.103115200996399,
|
| 1623 |
+
"learning_rate": 0.00031172147478485514,
|
| 1624 |
+
"loss": 42.077275390625,
|
| 1625 |
+
"step": 20500
|
| 1626 |
+
},
|
| 1627 |
+
{
|
| 1628 |
+
"epoch": 0.824,
|
| 1629 |
+
"grad_norm": 1.1964024305343628,
|
| 1630 |
+
"learning_rate": 0.00029837557918433946,
|
| 1631 |
+
"loss": 42.031953125,
|
| 1632 |
+
"step": 20600
|
| 1633 |
+
},
|
| 1634 |
+
{
|
| 1635 |
+
"epoch": 0.828,
|
| 1636 |
+
"grad_norm": 1.4636120796203613,
|
| 1637 |
+
"learning_rate": 0.00028529860489691903,
|
| 1638 |
+
"loss": 42.072587890625,
|
| 1639 |
+
"step": 20700
|
| 1640 |
+
},
|
| 1641 |
+
{
|
| 1642 |
+
"epoch": 0.832,
|
| 1643 |
+
"grad_norm": 1.1224356889724731,
|
| 1644 |
+
"learning_rate": 0.00027249261858140186,
|
| 1645 |
+
"loss": 42.538427734375,
|
| 1646 |
+
"step": 20800
|
| 1647 |
+
},
|
| 1648 |
+
{
|
| 1649 |
+
"epoch": 0.836,
|
| 1650 |
+
"grad_norm": 0.7813518643379211,
|
| 1651 |
+
"learning_rate": 0.00025995964407019923,
|
| 1652 |
+
"loss": 42.4304736328125,
|
| 1653 |
+
"step": 20900
|
| 1654 |
+
},
|
| 1655 |
+
{
|
| 1656 |
+
"epoch": 0.84,
|
| 1657 |
+
"grad_norm": 0.8456737399101257,
|
| 1658 |
+
"learning_rate": 0.0002477016620494856,
|
| 1659 |
+
"loss": 42.2731982421875,
|
| 1660 |
+
"step": 21000
|
| 1661 |
+
},
|
| 1662 |
+
{
|
| 1663 |
+
"epoch": 0.84,
|
| 1664 |
+
"eval_accuracy": 0.22656598240469208,
|
| 1665 |
+
"eval_loss": 42.09526824951172,
|
| 1666 |
+
"eval_runtime": 7.2222,
|
| 1667 |
+
"eval_samples_per_second": 17.308,
|
| 1668 |
+
"eval_steps_per_second": 0.554,
|
| 1669 |
+
"step": 21000
|
| 1670 |
+
},
|
| 1671 |
+
{
|
| 1672 |
+
"epoch": 0.844,
|
| 1673 |
+
"grad_norm": 1.4627004861831665,
|
| 1674 |
+
"learning_rate": 0.00023572060974617082,
|
| 1675 |
+
"loss": 42.2644970703125,
|
| 1676 |
+
"step": 21100
|
| 1677 |
+
},
|
| 1678 |
+
{
|
| 1679 |
+
"epoch": 0.848,
|
| 1680 |
+
"grad_norm": 1.2561938762664795,
|
| 1681 |
+
"learning_rate": 0.0002240183806217493,
|
| 1682 |
+
"loss": 42.0703955078125,
|
| 1683 |
+
"step": 21200
|
| 1684 |
+
},
|
| 1685 |
+
{
|
| 1686 |
+
"epoch": 0.852,
|
| 1687 |
+
"grad_norm": 1.2216907739639282,
|
| 1688 |
+
"learning_rate": 0.00021259682407305847,
|
| 1689 |
+
"loss": 42.2020556640625,
|
| 1690 |
+
"step": 21300
|
| 1691 |
+
},
|
| 1692 |
+
{
|
| 1693 |
+
"epoch": 0.856,
|
| 1694 |
+
"grad_norm": 0.8914040923118591,
|
| 1695 |
+
"learning_rate": 0.00020145774514000438,
|
| 1696 |
+
"loss": 42.357060546875,
|
| 1697 |
+
"step": 21400
|
| 1698 |
+
},
|
| 1699 |
+
{
|
| 1700 |
+
"epoch": 0.86,
|
| 1701 |
+
"grad_norm": 1.1476701498031616,
|
| 1702 |
+
"learning_rate": 0.00019060290422029657,
|
| 1703 |
+
"loss": 42.13251953125,
|
| 1704 |
+
"step": 21500
|
| 1705 |
+
},
|
| 1706 |
+
{
|
| 1707 |
+
"epoch": 0.864,
|
| 1708 |
+
"grad_norm": 1.119214653968811,
|
| 1709 |
+
"learning_rate": 0.00018003401679123887,
|
| 1710 |
+
"loss": 42.2200732421875,
|
| 1711 |
+
"step": 21600
|
| 1712 |
+
},
|
| 1713 |
+
{
|
| 1714 |
+
"epoch": 0.868,
|
| 1715 |
+
"grad_norm": 0.9182925820350647,
|
| 1716 |
+
"learning_rate": 0.0001697527531386196,
|
| 1717 |
+
"loss": 42.22140625,
|
| 1718 |
+
"step": 21700
|
| 1719 |
+
},
|
| 1720 |
+
{
|
| 1721 |
+
"epoch": 0.872,
|
| 1722 |
+
"grad_norm": 1.2138001918792725,
|
| 1723 |
+
"learning_rate": 0.00015976073809273973,
|
| 1724 |
+
"loss": 42.254619140625,
|
| 1725 |
+
"step": 21800
|
| 1726 |
+
},
|
| 1727 |
+
{
|
| 1728 |
+
"epoch": 0.876,
|
| 1729 |
+
"grad_norm": 1.321160078048706,
|
| 1730 |
+
"learning_rate": 0.00015005955077163137,
|
| 1731 |
+
"loss": 42.439716796875,
|
| 1732 |
+
"step": 21900
|
| 1733 |
+
},
|
| 1734 |
+
{
|
| 1735 |
+
"epoch": 0.88,
|
| 1736 |
+
"grad_norm": 0.8333455920219421,
|
| 1737 |
+
"learning_rate": 0.00014065072433149607,
|
| 1738 |
+
"loss": 42.4203271484375,
|
| 1739 |
+
"step": 22000
|
| 1740 |
+
},
|
| 1741 |
+
{
|
| 1742 |
+
"epoch": 0.88,
|
| 1743 |
+
"eval_accuracy": 0.2271329423264907,
|
| 1744 |
+
"eval_loss": 42.06480407714844,
|
| 1745 |
+
"eval_runtime": 6.8562,
|
| 1746 |
+
"eval_samples_per_second": 18.232,
|
| 1747 |
+
"eval_steps_per_second": 0.583,
|
| 1748 |
+
"step": 22000
|
| 1749 |
}
|
| 1750 |
],
|
| 1751 |
"logging_steps": 100,
|