Training in progress, step 12000, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +160 -2
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1600779
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:895b25ff701dcc070741cf42351fb18c429fa56d815b15c7ba3c867f23a10980
|
| 3 |
size 1600779
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 621149155
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1313ea19ff807c10147842a59b9051e7021a8c028fe68e36adab013243f36b38
|
| 3 |
size 621149155
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f794b047026bd2e455c2f0c1a40191bfadaa2e43fd08fedee4185a5059db1b48
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a5541088e5d274181d6c9ef46a5b103eb5b8bf402d2abae7634635f86f3865ab
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:da6b2d1e9271342e1e302bb1b26316b143cfc6c4934d2ed389c10bb161aa25ef
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7900dd4bec725410e9ff4f8c55274ea9809b523a68297edbd3a83622136595cd
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d05d56cf9a1650f01c1af9b1a787ec6db7126c64049f267e937381c9944e1282
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9be3515ba8ff34af4f0c40faacd8a5a9e8e607850863dcb77d772f0606a1ae0c
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d2e3e189e1259797127a8750045cf99ca33e22e9a0c01f4e0afcdf1fa22431dd
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3a9afcf27fe2dffd21d9ec60d55b8d2840fcdcf2a30cdfc9362419caa2813c93
|
| 3 |
size 14469
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:43ab2089bb2cb7b873a77ac8e24a3b3b2416ededca3811b6bb0922f61bea4c99
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 1000,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -798,6 +798,164 @@
|
|
| 798 |
"eval_samples_per_second": 39.294,
|
| 799 |
"eval_steps_per_second": 0.629,
|
| 800 |
"step": 10000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 801 |
}
|
| 802 |
],
|
| 803 |
"logging_steps": 100,
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.24,
|
| 6 |
"eval_steps": 1000,
|
| 7 |
+
"global_step": 12000,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 798 |
"eval_samples_per_second": 39.294,
|
| 799 |
"eval_steps_per_second": 0.629,
|
| 800 |
"step": 10000
|
| 801 |
+
},
|
| 802 |
+
{
|
| 803 |
+
"epoch": 0.202,
|
| 804 |
+
"grad_norm": 1.518142819404602,
|
| 805 |
+
"learning_rate": 0.0018763385407335963,
|
| 806 |
+
"loss": 32.7453,
|
| 807 |
+
"step": 10100
|
| 808 |
+
},
|
| 809 |
+
{
|
| 810 |
+
"epoch": 0.204,
|
| 811 |
+
"grad_norm": 1.908083200454712,
|
| 812 |
+
"learning_rate": 0.001873133519711602,
|
| 813 |
+
"loss": 32.9035,
|
| 814 |
+
"step": 10200
|
| 815 |
+
},
|
| 816 |
+
{
|
| 817 |
+
"epoch": 0.206,
|
| 818 |
+
"grad_norm": 2.4479877948760986,
|
| 819 |
+
"learning_rate": 0.0018698903050008956,
|
| 820 |
+
"loss": 32.3298,
|
| 821 |
+
"step": 10300
|
| 822 |
+
},
|
| 823 |
+
{
|
| 824 |
+
"epoch": 0.208,
|
| 825 |
+
"grad_norm": 1.403601050376892,
|
| 826 |
+
"learning_rate": 0.0018666090384701947,
|
| 827 |
+
"loss": 32.0576,
|
| 828 |
+
"step": 10400
|
| 829 |
+
},
|
| 830 |
+
{
|
| 831 |
+
"epoch": 0.21,
|
| 832 |
+
"grad_norm": 1.7655203342437744,
|
| 833 |
+
"learning_rate": 0.001863289863652727,
|
| 834 |
+
"loss": 32.8307,
|
| 835 |
+
"step": 10500
|
| 836 |
+
},
|
| 837 |
+
{
|
| 838 |
+
"epoch": 0.212,
|
| 839 |
+
"grad_norm": 1.317898154258728,
|
| 840 |
+
"learning_rate": 0.0018599329257399516,
|
| 841 |
+
"loss": 32.8206,
|
| 842 |
+
"step": 10600
|
| 843 |
+
},
|
| 844 |
+
{
|
| 845 |
+
"epoch": 0.214,
|
| 846 |
+
"grad_norm": 1.6275999546051025,
|
| 847 |
+
"learning_rate": 0.0018565383715752083,
|
| 848 |
+
"loss": 33.1208,
|
| 849 |
+
"step": 10700
|
| 850 |
+
},
|
| 851 |
+
{
|
| 852 |
+
"epoch": 0.216,
|
| 853 |
+
"grad_norm": 1.350576639175415,
|
| 854 |
+
"learning_rate": 0.0018531063496472927,
|
| 855 |
+
"loss": 33.09,
|
| 856 |
+
"step": 10800
|
| 857 |
+
},
|
| 858 |
+
{
|
| 859 |
+
"epoch": 0.218,
|
| 860 |
+
"grad_norm": 2.071432590484619,
|
| 861 |
+
"learning_rate": 0.0018496370100839622,
|
| 862 |
+
"loss": 32.4823,
|
| 863 |
+
"step": 10900
|
| 864 |
+
},
|
| 865 |
+
{
|
| 866 |
+
"epoch": 0.22,
|
| 867 |
+
"grad_norm": 1.756463646888733,
|
| 868 |
+
"learning_rate": 0.0018461305046453683,
|
| 869 |
+
"loss": 32.3173,
|
| 870 |
+
"step": 11000
|
| 871 |
+
},
|
| 872 |
+
{
|
| 873 |
+
"epoch": 0.22,
|
| 874 |
+
"eval_accuracy": 0.29537964774951075,
|
| 875 |
+
"eval_loss": 32.61259078979492,
|
| 876 |
+
"eval_runtime": 3.1999,
|
| 877 |
+
"eval_samples_per_second": 39.064,
|
| 878 |
+
"eval_steps_per_second": 0.625,
|
| 879 |
+
"step": 11000
|
| 880 |
+
},
|
| 881 |
+
{
|
| 882 |
+
"epoch": 0.222,
|
| 883 |
+
"grad_norm": 2.6366186141967773,
|
| 884 |
+
"learning_rate": 0.0018425869867174187,
|
| 885 |
+
"loss": 32.0905,
|
| 886 |
+
"step": 11100
|
| 887 |
+
},
|
| 888 |
+
{
|
| 889 |
+
"epoch": 0.224,
|
| 890 |
+
"grad_norm": 1.3121482133865356,
|
| 891 |
+
"learning_rate": 0.0018390066113050665,
|
| 892 |
+
"loss": 32.2553,
|
| 893 |
+
"step": 11200
|
| 894 |
+
},
|
| 895 |
+
{
|
| 896 |
+
"epoch": 0.226,
|
| 897 |
+
"grad_norm": 8.878124237060547,
|
| 898 |
+
"learning_rate": 0.0018353895350255317,
|
| 899 |
+
"loss": 33.1959,
|
| 900 |
+
"step": 11300
|
| 901 |
+
},
|
| 902 |
+
{
|
| 903 |
+
"epoch": 0.228,
|
| 904 |
+
"grad_norm": 2.0059814453125,
|
| 905 |
+
"learning_rate": 0.0018317359161014477,
|
| 906 |
+
"loss": 32.804,
|
| 907 |
+
"step": 11400
|
| 908 |
+
},
|
| 909 |
+
{
|
| 910 |
+
"epoch": 0.23,
|
| 911 |
+
"grad_norm": 1.3689004182815552,
|
| 912 |
+
"learning_rate": 0.001828045914353943,
|
| 913 |
+
"loss": 32.6974,
|
| 914 |
+
"step": 11500
|
| 915 |
+
},
|
| 916 |
+
{
|
| 917 |
+
"epoch": 0.232,
|
| 918 |
+
"grad_norm": 1.6248316764831543,
|
| 919 |
+
"learning_rate": 0.0018243196911956476,
|
| 920 |
+
"loss": 32.5622,
|
| 921 |
+
"step": 11600
|
| 922 |
+
},
|
| 923 |
+
{
|
| 924 |
+
"epoch": 0.234,
|
| 925 |
+
"grad_norm": 1.3218642473220825,
|
| 926 |
+
"learning_rate": 0.0018205574096236336,
|
| 927 |
+
"loss": 32.348,
|
| 928 |
+
"step": 11700
|
| 929 |
+
},
|
| 930 |
+
{
|
| 931 |
+
"epoch": 0.236,
|
| 932 |
+
"grad_norm": 1.4009215831756592,
|
| 933 |
+
"learning_rate": 0.0018167592342122857,
|
| 934 |
+
"loss": 31.677,
|
| 935 |
+
"step": 11800
|
| 936 |
+
},
|
| 937 |
+
{
|
| 938 |
+
"epoch": 0.238,
|
| 939 |
+
"grad_norm": 1.8492565155029297,
|
| 940 |
+
"learning_rate": 0.0018129253311061002,
|
| 941 |
+
"loss": 31.6381,
|
| 942 |
+
"step": 11900
|
| 943 |
+
},
|
| 944 |
+
{
|
| 945 |
+
"epoch": 0.24,
|
| 946 |
+
"grad_norm": 2.3902571201324463,
|
| 947 |
+
"learning_rate": 0.0018090558680124193,
|
| 948 |
+
"loss": 32.4851,
|
| 949 |
+
"step": 12000
|
| 950 |
+
},
|
| 951 |
+
{
|
| 952 |
+
"epoch": 0.24,
|
| 953 |
+
"eval_accuracy": 0.300720156555773,
|
| 954 |
+
"eval_loss": 32.26734161376953,
|
| 955 |
+
"eval_runtime": 3.3297,
|
| 956 |
+
"eval_samples_per_second": 37.54,
|
| 957 |
+
"eval_steps_per_second": 0.601,
|
| 958 |
+
"step": 12000
|
| 959 |
}
|
| 960 |
],
|
| 961 |
"logging_steps": 100,
|