Training in progress, step 12000, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +160 -2
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 386379
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4b8e8d3ec310b587a737b2ff1479785c95b383b0103a20c2616bcaa1b70cedfd
|
| 3 |
size 386379
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1540661735
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af1c4c4789099e3f06b7040705745688787a84de28b3c92b707a08bf66acedd5
|
| 3 |
size 1540661735
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:de97c976c1994a7f2b52cd661dd45e50f8c59ea7f29046e2b153d031a5329003
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:945dfe762c470357f0955a0e4fddcfadf68d636dfea2666def3fd34df1ab5189
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3883b9681a4e564010f0c080b9a5c164f6cfc69d29380b6cdfc7bd14e3061a6c
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1bfa4cd8ff4ea106b4895cc38559e225da4381a514e25136b0ad798f67a76ec0
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7bd3539150b8b4416c1ef3e8435df0b2b18db72a3b53e80a325b1d948246653b
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:414056302bb2c51030907ea288f267bbdafb5b349fed9bbe5df32046e41df462
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:217d507942f28f310e4eeed6bfba8d07fd2bb6dc167c4c4f66eda1c08b6318d6
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:06b43a58b72baf558b7e5624a89206c612752404e0e3755fa74aff6a8a856fea
|
| 3 |
size 14469
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ef456277f681db2c07f1ff63be489edd84235ae448b25e8872b7c6ddb3f68afb
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 1000,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -798,6 +798,164 @@
|
|
| 798 |
"eval_samples_per_second": 17.03,
|
| 799 |
"eval_steps_per_second": 0.545,
|
| 800 |
"step": 10000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 801 |
}
|
| 802 |
],
|
| 803 |
"logging_steps": 100,
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.48,
|
| 6 |
"eval_steps": 1000,
|
| 7 |
+
"global_step": 12000,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": false,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 798 |
"eval_samples_per_second": 17.03,
|
| 799 |
"eval_steps_per_second": 0.545,
|
| 800 |
"step": 10000
|
| 801 |
+
},
|
| 802 |
+
{
|
| 803 |
+
"epoch": 0.404,
|
| 804 |
+
"grad_norm": 3.5040667057037354,
|
| 805 |
+
"learning_rate": 0.0025957538900165333,
|
| 806 |
+
"loss": 44.2118505859375,
|
| 807 |
+
"step": 10100
|
| 808 |
+
},
|
| 809 |
+
{
|
| 810 |
+
"epoch": 0.408,
|
| 811 |
+
"grad_norm": 3.0154404640197754,
|
| 812 |
+
"learning_rate": 0.0025717060234955805,
|
| 813 |
+
"loss": 44.339365234375,
|
| 814 |
+
"step": 10200
|
| 815 |
+
},
|
| 816 |
+
{
|
| 817 |
+
"epoch": 0.412,
|
| 818 |
+
"grad_norm": 3.0024242401123047,
|
| 819 |
+
"learning_rate": 0.0025475678057004796,
|
| 820 |
+
"loss": 44.2085009765625,
|
| 821 |
+
"step": 10300
|
| 822 |
+
},
|
| 823 |
+
{
|
| 824 |
+
"epoch": 0.416,
|
| 825 |
+
"grad_norm": 2.868852138519287,
|
| 826 |
+
"learning_rate": 0.002523343051386793,
|
| 827 |
+
"loss": 44.1264453125,
|
| 828 |
+
"step": 10400
|
| 829 |
+
},
|
| 830 |
+
{
|
| 831 |
+
"epoch": 0.42,
|
| 832 |
+
"grad_norm": 3.098557472229004,
|
| 833 |
+
"learning_rate": 0.0024990355889861417,
|
| 834 |
+
"loss": 44.0060986328125,
|
| 835 |
+
"step": 10500
|
| 836 |
+
},
|
| 837 |
+
{
|
| 838 |
+
"epoch": 0.424,
|
| 839 |
+
"grad_norm": 2.94388747215271,
|
| 840 |
+
"learning_rate": 0.0024746492600011666,
|
| 841 |
+
"loss": 44.090751953125,
|
| 842 |
+
"step": 10600
|
| 843 |
+
},
|
| 844 |
+
{
|
| 845 |
+
"epoch": 0.428,
|
| 846 |
+
"grad_norm": 3.172335147857666,
|
| 847 |
+
"learning_rate": 0.002450187918398426,
|
| 848 |
+
"loss": 44.108330078125,
|
| 849 |
+
"step": 10700
|
| 850 |
+
},
|
| 851 |
+
{
|
| 852 |
+
"epoch": 0.432,
|
| 853 |
+
"grad_norm": 3.123621702194214,
|
| 854 |
+
"learning_rate": 0.0024256554299993214,
|
| 855 |
+
"loss": 44.00998046875,
|
| 856 |
+
"step": 10800
|
| 857 |
+
},
|
| 858 |
+
{
|
| 859 |
+
"epoch": 0.436,
|
| 860 |
+
"grad_norm": 2.8541388511657715,
|
| 861 |
+
"learning_rate": 0.002401055671869152,
|
| 862 |
+
"loss": 43.941904296875,
|
| 863 |
+
"step": 10900
|
| 864 |
+
},
|
| 865 |
+
{
|
| 866 |
+
"epoch": 0.44,
|
| 867 |
+
"grad_norm": 2.895085573196411,
|
| 868 |
+
"learning_rate": 0.0023763925317043903,
|
| 869 |
+
"loss": 43.792197265625,
|
| 870 |
+
"step": 11000
|
| 871 |
+
},
|
| 872 |
+
{
|
| 873 |
+
"epoch": 0.44,
|
| 874 |
+
"eval_accuracy": 0.20995894428152492,
|
| 875 |
+
"eval_loss": 43.750606536865234,
|
| 876 |
+
"eval_runtime": 6.9444,
|
| 877 |
+
"eval_samples_per_second": 18.0,
|
| 878 |
+
"eval_steps_per_second": 0.576,
|
| 879 |
+
"step": 11000
|
| 880 |
+
},
|
| 881 |
+
{
|
| 882 |
+
"epoch": 0.444,
|
| 883 |
+
"grad_norm": 2.346236228942871,
|
| 884 |
+
"learning_rate": 0.0023516699072182786,
|
| 885 |
+
"loss": 43.9089990234375,
|
| 886 |
+
"step": 11100
|
| 887 |
+
},
|
| 888 |
+
{
|
| 889 |
+
"epoch": 0.448,
|
| 890 |
+
"grad_norm": 2.3568971157073975,
|
| 891 |
+
"learning_rate": 0.002326891705524841,
|
| 892 |
+
"loss": 43.76560546875,
|
| 893 |
+
"step": 11200
|
| 894 |
+
},
|
| 895 |
+
{
|
| 896 |
+
"epoch": 0.452,
|
| 897 |
+
"grad_norm": 3.0081124305725098,
|
| 898 |
+
"learning_rate": 0.0023020618425214144,
|
| 899 |
+
"loss": 43.808974609375,
|
| 900 |
+
"step": 11300
|
| 901 |
+
},
|
| 902 |
+
{
|
| 903 |
+
"epoch": 0.456,
|
| 904 |
+
"grad_norm": 3.7043468952178955,
|
| 905 |
+
"learning_rate": 0.0022771842422697835,
|
| 906 |
+
"loss": 43.849228515625,
|
| 907 |
+
"step": 11400
|
| 908 |
+
},
|
| 909 |
+
{
|
| 910 |
+
"epoch": 0.46,
|
| 911 |
+
"grad_norm": 2.2964367866516113,
|
| 912 |
+
"learning_rate": 0.0022522628363760323,
|
| 913 |
+
"loss": 43.599345703125,
|
| 914 |
+
"step": 11500
|
| 915 |
+
},
|
| 916 |
+
{
|
| 917 |
+
"epoch": 0.464,
|
| 918 |
+
"grad_norm": 2.747478485107422,
|
| 919 |
+
"learning_rate": 0.002227301563369202,
|
| 920 |
+
"loss": 43.42494140625,
|
| 921 |
+
"step": 11600
|
| 922 |
+
},
|
| 923 |
+
{
|
| 924 |
+
"epoch": 0.468,
|
| 925 |
+
"grad_norm": 2.878131866455078,
|
| 926 |
+
"learning_rate": 0.0022023043680788512,
|
| 927 |
+
"loss": 43.6928955078125,
|
| 928 |
+
"step": 11700
|
| 929 |
+
},
|
| 930 |
+
{
|
| 931 |
+
"epoch": 0.472,
|
| 932 |
+
"grad_norm": 2.552713632583618,
|
| 933 |
+
"learning_rate": 0.0021772752010116247,
|
| 934 |
+
"loss": 43.6564453125,
|
| 935 |
+
"step": 11800
|
| 936 |
+
},
|
| 937 |
+
{
|
| 938 |
+
"epoch": 0.476,
|
| 939 |
+
"grad_norm": 2.9993340969085693,
|
| 940 |
+
"learning_rate": 0.002152218017726923,
|
| 941 |
+
"loss": 43.607763671875,
|
| 942 |
+
"step": 11900
|
| 943 |
+
},
|
| 944 |
+
{
|
| 945 |
+
"epoch": 0.48,
|
| 946 |
+
"grad_norm": 2.911576747894287,
|
| 947 |
+
"learning_rate": 0.002127136778211772,
|
| 948 |
+
"loss": 43.8162890625,
|
| 949 |
+
"step": 12000
|
| 950 |
+
},
|
| 951 |
+
{
|
| 952 |
+
"epoch": 0.48,
|
| 953 |
+
"eval_accuracy": 0.21305474095796675,
|
| 954 |
+
"eval_loss": 43.4426155090332,
|
| 955 |
+
"eval_runtime": 7.5722,
|
| 956 |
+
"eval_samples_per_second": 16.508,
|
| 957 |
+
"eval_steps_per_second": 0.528,
|
| 958 |
+
"step": 12000
|
| 959 |
}
|
| 960 |
],
|
| 961 |
"logging_steps": 100,
|