Training in progress, step 500, checkpoint
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/trainer_state.json +35 -35
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 386379
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:69ca6905e70e7a92d02eeddbdeb7103009b83d9f0f234789efb4904d68ed5838
|
| 3 |
size 386379
|
last-checkpoint/pytorch_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1540661735
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:82cc3acf680d85a95964b4366417613a590457a2020ce3a7574762bb50f5df4d
|
| 3 |
size 1540661735
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:88b48f43625e8919759a700846736ad1c67a9fb02b424ca27c045fdf11ebec4d
|
| 3 |
size 14469
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a3df8bbf0aa54958cebb063d8843254a6ca7fa9a4ecc3a91470b305106963beb
|
| 3 |
size 14469
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0c2aba2784b08e863f59f03cfa3309c63893fb49cde8f0070f10410339e4d7fd
|
| 3 |
size 14469
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:827e90af5517da4bb9113a66cb2136b09f8d4c852d96861d993b1d3b68201f38
|
| 3 |
size 14469
|
last-checkpoint/rng_state_4.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4db1556f9cda170711f81bfb9343feab14e545a26997a7968c5820a578a542c7
|
| 3 |
size 14469
|
last-checkpoint/rng_state_5.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:165ab5fc90c8c3dd2e6e924be06814806faf1d60acb8d14cc4b0676f6f5737f1
|
| 3 |
size 14469
|
last-checkpoint/rng_state_6.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:017ce61066c39ae46b1e2925435311fe07b1d409bfd8cdcf79a1c7f1a9b9127b
|
| 3 |
size 14469
|
last-checkpoint/rng_state_7.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14469
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:64b97a485d2b1a72ff869a0ba55eed82c522e3229735d8894bfab7c67a78fe16
|
| 3 |
size 14469
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -11,82 +11,82 @@
|
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
"epoch": 0.1,
|
| 14 |
-
"grad_norm":
|
| 15 |
"learning_rate": 0.003920764340590609,
|
| 16 |
-
"loss":
|
| 17 |
"step": 100
|
| 18 |
},
|
| 19 |
{
|
| 20 |
"epoch": 0.1,
|
| 21 |
-
"eval_accuracy": 0.
|
| 22 |
-
"eval_loss":
|
| 23 |
-
"eval_runtime":
|
| 24 |
-
"eval_samples_per_second": 5.
|
| 25 |
-
"eval_steps_per_second": 0.
|
| 26 |
"step": 100
|
| 27 |
},
|
| 28 |
{
|
| 29 |
"epoch": 0.2,
|
| 30 |
-
"grad_norm": 0.
|
| 31 |
"learning_rate": 0.003650943793925548,
|
| 32 |
-
"loss":
|
| 33 |
"step": 200
|
| 34 |
},
|
| 35 |
{
|
| 36 |
"epoch": 0.2,
|
| 37 |
-
"eval_accuracy": 0.
|
| 38 |
-
"eval_loss":
|
| 39 |
-
"eval_runtime":
|
| 40 |
-
"eval_samples_per_second":
|
| 41 |
-
"eval_steps_per_second": 0.
|
| 42 |
"step": 200
|
| 43 |
},
|
| 44 |
{
|
| 45 |
"epoch": 0.3,
|
| 46 |
-
"grad_norm": 0.
|
| 47 |
"learning_rate": 0.0032162636906540357,
|
| 48 |
-
"loss":
|
| 49 |
"step": 300
|
| 50 |
},
|
| 51 |
{
|
| 52 |
"epoch": 0.3,
|
| 53 |
-
"eval_accuracy": 0.
|
| 54 |
-
"eval_loss":
|
| 55 |
-
"eval_runtime":
|
| 56 |
-
"eval_samples_per_second":
|
| 57 |
-
"eval_steps_per_second": 0.
|
| 58 |
"step": 300
|
| 59 |
},
|
| 60 |
{
|
| 61 |
"epoch": 0.4,
|
| 62 |
-
"grad_norm": 0.
|
| 63 |
"learning_rate": 0.0026601302141692667,
|
| 64 |
-
"loss":
|
| 65 |
"step": 400
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"epoch": 0.4,
|
| 69 |
-
"eval_accuracy": 0.
|
| 70 |
-
"eval_loss":
|
| 71 |
-
"eval_runtime": 6.
|
| 72 |
-
"eval_samples_per_second": 18.
|
| 73 |
-
"eval_steps_per_second": 0.
|
| 74 |
"step": 400
|
| 75 |
},
|
| 76 |
{
|
| 77 |
"epoch": 0.5,
|
| 78 |
-
"grad_norm": 0.
|
| 79 |
"learning_rate": 0.002038077610206693,
|
| 80 |
-
"loss":
|
| 81 |
"step": 500
|
| 82 |
},
|
| 83 |
{
|
| 84 |
"epoch": 0.5,
|
| 85 |
-
"eval_accuracy": 0.
|
| 86 |
-
"eval_loss":
|
| 87 |
-
"eval_runtime": 6.
|
| 88 |
-
"eval_samples_per_second": 18.
|
| 89 |
-
"eval_steps_per_second": 0.
|
| 90 |
"step": 500
|
| 91 |
}
|
| 92 |
],
|
|
|
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
"epoch": 0.1,
|
| 14 |
+
"grad_norm": 1.0674346685409546,
|
| 15 |
"learning_rate": 0.003920764340590609,
|
| 16 |
+
"loss": 59.4722705078125,
|
| 17 |
"step": 100
|
| 18 |
},
|
| 19 |
{
|
| 20 |
"epoch": 0.1,
|
| 21 |
+
"eval_accuracy": 0.14505767350928642,
|
| 22 |
+
"eval_loss": 57.5521240234375,
|
| 23 |
+
"eval_runtime": 23.0144,
|
| 24 |
+
"eval_samples_per_second": 5.431,
|
| 25 |
+
"eval_steps_per_second": 0.174,
|
| 26 |
"step": 100
|
| 27 |
},
|
| 28 |
{
|
| 29 |
"epoch": 0.2,
|
| 30 |
+
"grad_norm": 0.8365403413772583,
|
| 31 |
"learning_rate": 0.003650943793925548,
|
| 32 |
+
"loss": 57.0307275390625,
|
| 33 |
"step": 200
|
| 34 |
},
|
| 35 |
{
|
| 36 |
"epoch": 0.2,
|
| 37 |
+
"eval_accuracy": 0.14682111436950146,
|
| 38 |
+
"eval_loss": 56.763153076171875,
|
| 39 |
+
"eval_runtime": 6.889,
|
| 40 |
+
"eval_samples_per_second": 18.145,
|
| 41 |
+
"eval_steps_per_second": 0.581,
|
| 42 |
"step": 200
|
| 43 |
},
|
| 44 |
{
|
| 45 |
"epoch": 0.3,
|
| 46 |
+
"grad_norm": 0.6420626044273376,
|
| 47 |
"learning_rate": 0.0032162636906540357,
|
| 48 |
+
"loss": 56.4287109375,
|
| 49 |
"step": 300
|
| 50 |
},
|
| 51 |
{
|
| 52 |
"epoch": 0.3,
|
| 53 |
+
"eval_accuracy": 0.15105865102639296,
|
| 54 |
+
"eval_loss": 56.29315185546875,
|
| 55 |
+
"eval_runtime": 7.4741,
|
| 56 |
+
"eval_samples_per_second": 16.724,
|
| 57 |
+
"eval_steps_per_second": 0.535,
|
| 58 |
"step": 300
|
| 59 |
},
|
| 60 |
{
|
| 61 |
"epoch": 0.4,
|
| 62 |
+
"grad_norm": 0.6022240519523621,
|
| 63 |
"learning_rate": 0.0026601302141692667,
|
| 64 |
+
"loss": 56.0290380859375,
|
| 65 |
"step": 400
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"epoch": 0.4,
|
| 69 |
+
"eval_accuracy": 0.1521642228739003,
|
| 70 |
+
"eval_loss": 55.97286605834961,
|
| 71 |
+
"eval_runtime": 6.9186,
|
| 72 |
+
"eval_samples_per_second": 18.067,
|
| 73 |
+
"eval_steps_per_second": 0.578,
|
| 74 |
"step": 400
|
| 75 |
},
|
| 76 |
{
|
| 77 |
"epoch": 0.5,
|
| 78 |
+
"grad_norm": 0.6007410287857056,
|
| 79 |
"learning_rate": 0.002038077610206693,
|
| 80 |
+
"loss": 56.0603759765625,
|
| 81 |
"step": 500
|
| 82 |
},
|
| 83 |
{
|
| 84 |
"epoch": 0.5,
|
| 85 |
+
"eval_accuracy": 0.15361681329423266,
|
| 86 |
+
"eval_loss": 55.698612213134766,
|
| 87 |
+
"eval_runtime": 6.7281,
|
| 88 |
+
"eval_samples_per_second": 18.579,
|
| 89 |
+
"eval_steps_per_second": 0.595,
|
| 90 |
"step": 500
|
| 91 |
}
|
| 92 |
],
|