Training in progress, step 800, checkpoint
Browse files- last-checkpoint/model.safetensors +1 -1
- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +20 -5
last-checkpoint/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 136000488
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f449ecaea05023856701cc6a7482fc4575a7d78905c0579ac3cd784669119041
|
| 3 |
size 136000488
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 268176506
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:64c3cdc6de437bb4cd73d82d3785754a70052aee12cbfbea25694372cfbd99cf
|
| 3 |
size 268176506
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1a2f28005ae080d5f74ee78733fad5c069f8b08635360a4876733f9e12e33dea
|
| 3 |
size 15024
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cb645dc2e4148818e3b8abafb0475d084098a031401b3762383d7bb8284eea3a
|
| 3 |
size 15024
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:adb042206d9d129a5489fd88e856b87f650944ac6e30a8e680a63423187eafd8
|
| 3 |
size 15024
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:38e70ba9ff6e871b9d42454856b4698e04796c285cbeeab7718443e202a72c0d
|
| 3 |
size 15024
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:60f77b738f0f28a29c728f4c7ce2972e380af6a14dc67f95a2dd3348d64628be
|
| 3 |
size 1064
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
-
"best_metric": 0.
|
| 3 |
-
"best_model_checkpoint": "mgh6/TCS_MLM/checkpoint-
|
| 4 |
-
"epoch":
|
| 5 |
"eval_steps": 100,
|
| 6 |
-
"global_step":
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
@@ -112,6 +112,21 @@
|
|
| 112 |
"eval_samples_per_second": 890.852,
|
| 113 |
"eval_steps_per_second": 3.6,
|
| 114 |
"step": 700
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 115 |
}
|
| 116 |
],
|
| 117 |
"logging_steps": 100,
|
|
@@ -140,7 +155,7 @@
|
|
| 140 |
"attributes": {}
|
| 141 |
}
|
| 142 |
},
|
| 143 |
-
"total_flos": 2.
|
| 144 |
"train_batch_size": 64,
|
| 145 |
"trial_name": null,
|
| 146 |
"trial_params": null
|
|
|
|
| 1 |
{
|
| 2 |
+
"best_metric": 0.9347544312477112,
|
| 3 |
+
"best_model_checkpoint": "mgh6/TCS_MLM/checkpoint-800",
|
| 4 |
+
"epoch": 1.07095046854083,
|
| 5 |
"eval_steps": 100,
|
| 6 |
+
"global_step": 800,
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
|
|
| 112 |
"eval_samples_per_second": 890.852,
|
| 113 |
"eval_steps_per_second": 3.6,
|
| 114 |
"step": 700
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"epoch": 1.07095046854083,
|
| 118 |
+
"grad_norm": 0.19669267535209656,
|
| 119 |
+
"learning_rate": 0.000892904953145917,
|
| 120 |
+
"loss": 0.9746,
|
| 121 |
+
"step": 800
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"epoch": 1.07095046854083,
|
| 125 |
+
"eval_loss": 0.9347544312477112,
|
| 126 |
+
"eval_runtime": 6.3997,
|
| 127 |
+
"eval_samples_per_second": 889.412,
|
| 128 |
+
"eval_steps_per_second": 3.594,
|
| 129 |
+
"step": 800
|
| 130 |
}
|
| 131 |
],
|
| 132 |
"logging_steps": 100,
|
|
|
|
| 155 |
"attributes": {}
|
| 156 |
}
|
| 157 |
},
|
| 158 |
+
"total_flos": 2.9049816612864e+16,
|
| 159 |
"train_batch_size": 64,
|
| 160 |
"trial_name": null,
|
| 161 |
"trial_params": null
|