Training in progress, step 100, checkpoint
Browse files- last-checkpoint/model.safetensors +1 -1
- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +26 -4
last-checkpoint/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 662430992
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b356afc6d082488f2555ef406378069504708e31271d6e086beee3c91f1f5c83
|
| 3 |
size 662430992
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 674384884
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:88b27e660c26757696366e20c37f079effc428107f2e9ffa74898f7f9b47361d
|
| 3 |
size 674384884
|
last-checkpoint/rng_state_0.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c63e25c94fd32bbfac74e77e235933e40e71931ccdab4688693badf62fc9d895
|
| 3 |
size 15024
|
last-checkpoint/rng_state_1.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42d5a9f1444725574e6e96d7460d2ae867d4c9d4d70a147ad1691b8ce1c4b0b8
|
| 3 |
size 15024
|
last-checkpoint/rng_state_2.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:86e844065e2ec1428132da97db98340b7020ef84b112f1abe984badffc1a1d8a
|
| 3 |
size 15024
|
last-checkpoint/rng_state_3.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 15024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3774449d16a3fcd1d29aba215bc5841e6b8bc77f3a4c81e9a133c2a3c87bcc8d
|
| 3 |
size 15024
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3a60c7d771c1fd156acee762fba03c724cb41829a3f71df370ecd1d20b134982
|
| 3 |
size 1064
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
-
"epoch":
|
| 5 |
"eval_steps": 20,
|
| 6 |
-
"global_step":
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
@@ -103,6 +103,28 @@
|
|
| 103 |
"eval_samples_per_second": 304.944,
|
| 104 |
"eval_steps_per_second": 3.251,
|
| 105 |
"step": 80
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 106 |
}
|
| 107 |
],
|
| 108 |
"logging_steps": 10,
|
|
@@ -117,12 +139,12 @@
|
|
| 117 |
"should_evaluate": false,
|
| 118 |
"should_log": false,
|
| 119 |
"should_save": true,
|
| 120 |
-
"should_training_stop":
|
| 121 |
},
|
| 122 |
"attributes": {}
|
| 123 |
}
|
| 124 |
},
|
| 125 |
-
"total_flos":
|
| 126 |
"train_batch_size": 24,
|
| 127 |
"trial_name": null,
|
| 128 |
"trial_params": null
|
|
|
|
| 1 |
{
|
| 2 |
"best_metric": null,
|
| 3 |
"best_model_checkpoint": null,
|
| 4 |
+
"epoch": 14.285714285714286,
|
| 5 |
"eval_steps": 20,
|
| 6 |
+
"global_step": 100,
|
| 7 |
"is_hyper_param_search": false,
|
| 8 |
"is_local_process_zero": true,
|
| 9 |
"is_world_process_zero": true,
|
|
|
|
| 103 |
"eval_samples_per_second": 304.944,
|
| 104 |
"eval_steps_per_second": 3.251,
|
| 105 |
"step": 80
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"epoch": 12.857142857142858,
|
| 109 |
+
"grad_norm": 2.046875,
|
| 110 |
+
"learning_rate": 5.418275829936537e-06,
|
| 111 |
+
"loss": 3.7097,
|
| 112 |
+
"step": 90
|
| 113 |
+
},
|
| 114 |
+
{
|
| 115 |
+
"epoch": 14.285714285714286,
|
| 116 |
+
"grad_norm": 1.6171875,
|
| 117 |
+
"learning_rate": 0.0,
|
| 118 |
+
"loss": 3.6947,
|
| 119 |
+
"step": 100
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"epoch": 14.285714285714286,
|
| 123 |
+
"eval_loss": 3.0370891094207764,
|
| 124 |
+
"eval_runtime": 5.0102,
|
| 125 |
+
"eval_samples_per_second": 299.59,
|
| 126 |
+
"eval_steps_per_second": 3.194,
|
| 127 |
+
"step": 100
|
| 128 |
}
|
| 129 |
],
|
| 130 |
"logging_steps": 10,
|
|
|
|
| 139 |
"should_evaluate": false,
|
| 140 |
"should_log": false,
|
| 141 |
"should_save": true,
|
| 142 |
+
"should_training_stop": true
|
| 143 |
},
|
| 144 |
"attributes": {}
|
| 145 |
}
|
| 146 |
},
|
| 147 |
+
"total_flos": 3.57855601360896e+16,
|
| 148 |
"train_batch_size": 24,
|
| 149 |
"trial_name": null,
|
| 150 |
"trial_params": null
|