Auto upload zain 2026-08-19T17:31:52.723720 (part 2)
Browse files- .gitattributes +1 -0
- zain/Activation/out/mlp-linear-9L_run/checkpoint-400/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-400/trainer_state.json +104 -104
- zain/Activation/out/mlp-linear-9L_run/checkpoint-400/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-500/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-500/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-500/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-500/trainer_state.json +129 -129
- zain/Activation/out/mlp-linear-9L_run/checkpoint-500/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-600/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-600/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-600/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-600/trainer_state.json +154 -154
- zain/Activation/out/mlp-linear-9L_run/checkpoint-600/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-700/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-700/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-700/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-700/trainer_state.json +179 -179
- zain/Activation/out/mlp-linear-9L_run/checkpoint-700/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-800/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-800/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-800/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-800/trainer_state.json +204 -204
- zain/Activation/out/mlp-linear-9L_run/checkpoint-800/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-900/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-900/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-900/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-9L_run/checkpoint-900/trainer_state.json +229 -229
- zain/Activation/out/mlp-linear-9L_run/checkpoint-900/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-9L_run/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-9L_run/training_log.jsonl +62 -152
- zain/Activation/out/sweep_summary.json +4 -4
- zain/Activation/wandb/debug-internal.log +32 -0
- zain/Activation/wandb/debug.log +5 -0
- zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log +9 -0
- zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log +9 -0
- zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log +9 -0
- zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log +9 -0
- zain/Activation/wandb/run-20260819_171215-2nil4rqn/logs/debug-core.log +12 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/config.yaml +435 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/output.log +121 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/requirements.txt +149 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-metadata.json +95 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-summary.json +1 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-core.log +76 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log +41 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log +28 -0
- zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb +3 -0
.gitattributes
CHANGED
|
@@ -143,3 +143,4 @@ zain/Activation/wandb/run-20260819_163845-74tq2syl/run-74tq2syl.wandb filter=lfs
|
|
| 143 |
zain/Activation/wandb/run-20260819_164553-44x03ghu/run-44x03ghu.wandb filter=lfs diff=lfs merge=lfs -text
|
| 144 |
zain/Activation/wandb/run-20260819_170854-zhg4u13t/run-zhg4u13t.wandb filter=lfs diff=lfs merge=lfs -text
|
| 145 |
zain/Activation/wandb/run-20260819_171215-2nil4rqn/run-2nil4rqn.wandb filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 143 |
zain/Activation/wandb/run-20260819_164553-44x03ghu/run-44x03ghu.wandb filter=lfs diff=lfs merge=lfs -text
|
| 144 |
zain/Activation/wandb/run-20260819_170854-zhg4u13t/run-zhg4u13t.wandb filter=lfs diff=lfs merge=lfs -text
|
| 145 |
zain/Activation/wandb/run-20260819_171215-2nil4rqn/run-2nil4rqn.wandb filter=lfs diff=lfs merge=lfs -text
|
| 146 |
+
zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb filter=lfs diff=lfs merge=lfs -text
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af1ea62c89448929159c1b05ebc75e926e6be85e7ce3551310793c351ba3ade1
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 400,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,180 +10,180 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
}
|
| 184 |
],
|
| 185 |
"logging_steps": 20,
|
| 186 |
-
"max_steps":
|
| 187 |
"num_input_tokens_seen": 0,
|
| 188 |
"num_train_epochs": 1,
|
| 189 |
"save_steps": 100,
|
|
@@ -199,8 +199,8 @@
|
|
| 199 |
"attributes": {}
|
| 200 |
}
|
| 201 |
},
|
| 202 |
-
"total_flos":
|
| 203 |
-
"train_batch_size":
|
| 204 |
"trial_name": null,
|
| 205 |
"trial_params": null
|
| 206 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.03370407819346141,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 400,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
}
|
| 184 |
],
|
| 185 |
"logging_steps": 20,
|
| 186 |
+
"max_steps": 1000,
|
| 187 |
"num_input_tokens_seen": 0,
|
| 188 |
"num_train_epochs": 1,
|
| 189 |
"save_steps": 100,
|
|
|
|
| 199 |
"attributes": {}
|
| 200 |
}
|
| 201 |
},
|
| 202 |
+
"total_flos": 145194221568000.0,
|
| 203 |
+
"train_batch_size": 80,
|
| 204 |
"trial_name": null,
|
| 205 |
"trial_params": null
|
| 206 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5712d6f569182e98a2720f28f3ef489d602a3c6473ce0b6e3c4f1f7907e3c8cf
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8068282
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:63b2d05dc2871dc45cdfc3ecc136218a493535e4bde6da2a670418746fe34818
|
| 3 |
size 8068282
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b11eb900cb7af9ecb8797771dfa83159f999b3d7b4f3c5b2ff6e19c51da8de4a
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,223 +10,223 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm":
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm":
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm":
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"eval_loss":
|
| 222 |
-
"eval_runtime":
|
| 223 |
-
"eval_samples_per_second":
|
| 224 |
-
"eval_steps_per_second": 0.
|
| 225 |
"step": 500
|
| 226 |
}
|
| 227 |
],
|
| 228 |
"logging_steps": 20,
|
| 229 |
-
"max_steps":
|
| 230 |
"num_input_tokens_seen": 0,
|
| 231 |
"num_train_epochs": 1,
|
| 232 |
"save_steps": 100,
|
|
@@ -242,8 +242,8 @@
|
|
| 242 |
"attributes": {}
|
| 243 |
}
|
| 244 |
},
|
| 245 |
-
"total_flos":
|
| 246 |
-
"train_batch_size":
|
| 247 |
"trial_name": null,
|
| 248 |
"trial_params": null
|
| 249 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.04213009774182676,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
+
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 0.7109375,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.3047794342041015,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
+
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.71875,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.258700942993164,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
+
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.73046875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.229313278198242,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
+
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.703125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.202067565917969,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
+
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.7265625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.163772201538086,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
+
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.157355308532715,
|
| 222 |
+
"eval_runtime": 7.9503,
|
| 223 |
+
"eval_samples_per_second": 1198.315,
|
| 224 |
+
"eval_steps_per_second": 0.88,
|
| 225 |
"step": 500
|
| 226 |
}
|
| 227 |
],
|
| 228 |
"logging_steps": 20,
|
| 229 |
+
"max_steps": 1000,
|
| 230 |
"num_input_tokens_seen": 0,
|
| 231 |
"num_train_epochs": 1,
|
| 232 |
"save_steps": 100,
|
|
|
|
| 242 |
"attributes": {}
|
| 243 |
}
|
| 244 |
},
|
| 245 |
+
"total_flos": 181492776960000.0,
|
| 246 |
+
"train_batch_size": 80,
|
| 247 |
"trial_name": null,
|
| 248 |
"trial_params": null
|
| 249 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fdd583528bce16f5f52bc31609f36b44fc4dc0de87540acf616f1442cb07fde9
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8068282
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8fe1053b1cf05e779ceab88f2312865667a8cb4999b4439b1a72908989024233
|
| 3 |
size 8068282
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:31ff2c480a5c30c49d6c8b2f5ae72d7dc39d2d3f7060f88a23705954d049a506
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 600,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,266 +10,266 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm":
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm":
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm":
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"eval_loss":
|
| 222 |
-
"eval_runtime":
|
| 223 |
-
"eval_samples_per_second":
|
| 224 |
-
"eval_steps_per_second": 0.
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm":
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm":
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm":
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 8.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
}
|
| 270 |
],
|
| 271 |
"logging_steps": 20,
|
| 272 |
-
"max_steps":
|
| 273 |
"num_input_tokens_seen": 0,
|
| 274 |
"num_train_epochs": 1,
|
| 275 |
"save_steps": 100,
|
|
@@ -285,8 +285,8 @@
|
|
| 285 |
"attributes": {}
|
| 286 |
}
|
| 287 |
},
|
| 288 |
-
"total_flos":
|
| 289 |
-
"train_batch_size":
|
| 290 |
"trial_name": null,
|
| 291 |
"trial_params": null
|
| 292 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.05055611729019211,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 600,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
+
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 0.7109375,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.3047794342041015,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
+
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.71875,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.258700942993164,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
+
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.73046875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.229313278198242,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
+
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.703125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.202067565917969,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
+
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.7265625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.163772201538086,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
+
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.157355308532715,
|
| 222 |
+
"eval_runtime": 7.9503,
|
| 223 |
+
"eval_samples_per_second": 1198.315,
|
| 224 |
+
"eval_steps_per_second": 0.88,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
+
"epoch": 0.043815301651499834,
|
| 229 |
+
"grad_norm": 0.6953125,
|
| 230 |
+
"learning_rate": 0.001,
|
| 231 |
+
"loss": 3.1448848724365233,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
+
"epoch": 0.0455005055611729,
|
| 236 |
+
"grad_norm": 0.828125,
|
| 237 |
+
"learning_rate": 0.001,
|
| 238 |
+
"loss": 3.103527069091797,
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
+
"epoch": 0.04718570947084597,
|
| 243 |
+
"grad_norm": 0.7578125,
|
| 244 |
+
"learning_rate": 0.001,
|
| 245 |
+
"loss": 3.08404541015625,
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
+
"epoch": 0.04887091338051904,
|
| 250 |
+
"grad_norm": 0.76953125,
|
| 251 |
+
"learning_rate": 0.001,
|
| 252 |
+
"loss": 3.0501741409301757,
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
+
"epoch": 0.05055611729019211,
|
| 257 |
+
"grad_norm": 0.96875,
|
| 258 |
+
"learning_rate": 0.001,
|
| 259 |
+
"loss": 3.0376760482788088,
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
+
"epoch": 0.05055611729019211,
|
| 264 |
+
"eval_loss": 3.0310890674591064,
|
| 265 |
+
"eval_runtime": 8.1195,
|
| 266 |
+
"eval_samples_per_second": 1173.346,
|
| 267 |
+
"eval_steps_per_second": 0.862,
|
| 268 |
"step": 600
|
| 269 |
}
|
| 270 |
],
|
| 271 |
"logging_steps": 20,
|
| 272 |
+
"max_steps": 1000,
|
| 273 |
"num_input_tokens_seen": 0,
|
| 274 |
"num_train_epochs": 1,
|
| 275 |
"save_steps": 100,
|
|
|
|
| 285 |
"attributes": {}
|
| 286 |
}
|
| 287 |
},
|
| 288 |
+
"total_flos": 217791332352000.0,
|
| 289 |
+
"train_batch_size": 80,
|
| 290 |
"trial_name": null,
|
| 291 |
"trial_params": null
|
| 292 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:97deb4d2fe5491f671435ead6997073978dc5974641a5f915ea901926274c1a1
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8068282
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1f5e62abb7b4309d85d8e7e3f37e3449916b0bac155e62f602e3bbd0aa0a7951
|
| 3 |
size 8068282
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dd7c397523408b6e567c3990f49ad726aa8ff2c476b37da2c996606b1b8ddb75
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 700,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,309 +10,309 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm":
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm":
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm":
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"eval_loss":
|
| 222 |
-
"eval_runtime":
|
| 223 |
-
"eval_samples_per_second":
|
| 224 |
-
"eval_steps_per_second": 0.
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm":
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm":
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm":
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 8.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm":
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm":
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss":
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm":
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss":
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm":
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss":
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
-
"epoch": 0.
|
| 307 |
-
"eval_loss":
|
| 308 |
-
"eval_runtime":
|
| 309 |
-
"eval_samples_per_second":
|
| 310 |
-
"eval_steps_per_second": 0.
|
| 311 |
"step": 700
|
| 312 |
}
|
| 313 |
],
|
| 314 |
"logging_steps": 20,
|
| 315 |
-
"max_steps":
|
| 316 |
"num_input_tokens_seen": 0,
|
| 317 |
"num_train_epochs": 1,
|
| 318 |
"save_steps": 100,
|
|
@@ -328,8 +328,8 @@
|
|
| 328 |
"attributes": {}
|
| 329 |
}
|
| 330 |
},
|
| 331 |
-
"total_flos":
|
| 332 |
-
"train_batch_size":
|
| 333 |
"trial_name": null,
|
| 334 |
"trial_params": null
|
| 335 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.058982136838557464,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 700,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
+
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 0.7109375,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.3047794342041015,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
+
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.71875,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.258700942993164,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
+
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.73046875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.229313278198242,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
+
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.703125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.202067565917969,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
+
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.7265625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.163772201538086,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
+
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.157355308532715,
|
| 222 |
+
"eval_runtime": 7.9503,
|
| 223 |
+
"eval_samples_per_second": 1198.315,
|
| 224 |
+
"eval_steps_per_second": 0.88,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
+
"epoch": 0.043815301651499834,
|
| 229 |
+
"grad_norm": 0.6953125,
|
| 230 |
+
"learning_rate": 0.001,
|
| 231 |
+
"loss": 3.1448848724365233,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
+
"epoch": 0.0455005055611729,
|
| 236 |
+
"grad_norm": 0.828125,
|
| 237 |
+
"learning_rate": 0.001,
|
| 238 |
+
"loss": 3.103527069091797,
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
+
"epoch": 0.04718570947084597,
|
| 243 |
+
"grad_norm": 0.7578125,
|
| 244 |
+
"learning_rate": 0.001,
|
| 245 |
+
"loss": 3.08404541015625,
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
+
"epoch": 0.04887091338051904,
|
| 250 |
+
"grad_norm": 0.76953125,
|
| 251 |
+
"learning_rate": 0.001,
|
| 252 |
+
"loss": 3.0501741409301757,
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
+
"epoch": 0.05055611729019211,
|
| 257 |
+
"grad_norm": 0.96875,
|
| 258 |
+
"learning_rate": 0.001,
|
| 259 |
+
"loss": 3.0376760482788088,
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
+
"epoch": 0.05055611729019211,
|
| 264 |
+
"eval_loss": 3.0310890674591064,
|
| 265 |
+
"eval_runtime": 8.1195,
|
| 266 |
+
"eval_samples_per_second": 1173.346,
|
| 267 |
+
"eval_steps_per_second": 0.862,
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
+
"epoch": 0.05224132119986518,
|
| 272 |
+
"grad_norm": 0.82421875,
|
| 273 |
+
"learning_rate": 0.001,
|
| 274 |
+
"loss": 3.0314205169677733,
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
+
"epoch": 0.053926525109538256,
|
| 279 |
+
"grad_norm": 0.640625,
|
| 280 |
+
"learning_rate": 0.001,
|
| 281 |
+
"loss": 3.0014928817749023,
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
+
"epoch": 0.055611729019211326,
|
| 286 |
+
"grad_norm": 0.70703125,
|
| 287 |
+
"learning_rate": 0.001,
|
| 288 |
+
"loss": 2.9963293075561523,
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
+
"epoch": 0.057296932928884395,
|
| 293 |
+
"grad_norm": 0.7109375,
|
| 294 |
+
"learning_rate": 0.001,
|
| 295 |
+
"loss": 2.9568761825561523,
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
+
"epoch": 0.058982136838557464,
|
| 300 |
+
"grad_norm": 0.69921875,
|
| 301 |
+
"learning_rate": 0.001,
|
| 302 |
+
"loss": 2.9395275115966797,
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
+
"epoch": 0.058982136838557464,
|
| 307 |
+
"eval_loss": 2.939429759979248,
|
| 308 |
+
"eval_runtime": 7.975,
|
| 309 |
+
"eval_samples_per_second": 1194.602,
|
| 310 |
+
"eval_steps_per_second": 0.878,
|
| 311 |
"step": 700
|
| 312 |
}
|
| 313 |
],
|
| 314 |
"logging_steps": 20,
|
| 315 |
+
"max_steps": 1000,
|
| 316 |
"num_input_tokens_seen": 0,
|
| 317 |
"num_train_epochs": 1,
|
| 318 |
"save_steps": 100,
|
|
|
|
| 328 |
"attributes": {}
|
| 329 |
}
|
| 330 |
},
|
| 331 |
+
"total_flos": 254089887744000.0,
|
| 332 |
+
"train_batch_size": 80,
|
| 333 |
"trial_name": null,
|
| 334 |
"trial_params": null
|
| 335 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9c337edbdecc4bf136ef452013ba3dadf0cb21c44e46ffe85d9ad094448bd625
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8068282
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:99e89d2c66df90ceb6b05584fce25f5cd08caf172652dfb64c617b95863d99bf
|
| 3 |
size 8068282
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:71e4310b54de76d098945960239794c83dba2710505618327c0dcd1468c3497f
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 800,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,352 +10,352 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm":
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm":
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm":
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"eval_loss":
|
| 222 |
-
"eval_runtime":
|
| 223 |
-
"eval_samples_per_second":
|
| 224 |
-
"eval_steps_per_second": 0.
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm":
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm":
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm":
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 8.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm":
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm":
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss":
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm":
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss":
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm":
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss":
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
-
"epoch": 0.
|
| 307 |
-
"eval_loss":
|
| 308 |
-
"eval_runtime":
|
| 309 |
-
"eval_samples_per_second":
|
| 310 |
-
"eval_steps_per_second": 0.
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
-
"epoch": 0.
|
| 315 |
-
"grad_norm":
|
| 316 |
-
"learning_rate": 0.
|
| 317 |
-
"loss":
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
-
"epoch": 0.
|
| 322 |
-
"grad_norm":
|
| 323 |
-
"learning_rate": 0.
|
| 324 |
-
"loss":
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
-
"epoch": 0.
|
| 329 |
-
"grad_norm":
|
| 330 |
-
"learning_rate": 0.
|
| 331 |
-
"loss":
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
-
"epoch": 0.
|
| 336 |
-
"grad_norm":
|
| 337 |
-
"learning_rate": 0.
|
| 338 |
-
"loss":
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
-
"epoch": 0.
|
| 343 |
-
"grad_norm":
|
| 344 |
-
"learning_rate": 0.
|
| 345 |
-
"loss":
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
-
"epoch": 0.
|
| 350 |
-
"eval_loss":
|
| 351 |
-
"eval_runtime": 8.
|
| 352 |
-
"eval_samples_per_second":
|
| 353 |
-
"eval_steps_per_second": 0.
|
| 354 |
"step": 800
|
| 355 |
}
|
| 356 |
],
|
| 357 |
"logging_steps": 20,
|
| 358 |
-
"max_steps":
|
| 359 |
"num_input_tokens_seen": 0,
|
| 360 |
"num_train_epochs": 1,
|
| 361 |
"save_steps": 100,
|
|
@@ -371,8 +371,8 @@
|
|
| 371 |
"attributes": {}
|
| 372 |
}
|
| 373 |
},
|
| 374 |
-
"total_flos":
|
| 375 |
-
"train_batch_size":
|
| 376 |
"trial_name": null,
|
| 377 |
"trial_params": null
|
| 378 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.06740815638692282,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 800,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
+
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 0.7109375,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.3047794342041015,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
+
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.71875,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.258700942993164,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
+
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.73046875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.229313278198242,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
+
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.703125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.202067565917969,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
+
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.7265625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.163772201538086,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
+
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.157355308532715,
|
| 222 |
+
"eval_runtime": 7.9503,
|
| 223 |
+
"eval_samples_per_second": 1198.315,
|
| 224 |
+
"eval_steps_per_second": 0.88,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
+
"epoch": 0.043815301651499834,
|
| 229 |
+
"grad_norm": 0.6953125,
|
| 230 |
+
"learning_rate": 0.001,
|
| 231 |
+
"loss": 3.1448848724365233,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
+
"epoch": 0.0455005055611729,
|
| 236 |
+
"grad_norm": 0.828125,
|
| 237 |
+
"learning_rate": 0.001,
|
| 238 |
+
"loss": 3.103527069091797,
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
+
"epoch": 0.04718570947084597,
|
| 243 |
+
"grad_norm": 0.7578125,
|
| 244 |
+
"learning_rate": 0.001,
|
| 245 |
+
"loss": 3.08404541015625,
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
+
"epoch": 0.04887091338051904,
|
| 250 |
+
"grad_norm": 0.76953125,
|
| 251 |
+
"learning_rate": 0.001,
|
| 252 |
+
"loss": 3.0501741409301757,
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
+
"epoch": 0.05055611729019211,
|
| 257 |
+
"grad_norm": 0.96875,
|
| 258 |
+
"learning_rate": 0.001,
|
| 259 |
+
"loss": 3.0376760482788088,
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
+
"epoch": 0.05055611729019211,
|
| 264 |
+
"eval_loss": 3.0310890674591064,
|
| 265 |
+
"eval_runtime": 8.1195,
|
| 266 |
+
"eval_samples_per_second": 1173.346,
|
| 267 |
+
"eval_steps_per_second": 0.862,
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
+
"epoch": 0.05224132119986518,
|
| 272 |
+
"grad_norm": 0.82421875,
|
| 273 |
+
"learning_rate": 0.001,
|
| 274 |
+
"loss": 3.0314205169677733,
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
+
"epoch": 0.053926525109538256,
|
| 279 |
+
"grad_norm": 0.640625,
|
| 280 |
+
"learning_rate": 0.001,
|
| 281 |
+
"loss": 3.0014928817749023,
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
+
"epoch": 0.055611729019211326,
|
| 286 |
+
"grad_norm": 0.70703125,
|
| 287 |
+
"learning_rate": 0.001,
|
| 288 |
+
"loss": 2.9963293075561523,
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
+
"epoch": 0.057296932928884395,
|
| 293 |
+
"grad_norm": 0.7109375,
|
| 294 |
+
"learning_rate": 0.001,
|
| 295 |
+
"loss": 2.9568761825561523,
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
+
"epoch": 0.058982136838557464,
|
| 300 |
+
"grad_norm": 0.69921875,
|
| 301 |
+
"learning_rate": 0.001,
|
| 302 |
+
"loss": 2.9395275115966797,
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
+
"epoch": 0.058982136838557464,
|
| 307 |
+
"eval_loss": 2.939429759979248,
|
| 308 |
+
"eval_runtime": 7.975,
|
| 309 |
+
"eval_samples_per_second": 1194.602,
|
| 310 |
+
"eval_steps_per_second": 0.878,
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
+
"epoch": 0.06066734074823053,
|
| 315 |
+
"grad_norm": 0.6484375,
|
| 316 |
+
"learning_rate": 0.001,
|
| 317 |
+
"loss": 2.922671890258789,
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
+
"epoch": 0.06235254465790361,
|
| 322 |
+
"grad_norm": 0.95703125,
|
| 323 |
+
"learning_rate": 0.001,
|
| 324 |
+
"loss": 2.9145633697509767,
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
+
"epoch": 0.06403774856757667,
|
| 329 |
+
"grad_norm": 0.7109375,
|
| 330 |
+
"learning_rate": 0.001,
|
| 331 |
+
"loss": 2.8876668930053713,
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
+
"epoch": 0.06572295247724974,
|
| 336 |
+
"grad_norm": 0.69140625,
|
| 337 |
+
"learning_rate": 0.001,
|
| 338 |
+
"loss": 2.8692466735839846,
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
+
"epoch": 0.06740815638692282,
|
| 343 |
+
"grad_norm": 0.68359375,
|
| 344 |
+
"learning_rate": 0.001,
|
| 345 |
+
"loss": 2.8788633346557617,
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
+
"epoch": 0.06740815638692282,
|
| 350 |
+
"eval_loss": 2.8632190227508545,
|
| 351 |
+
"eval_runtime": 8.2773,
|
| 352 |
+
"eval_samples_per_second": 1150.973,
|
| 353 |
+
"eval_steps_per_second": 0.846,
|
| 354 |
"step": 800
|
| 355 |
}
|
| 356 |
],
|
| 357 |
"logging_steps": 20,
|
| 358 |
+
"max_steps": 1000,
|
| 359 |
"num_input_tokens_seen": 0,
|
| 360 |
"num_train_epochs": 1,
|
| 361 |
"save_steps": 100,
|
|
|
|
| 371 |
"attributes": {}
|
| 372 |
}
|
| 373 |
},
|
| 374 |
+
"total_flos": 290388443136000.0,
|
| 375 |
+
"train_batch_size": 80,
|
| 376 |
"trial_name": null,
|
| 377 |
"trial_params": null
|
| 378 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:82e06f7126a42b84896bee7f96709259393d400ad27e92239a7c07ad6c0e0fd4
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 8068282
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:412b1a58a5c56f4c0d4c86023f862cac2da0249009c2d10b79e1506090879e57
|
| 3 |
size 8068282
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:819e7677327ec819a829273ee6bf375a38913eb519826c73009d0dd3825f5480
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/trainer_state.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 900,
|
| 8 |
"is_hyper_param_search": false,
|
|
@@ -10,395 +10,395 @@
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
-
"epoch": 0.
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
-
"epoch": 0.
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate":
|
| 23 |
-
"loss":
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
-
"epoch": 0.
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate":
|
| 30 |
-
"loss":
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
-
"epoch": 0.
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate":
|
| 37 |
-
"loss":
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
-
"epoch": 0.
|
| 42 |
-
"grad_norm": 1.
|
| 43 |
-
"learning_rate":
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
-
"epoch": 0.
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime":
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
-
"epoch": 0.
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate":
|
| 59 |
-
"loss":
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
-
"epoch": 0.
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate":
|
| 66 |
-
"loss":
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
-
"epoch": 0.
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate":
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
-
"epoch": 0.
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss":
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
-
"epoch": 0.
|
| 85 |
-
"grad_norm":
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss":
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
-
"epoch": 0.
|
| 92 |
-
"eval_loss":
|
| 93 |
-
"eval_runtime":
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
-
"epoch": 0.
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss":
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
-
"epoch": 0.
|
| 107 |
-
"grad_norm":
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
-
"epoch": 0.
|
| 114 |
-
"grad_norm":
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
-
"epoch": 0.
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss":
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
-
"epoch": 0.
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss":
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"epoch": 0.
|
| 135 |
-
"eval_loss":
|
| 136 |
-
"eval_runtime":
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
-
"epoch": 0.
|
| 143 |
-
"grad_norm":
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss":
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
-
"epoch": 0.
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss":
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
-
"epoch": 0.
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss":
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
-
"epoch": 0.
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss":
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
-
"epoch": 0.
|
| 171 |
-
"grad_norm":
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss":
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
-
"epoch": 0.
|
| 178 |
-
"eval_loss":
|
| 179 |
-
"eval_runtime": 8.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
-
"epoch": 0.
|
| 186 |
-
"grad_norm":
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss":
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
-
"epoch": 0.
|
| 193 |
-
"grad_norm":
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss":
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
-
"epoch": 0.
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss":
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
-
"epoch": 0.
|
| 207 |
-
"grad_norm":
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss":
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
-
"epoch": 0.
|
| 214 |
-
"grad_norm":
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss":
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
-
"epoch": 0.
|
| 221 |
-
"eval_loss":
|
| 222 |
-
"eval_runtime":
|
| 223 |
-
"eval_samples_per_second":
|
| 224 |
-
"eval_steps_per_second": 0.
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
-
"epoch": 0.
|
| 229 |
-
"grad_norm":
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss":
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
-
"epoch": 0.
|
| 236 |
-
"grad_norm":
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss":
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
-
"epoch": 0.
|
| 243 |
-
"grad_norm":
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss":
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
-
"epoch": 0.
|
| 250 |
-
"grad_norm":
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss":
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
-
"epoch": 0.
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss":
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
-
"epoch": 0.
|
| 264 |
-
"eval_loss":
|
| 265 |
-
"eval_runtime": 8.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
-
"epoch": 0.
|
| 272 |
-
"grad_norm":
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss":
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
-
"epoch": 0.
|
| 279 |
-
"grad_norm":
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss":
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
-
"epoch": 0.
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss":
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"epoch": 0.
|
| 293 |
-
"grad_norm":
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss":
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
-
"epoch": 0.
|
| 300 |
-
"grad_norm":
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss":
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
-
"epoch": 0.
|
| 307 |
-
"eval_loss":
|
| 308 |
-
"eval_runtime":
|
| 309 |
-
"eval_samples_per_second":
|
| 310 |
-
"eval_steps_per_second": 0.
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
-
"epoch": 0.
|
| 315 |
-
"grad_norm":
|
| 316 |
-
"learning_rate": 0.
|
| 317 |
-
"loss":
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
-
"epoch": 0.
|
| 322 |
-
"grad_norm":
|
| 323 |
-
"learning_rate": 0.
|
| 324 |
-
"loss":
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
-
"epoch": 0.
|
| 329 |
-
"grad_norm":
|
| 330 |
-
"learning_rate": 0.
|
| 331 |
-
"loss":
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
-
"epoch": 0.
|
| 336 |
-
"grad_norm":
|
| 337 |
-
"learning_rate": 0.
|
| 338 |
-
"loss":
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
-
"epoch": 0.
|
| 343 |
-
"grad_norm":
|
| 344 |
-
"learning_rate": 0.
|
| 345 |
-
"loss":
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
-
"epoch": 0.
|
| 350 |
-
"eval_loss":
|
| 351 |
-
"eval_runtime": 8.
|
| 352 |
-
"eval_samples_per_second":
|
| 353 |
-
"eval_steps_per_second": 0.
|
| 354 |
"step": 800
|
| 355 |
},
|
| 356 |
{
|
| 357 |
-
"epoch": 0.
|
| 358 |
-
"grad_norm":
|
| 359 |
-
"learning_rate": 0.
|
| 360 |
-
"loss":
|
| 361 |
"step": 820
|
| 362 |
},
|
| 363 |
{
|
| 364 |
-
"epoch": 0.
|
| 365 |
-
"grad_norm":
|
| 366 |
-
"learning_rate": 0.
|
| 367 |
-
"loss":
|
| 368 |
"step": 840
|
| 369 |
},
|
| 370 |
{
|
| 371 |
-
"epoch": 0.
|
| 372 |
-
"grad_norm":
|
| 373 |
-
"learning_rate": 0.
|
| 374 |
-
"loss":
|
| 375 |
"step": 860
|
| 376 |
},
|
| 377 |
{
|
| 378 |
-
"epoch": 0.
|
| 379 |
-
"grad_norm":
|
| 380 |
-
"learning_rate": 0.
|
| 381 |
-
"loss":
|
| 382 |
"step": 880
|
| 383 |
},
|
| 384 |
{
|
| 385 |
-
"epoch": 0.
|
| 386 |
-
"grad_norm":
|
| 387 |
-
"learning_rate": 0.
|
| 388 |
-
"loss":
|
| 389 |
"step": 900
|
| 390 |
},
|
| 391 |
{
|
| 392 |
-
"epoch": 0.
|
| 393 |
-
"eval_loss":
|
| 394 |
-
"eval_runtime":
|
| 395 |
-
"eval_samples_per_second":
|
| 396 |
-
"eval_steps_per_second": 0.
|
| 397 |
"step": 900
|
| 398 |
}
|
| 399 |
],
|
| 400 |
"logging_steps": 20,
|
| 401 |
-
"max_steps":
|
| 402 |
"num_input_tokens_seen": 0,
|
| 403 |
"num_train_epochs": 1,
|
| 404 |
"save_steps": 100,
|
|
@@ -414,8 +414,8 @@
|
|
| 414 |
"attributes": {}
|
| 415 |
}
|
| 416 |
},
|
| 417 |
-
"total_flos":
|
| 418 |
-
"train_batch_size":
|
| 419 |
"trial_name": null,
|
| 420 |
"trial_params": null
|
| 421 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.07583417593528817,
|
| 6 |
"eval_steps": 100,
|
| 7 |
"global_step": 900,
|
| 8 |
"is_hyper_param_search": false,
|
|
|
|
| 10 |
"is_world_process_zero": true,
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
+
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.2421875,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.200721740722656,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
+
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.2421875,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.764720153808594,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
+
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.2265625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.160160064697266,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
+
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.2109375,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.480591583251953,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
+
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.0078125,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.919136428833008,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
+
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.634258270263672,
|
| 50 |
+
"eval_runtime": 7.97,
|
| 51 |
+
"eval_samples_per_second": 1195.356,
|
| 52 |
+
"eval_steps_per_second": 0.878,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
+
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.171875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.412916946411133,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
+
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.515625,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.0327880859375,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
+
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.9765625,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.69476089477539,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
+
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.8828125,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.4246673583984375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
+
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 0.87890625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.204695129394532,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
+
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.133424282073975,
|
| 93 |
+
"eval_runtime": 7.9329,
|
| 94 |
+
"eval_samples_per_second": 1200.942,
|
| 95 |
+
"eval_steps_per_second": 0.882,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
+
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 0.6875,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.048733520507812,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
+
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.73828125,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.9190834045410154,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
+
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.68359375,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7833847045898437,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
+
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.82421875,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.700960159301758,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
+
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 0.984375,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6359439849853517,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5975606441497803,
|
| 136 |
+
"eval_runtime": 7.973,
|
| 137 |
+
"eval_samples_per_second": 1194.91,
|
| 138 |
+
"eval_steps_per_second": 0.878,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
+
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.80859375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5393722534179686,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
+
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.72265625,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.492219924926758,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
+
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 0.8203125,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4430694580078125,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
+
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.8203125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.3989883422851563,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
+
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 0.6484375,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.341664123535156,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
+
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3295202255249023,
|
| 179 |
+
"eval_runtime": 8.1687,
|
| 180 |
+
"eval_samples_per_second": 1166.287,
|
| 181 |
+
"eval_steps_per_second": 0.857,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
+
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 0.7109375,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.3047794342041015,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
+
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.71875,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.258700942993164,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
+
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.73046875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.229313278198242,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
+
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.703125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.202067565917969,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
+
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.7265625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.163772201538086,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
+
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.157355308532715,
|
| 222 |
+
"eval_runtime": 7.9503,
|
| 223 |
+
"eval_samples_per_second": 1198.315,
|
| 224 |
+
"eval_steps_per_second": 0.88,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
+
"epoch": 0.043815301651499834,
|
| 229 |
+
"grad_norm": 0.6953125,
|
| 230 |
+
"learning_rate": 0.001,
|
| 231 |
+
"loss": 3.1448848724365233,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
+
"epoch": 0.0455005055611729,
|
| 236 |
+
"grad_norm": 0.828125,
|
| 237 |
+
"learning_rate": 0.001,
|
| 238 |
+
"loss": 3.103527069091797,
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
+
"epoch": 0.04718570947084597,
|
| 243 |
+
"grad_norm": 0.7578125,
|
| 244 |
+
"learning_rate": 0.001,
|
| 245 |
+
"loss": 3.08404541015625,
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
+
"epoch": 0.04887091338051904,
|
| 250 |
+
"grad_norm": 0.76953125,
|
| 251 |
+
"learning_rate": 0.001,
|
| 252 |
+
"loss": 3.0501741409301757,
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
+
"epoch": 0.05055611729019211,
|
| 257 |
+
"grad_norm": 0.96875,
|
| 258 |
+
"learning_rate": 0.001,
|
| 259 |
+
"loss": 3.0376760482788088,
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
+
"epoch": 0.05055611729019211,
|
| 264 |
+
"eval_loss": 3.0310890674591064,
|
| 265 |
+
"eval_runtime": 8.1195,
|
| 266 |
+
"eval_samples_per_second": 1173.346,
|
| 267 |
+
"eval_steps_per_second": 0.862,
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
+
"epoch": 0.05224132119986518,
|
| 272 |
+
"grad_norm": 0.82421875,
|
| 273 |
+
"learning_rate": 0.001,
|
| 274 |
+
"loss": 3.0314205169677733,
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
+
"epoch": 0.053926525109538256,
|
| 279 |
+
"grad_norm": 0.640625,
|
| 280 |
+
"learning_rate": 0.001,
|
| 281 |
+
"loss": 3.0014928817749023,
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
+
"epoch": 0.055611729019211326,
|
| 286 |
+
"grad_norm": 0.70703125,
|
| 287 |
+
"learning_rate": 0.001,
|
| 288 |
+
"loss": 2.9963293075561523,
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
+
"epoch": 0.057296932928884395,
|
| 293 |
+
"grad_norm": 0.7109375,
|
| 294 |
+
"learning_rate": 0.001,
|
| 295 |
+
"loss": 2.9568761825561523,
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
+
"epoch": 0.058982136838557464,
|
| 300 |
+
"grad_norm": 0.69921875,
|
| 301 |
+
"learning_rate": 0.001,
|
| 302 |
+
"loss": 2.9395275115966797,
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
+
"epoch": 0.058982136838557464,
|
| 307 |
+
"eval_loss": 2.939429759979248,
|
| 308 |
+
"eval_runtime": 7.975,
|
| 309 |
+
"eval_samples_per_second": 1194.602,
|
| 310 |
+
"eval_steps_per_second": 0.878,
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
+
"epoch": 0.06066734074823053,
|
| 315 |
+
"grad_norm": 0.6484375,
|
| 316 |
+
"learning_rate": 0.001,
|
| 317 |
+
"loss": 2.922671890258789,
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
+
"epoch": 0.06235254465790361,
|
| 322 |
+
"grad_norm": 0.95703125,
|
| 323 |
+
"learning_rate": 0.001,
|
| 324 |
+
"loss": 2.9145633697509767,
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
+
"epoch": 0.06403774856757667,
|
| 329 |
+
"grad_norm": 0.7109375,
|
| 330 |
+
"learning_rate": 0.001,
|
| 331 |
+
"loss": 2.8876668930053713,
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
+
"epoch": 0.06572295247724974,
|
| 336 |
+
"grad_norm": 0.69140625,
|
| 337 |
+
"learning_rate": 0.001,
|
| 338 |
+
"loss": 2.8692466735839846,
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
+
"epoch": 0.06740815638692282,
|
| 343 |
+
"grad_norm": 0.68359375,
|
| 344 |
+
"learning_rate": 0.001,
|
| 345 |
+
"loss": 2.8788633346557617,
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
+
"epoch": 0.06740815638692282,
|
| 350 |
+
"eval_loss": 2.8632190227508545,
|
| 351 |
+
"eval_runtime": 8.2773,
|
| 352 |
+
"eval_samples_per_second": 1150.973,
|
| 353 |
+
"eval_steps_per_second": 0.846,
|
| 354 |
"step": 800
|
| 355 |
},
|
| 356 |
{
|
| 357 |
+
"epoch": 0.0690933602965959,
|
| 358 |
+
"grad_norm": 0.6796875,
|
| 359 |
+
"learning_rate": 0.001,
|
| 360 |
+
"loss": 2.8496471405029298,
|
| 361 |
"step": 820
|
| 362 |
},
|
| 363 |
{
|
| 364 |
+
"epoch": 0.07077856420626896,
|
| 365 |
+
"grad_norm": 0.62109375,
|
| 366 |
+
"learning_rate": 0.001,
|
| 367 |
+
"loss": 2.847636604309082,
|
| 368 |
"step": 840
|
| 369 |
},
|
| 370 |
{
|
| 371 |
+
"epoch": 0.07246376811594203,
|
| 372 |
+
"grad_norm": 0.73046875,
|
| 373 |
+
"learning_rate": 0.001,
|
| 374 |
+
"loss": 2.8339916229248048,
|
| 375 |
"step": 860
|
| 376 |
},
|
| 377 |
{
|
| 378 |
+
"epoch": 0.0741489720256151,
|
| 379 |
+
"grad_norm": 0.69921875,
|
| 380 |
+
"learning_rate": 0.001,
|
| 381 |
+
"loss": 2.8253028869628904,
|
| 382 |
"step": 880
|
| 383 |
},
|
| 384 |
{
|
| 385 |
+
"epoch": 0.07583417593528817,
|
| 386 |
+
"grad_norm": 0.6875,
|
| 387 |
+
"learning_rate": 0.001,
|
| 388 |
+
"loss": 2.7853120803833007,
|
| 389 |
"step": 900
|
| 390 |
},
|
| 391 |
{
|
| 392 |
+
"epoch": 0.07583417593528817,
|
| 393 |
+
"eval_loss": 2.8034849166870117,
|
| 394 |
+
"eval_runtime": 7.9593,
|
| 395 |
+
"eval_samples_per_second": 1196.967,
|
| 396 |
+
"eval_steps_per_second": 0.879,
|
| 397 |
"step": 900
|
| 398 |
}
|
| 399 |
],
|
| 400 |
"logging_steps": 20,
|
| 401 |
+
"max_steps": 1000,
|
| 402 |
"num_input_tokens_seen": 0,
|
| 403 |
"num_train_epochs": 1,
|
| 404 |
"save_steps": 100,
|
|
|
|
| 414 |
"attributes": {}
|
| 415 |
}
|
| 416 |
},
|
| 417 |
+
"total_flos": 326686998528000.0,
|
| 418 |
+
"train_batch_size": 80,
|
| 419 |
"trial_name": null,
|
| 420 |
"trial_params": null
|
| 421 |
}
|
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4010544
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9e9c917aa3b25bdbc40f473e1cf017022bafceb91187441f0ef9ae86b2f5e2e6
|
| 3 |
size 4010544
|
zain/Activation/out/mlp-linear-9L_run/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-9L_run/training_log.jsonl
CHANGED
|
@@ -1,152 +1,62 @@
|
|
| 1 |
-
{"step": 20, "epoch": 0.
|
| 2 |
-
{"step": 40, "epoch": 0.
|
| 3 |
-
{"step": 60, "epoch": 0.
|
| 4 |
-
{"step": 80, "epoch": 0.
|
| 5 |
-
{"step": 100, "epoch": 0.
|
| 6 |
-
{"step": 100, "epoch": 0.
|
| 7 |
-
{"step": 120, "epoch": 0.
|
| 8 |
-
{"step": 140, "epoch": 0.
|
| 9 |
-
{"step": 160, "epoch": 0.
|
| 10 |
-
{"step": 180, "epoch": 0.
|
| 11 |
-
{"step": 200, "epoch": 0.
|
| 12 |
-
{"step": 200, "epoch": 0.
|
| 13 |
-
{"step": 220, "epoch": 0.
|
| 14 |
-
{"step": 240, "epoch": 0.
|
| 15 |
-
{"step": 260, "epoch": 0.
|
| 16 |
-
{"step": 280, "epoch": 0.
|
| 17 |
-
{"step": 300, "epoch": 0.
|
| 18 |
-
{"step": 300, "epoch": 0.
|
| 19 |
-
{"step": 320, "epoch": 0.
|
| 20 |
-
{"step": 340, "epoch": 0.
|
| 21 |
-
{"step": 360, "epoch": 0.
|
| 22 |
-
{"step": 380, "epoch": 0.
|
| 23 |
-
{"step": 400, "epoch": 0.
|
| 24 |
-
{"step": 400, "epoch": 0.
|
| 25 |
-
{"step": 420, "epoch": 0.
|
| 26 |
-
{"step": 440, "epoch": 0.
|
| 27 |
-
{"step": 460, "epoch": 0.
|
| 28 |
-
{"step": 480, "epoch": 0.
|
| 29 |
-
{"step": 500, "epoch": 0.
|
| 30 |
-
{"step": 500, "epoch": 0.
|
| 31 |
-
{"step": 520, "epoch": 0.
|
| 32 |
-
{"step": 540, "epoch": 0.
|
| 33 |
-
{"step": 560, "epoch": 0.
|
| 34 |
-
{"step": 580, "epoch": 0.
|
| 35 |
-
{"step": 600, "epoch": 0.
|
| 36 |
-
{"step": 600, "epoch": 0.
|
| 37 |
-
{"step": 620, "epoch": 0.
|
| 38 |
-
{"step": 640, "epoch": 0.
|
| 39 |
-
{"step": 660, "epoch": 0.
|
| 40 |
-
{"step": 680, "epoch": 0.
|
| 41 |
-
{"step": 700, "epoch": 0.
|
| 42 |
-
{"step": 700, "epoch": 0.
|
| 43 |
-
{"step": 720, "epoch": 0.
|
| 44 |
-
{"step": 740, "epoch": 0.
|
| 45 |
-
{"step": 760, "epoch": 0.
|
| 46 |
-
{"step": 780, "epoch": 0.
|
| 47 |
-
{"step": 800, "epoch": 0.
|
| 48 |
-
{"step": 800, "epoch": 0.
|
| 49 |
-
{"step": 820, "epoch": 0.
|
| 50 |
-
{"step": 840, "epoch": 0.
|
| 51 |
-
{"step": 860, "epoch": 0.
|
| 52 |
-
{"step": 880, "epoch": 0.
|
| 53 |
-
{"step": 900, "epoch": 0.
|
| 54 |
-
{"step": 900, "epoch": 0.
|
| 55 |
-
{"step": 920, "epoch": 0.
|
| 56 |
-
{"step": 940, "epoch": 0.
|
| 57 |
-
{"step": 960, "epoch": 0.
|
| 58 |
-
{"step": 980, "epoch": 0.
|
| 59 |
-
{"step": 1000, "epoch": 0.
|
| 60 |
-
{"step": 1000, "epoch": 0.
|
| 61 |
-
{"step":
|
| 62 |
-
{"step":
|
| 63 |
-
{"step": 1060, "epoch": 0.03572632288506909, "timestamp": 1787090158.801586, "loss": 3.7539413452148436, "grad_norm": 1.734375, "learning_rate": 0.0003, "train/total_time_seconds": 22.594436164945364, "train/time_per_step_avg": 0.021287702061235904, "train/epoch_time_elapsed": 135.50294019281864, "train/estimated_remaining_minutes": 0.5115721395836687}
|
| 64 |
-
{"step": 1080, "epoch": 0.03640040444893832, "timestamp": 1787090159.7369173, "loss": 3.742329406738281, "grad_norm": 2.109375, "learning_rate": 0.0003, "train/total_time_seconds": 23.02944030240178, "train/time_per_step_avg": 0.021505829468369483, "train/epoch_time_elapsed": 136.4382721632719, "train/estimated_remaining_minutes": 0.5046574881081871}
|
| 65 |
-
{"step": 1100, "epoch": 0.03707448601280755, "timestamp": 1787090160.6459656, "loss": 3.737290954589844, "grad_norm": 1.84375, "learning_rate": 0.0003, "train/total_time_seconds": 23.445339139550924, "train/time_per_step_avg": 0.02136110991239548, "train/epoch_time_elapsed": 137.34732024744153, "train/estimated_remaining_minutes": 0.49732537568744384}
|
| 66 |
-
{"step": 1100, "epoch": 0.03707448601280755, "timestamp": 1787090169.0898743, "eval_loss": 3.744091272354126, "eval_runtime": 8.4423, "eval_samples_per_second": 1128.49, "eval_steps_per_second": 0.829, "train/total_time_seconds": 23.445339139550924, "train/time_per_step_avg": 0.02136110991239548, "train/epoch_time_elapsed": 145.79122596606612, "train/estimated_remaining_minutes": 0.49732537568744384}
|
| 67 |
-
{"step": 1120, "epoch": 0.03774856757667678, "timestamp": 1787090170.1453664, "loss": 3.7377304077148437, "grad_norm": 1.2890625, "learning_rate": 0.0003, "train/total_time_seconds": 23.914364136755466, "train/time_per_step_avg": 0.021765267327427864, "train/epoch_time_elapsed": 146.84672116488218, "train/estimated_remaining_minutes": 0.49109854923694257}
|
| 68 |
-
{"step": 1140, "epoch": 0.03842264914054601, "timestamp": 1787090171.1464329, "loss": 3.713370513916016, "grad_norm": 1.9140625, "learning_rate": 0.0003, "train/total_time_seconds": 24.421343587338924, "train/time_per_step_avg": 0.02250323589891195, "train/epoch_time_elapsed": 147.8477869592607, "train/estimated_remaining_minutes": 0.4855705742511833}
|
| 69 |
-
{"step": 1160, "epoch": 0.03909673070441524, "timestamp": 1787090172.061697, "loss": 3.7126068115234374, "grad_norm": 1.703125, "learning_rate": 0.0003, "train/total_time_seconds": 24.84479048475623, "train/time_per_step_avg": 0.022503543198108673, "train/epoch_time_elapsed": 148.76305164396763, "train/estimated_remaining_minutes": 0.4783336099076631}
|
| 70 |
-
{"step": 1180, "epoch": 0.03977081226828446, "timestamp": 1787090172.96644, "loss": 3.7255340576171876, "grad_norm": 1.3515625, "learning_rate": 0.0003, "train/total_time_seconds": 25.258656304329634, "train/time_per_step_avg": 0.022292160019278525, "train/epoch_time_elapsed": 149.66779493540525, "train/estimated_remaining_minutes": 0.47092410058919654}
|
| 71 |
-
{"step": 1200, "epoch": 0.04044489383215369, "timestamp": 1787090173.8752575, "loss": 3.6966705322265625, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 25.67636000365019, "train/time_per_step_avg": 0.022310208640992642, "train/epoch_time_elapsed": 150.57661206647754, "train/estimated_remaining_minutes": 0.4636009445103506}
|
| 72 |
-
{"step": 1200, "epoch": 0.04044489383215369, "timestamp": 1787090182.3466508, "eval_loss": 3.6958041191101074, "eval_runtime": 8.4698, "eval_samples_per_second": 1124.816, "eval_steps_per_second": 0.826, "train/total_time_seconds": 25.67636000365019, "train/time_per_step_avg": 0.022310208640992642, "train/epoch_time_elapsed": 159.04800394922495, "train/estimated_remaining_minutes": 0.4636009445103506}
|
| 73 |
-
{"step": 1220, "epoch": 0.04111897539602292, "timestamp": 1787090183.3028944, "loss": 3.6724082946777346, "grad_norm": 1.6171875, "learning_rate": 0.0003, "train/total_time_seconds": 26.09580424427986, "train/time_per_step_avg": 0.02181440107524395, "train/epoch_time_elapsed": 160.0042488835752, "train/estimated_remaining_minutes": 0.45632007421691567}
|
| 74 |
-
{"step": 1240, "epoch": 0.04179305695989215, "timestamp": 1787090184.2335675, "loss": 3.6844207763671877, "grad_norm": 1.5234375, "learning_rate": 0.0003, "train/total_time_seconds": 26.528886377811432, "train/time_per_step_avg": 0.021075427904725073, "train/epoch_time_elapsed": 160.9349226243794, "train/estimated_remaining_minutes": 0.44927952736616134}
|
| 75 |
-
{"step": 1260, "epoch": 0.042467138523761376, "timestamp": 1787090185.199458, "loss": 3.6926769256591796, "grad_norm": 1.375, "learning_rate": 0.0003, "train/total_time_seconds": 26.96605222299695, "train/time_per_step_avg": 0.02121261738240719, "train/epoch_time_elapsed": 161.90081299096346, "train/estimated_remaining_minutes": 0.4423003274671457}
|
| 76 |
-
{"step": 1280, "epoch": 0.043141220087630605, "timestamp": 1787090186.1345897, "loss": 3.6725536346435548, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 27.402600783854723, "train/time_per_step_avg": 0.02143944479525089, "train/epoch_time_elapsed": 162.83594417572021, "train/estimated_remaining_minutes": 0.4353017312018589}
|
| 77 |
-
{"step": 1300, "epoch": 0.043815301651499834, "timestamp": 1787090187.0789473, "loss": 3.6652542114257813, "grad_norm": 1.4921875, "learning_rate": 0.0003, "train/total_time_seconds": 27.839303817600012, "train/time_per_step_avg": 0.021629438139498233, "train/epoch_time_elapsed": 163.78030264377594, "train/estimated_remaining_minutes": 0.4282969818092309}
|
| 78 |
-
{"step": 1300, "epoch": 0.043815301651499834, "timestamp": 1787090195.647478, "eval_loss": 3.6601669788360596, "eval_runtime": 8.567, "eval_samples_per_second": 1112.052, "eval_steps_per_second": 0.817, "train/total_time_seconds": 27.839303817600012, "train/time_per_step_avg": 0.021629438139498233, "train/epoch_time_elapsed": 172.34883080422878, "train/estimated_remaining_minutes": 0.4282969818092309}
|
| 79 |
-
{"step": 1320, "epoch": 0.044489383215369056, "timestamp": 1787090196.6083207, "loss": 3.6304325103759765, "grad_norm": 1.203125, "learning_rate": 0.0003, "train/total_time_seconds": 28.265503082424402, "train/time_per_step_avg": 0.021696988381445407, "train/epoch_time_elapsed": 173.3096746765077, "train/estimated_remaining_minutes": 0.421127444914909}
|
| 80 |
-
{"step": 1340, "epoch": 0.045163464779238285, "timestamp": 1787090197.5520518, "loss": 3.654827880859375, "grad_norm": 1.328125, "learning_rate": 0.0003, "train/total_time_seconds": 28.698285780847073, "train/time_per_step_avg": 0.021693994030356406, "train/epoch_time_elapsed": 174.25340588018298, "train/estimated_remaining_minutes": 0.414054869474908}
|
| 81 |
-
{"step": 1360, "epoch": 0.045837546343107514, "timestamp": 1787090198.5060043, "loss": 3.634907531738281, "grad_norm": 1.3671875, "learning_rate": 0.0003, "train/total_time_seconds": 29.14553214609623, "train/time_per_step_avg": 0.021794799230992794, "train/epoch_time_elapsed": 175.20735903084278, "train/estimated_remaining_minutes": 0.4071802285116385}
|
| 82 |
-
{"step": 1380, "epoch": 0.046511627906976744, "timestamp": 1787090199.4406104, "loss": 3.631271743774414, "grad_norm": 1.3203125, "learning_rate": 0.0003, "train/total_time_seconds": 29.57575212791562, "train/time_per_step_avg": 0.021731513440608977, "train/epoch_time_elapsed": 176.14196480065584, "train/estimated_remaining_minutes": 0.4000584828896799}
|
| 83 |
-
{"step": 1400, "epoch": 0.04718570947084597, "timestamp": 1787090200.3831522, "loss": 3.6088893890380858, "grad_norm": 1.4765625, "learning_rate": 0.0003, "train/total_time_seconds": 30.00523616373539, "train/time_per_step_avg": 0.02165932346135378, "train/epoch_time_elapsed": 177.08450701460242, "train/estimated_remaining_minutes": 0.3929257116679634}
|
| 84 |
-
{"step": 1400, "epoch": 0.04718570947084597, "timestamp": 1787090208.7806206, "eval_loss": 3.620887517929077, "eval_runtime": 8.3958, "eval_samples_per_second": 1134.733, "eval_steps_per_second": 0.834, "train/total_time_seconds": 30.00523616373539, "train/time_per_step_avg": 0.02165932346135378, "train/epoch_time_elapsed": 185.48197343945503, "train/estimated_remaining_minutes": 0.3929257116679634}
|
| 85 |
-
{"step": 1420, "epoch": 0.0478597910347152, "timestamp": 1787090209.758754, "loss": 3.6118431091308594, "grad_norm": 1.3125, "learning_rate": 0.0003, "train/total_time_seconds": 30.44477339833975, "train/time_per_step_avg": 0.02179270315915346, "train/epoch_time_elapsed": 186.46010866761208, "train/estimated_remaining_minutes": 0.385919662795856}
|
| 86 |
-
{"step": 1440, "epoch": 0.04853387259858443, "timestamp": 1787090210.6889102, "loss": 3.6007781982421876, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 30.87500672414899, "train/time_per_step_avg": 0.02176720943301916, "train/epoch_time_elapsed": 187.390264775604, "train/estimated_remaining_minutes": 0.37879059175460567}
|
| 87 |
-
{"step": 1460, "epoch": 0.04920795416245366, "timestamp": 1787090211.629568, "loss": 3.577118682861328, "grad_norm": 1.171875, "learning_rate": 0.0003, "train/total_time_seconds": 31.31340730935335, "train/time_per_step_avg": 0.02167875163257122, "train/epoch_time_elapsed": 188.33092243596911, "train/estimated_remaining_minutes": 0.3717573470516837}
|
| 88 |
-
{"step": 1480, "epoch": 0.04988203572632288, "timestamp": 1787090212.5728993, "loss": 3.600400924682617, "grad_norm": 1.453125, "learning_rate": 0.0003, "train/total_time_seconds": 31.75088044628501, "train/time_per_step_avg": 0.021751283183693886, "train/epoch_time_elapsed": 189.27425381541252, "train/estimated_remaining_minutes": 0.3647060591803008}
|
| 89 |
-
{"step": 1500, "epoch": 0.05055611729019211, "timestamp": 1787090213.5248115, "loss": 3.5930843353271484, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 32.199186488986015, "train/time_per_step_avg": 0.021939503252506255, "train/epoch_time_elapsed": 190.22616611793637, "train/estimated_remaining_minutes": 0.3577687387665113}
|
| 90 |
-
{"step": 1500, "epoch": 0.05055611729019211, "timestamp": 1787090222.1475577, "eval_loss": 3.5886731147766113, "eval_runtime": 8.6212, "eval_samples_per_second": 1105.07, "eval_steps_per_second": 0.812, "train/total_time_seconds": 32.199186488986015, "train/time_per_step_avg": 0.021939503252506255, "train/epoch_time_elapsed": 198.84891088306904, "train/estimated_remaining_minutes": 0.3577687387665113}
|
| 91 |
-
{"step": 1520, "epoch": 0.05123019885406134, "timestamp": 1787090223.1094587, "loss": 3.5856658935546877, "grad_norm": 1.703125, "learning_rate": 0.0003, "train/total_time_seconds": 32.62315646559, "train/time_per_step_avg": 0.021783830672502516, "train/epoch_time_elapsed": 199.81081285327673, "train/estimated_remaining_minutes": 0.35055584798550654}
|
| 92 |
-
{"step": 1540, "epoch": 0.05190428041793057, "timestamp": 1787090224.036868, "loss": 3.592951202392578, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 33.04965380206704, "train/time_per_step_avg": 0.021746470779180526, "train/epoch_time_elapsed": 200.7382232248783, "train/estimated_remaining_minutes": 0.34337302651498225}
|
| 93 |
-
{"step": 1560, "epoch": 0.0525783619817998, "timestamp": 1787090224.9690807, "loss": 3.5940937042236327, "grad_norm": 1.8046875, "learning_rate": 0.0003, "train/total_time_seconds": 33.478857297450304, "train/time_per_step_avg": 0.021654499880969524, "train/epoch_time_elapsed": 201.67043430358171, "train/estimated_remaining_minutes": 0.33621929337183}
|
| 94 |
-
{"step": 1580, "epoch": 0.05325244354566903, "timestamp": 1787090225.8977115, "loss": 3.5703506469726562, "grad_norm": 1.375, "learning_rate": 0.0003, "train/total_time_seconds": 33.90842756256461, "train/time_per_step_avg": 0.021575471162796022, "train/epoch_time_elapsed": 202.59906630590558, "train/estimated_remaining_minutes": 0.32906912824429796}
|
| 95 |
-
{"step": 1600, "epoch": 0.053926525109538256, "timestamp": 1787090226.8264098, "loss": 3.5667308807373046, "grad_norm": 1.2578125, "learning_rate": 0.0003, "train/total_time_seconds": 34.33643752709031, "train/time_per_step_avg": 0.02137251038104296, "train/epoch_time_elapsed": 203.5277648679912, "train/estimated_remaining_minutes": 0.32190410181647167}
|
| 96 |
-
{"step": 1600, "epoch": 0.053926525109538256, "timestamp": 1787090235.2771208, "eval_loss": 3.55812931060791, "eval_runtime": 8.4492, "eval_samples_per_second": 1127.557, "eval_steps_per_second": 0.828, "train/total_time_seconds": 34.33643752709031, "train/time_per_step_avg": 0.02137251038104296, "train/epoch_time_elapsed": 211.97847399488091, "train/estimated_remaining_minutes": 0.32190410181647167}
|
| 97 |
-
{"step": 1620, "epoch": 0.054600606673407485, "timestamp": 1787090236.2274826, "loss": 3.578482437133789, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 34.754911847412586, "train/time_per_step_avg": 0.02131755381822586, "train/epoch_time_elapsed": 212.92883710190654, "train/estimated_remaining_minutes": 0.31465352289838555}
|
| 98 |
-
{"step": 1640, "epoch": 0.05527468823727671, "timestamp": 1787090237.15594, "loss": 3.5558090209960938, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 35.1784917190671, "train/time_per_step_avg": 0.02128837917000055, "train/epoch_time_elapsed": 213.8572948165238, "train/estimated_remaining_minutes": 0.30745429754469206}
|
| 99 |
-
{"step": 1660, "epoch": 0.05594876980114594, "timestamp": 1787090238.085788, "loss": 3.566594696044922, "grad_norm": 1.359375, "learning_rate": 0.0003, "train/total_time_seconds": 35.59984588995576, "train/time_per_step_avg": 0.02120988592505455, "train/epoch_time_elapsed": 214.78714298456907, "train/estimated_remaining_minutes": 0.30023966413215697}
|
| 100 |
-
{"step": 1680, "epoch": 0.056622851365015166, "timestamp": 1787090239.0095909, "loss": 3.5346706390380858, "grad_norm": 1.6015625, "learning_rate": 0.0003, "train/total_time_seconds": 36.021644342690706, "train/time_per_step_avg": 0.021132167801260947, "train/epoch_time_elapsed": 215.71094546467066, "train/estimated_remaining_minutes": 0.2930332178671267}
|
| 101 |
-
{"step": 1700, "epoch": 0.057296932928884395, "timestamp": 1787090239.9389343, "loss": 3.530739974975586, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 36.447514064610004, "train/time_per_step_avg": 0.021110765375196933, "train/epoch_time_elapsed": 216.64028869196773, "train/estimated_remaining_minutes": 0.2858628554087059}
|
| 102 |
-
{"step": 1700, "epoch": 0.057296932928884395, "timestamp": 1787090248.842449, "eval_loss": 3.530102014541626, "eval_runtime": 8.9019, "eval_samples_per_second": 1070.223, "eval_steps_per_second": 0.786, "train/total_time_seconds": 36.447514064610004, "train/time_per_step_avg": 0.021110765375196933, "train/epoch_time_elapsed": 225.54380098730326, "train/estimated_remaining_minutes": 0.2858628554087059}
|
| 103 |
-
{"step": 1720, "epoch": 0.057971014492753624, "timestamp": 1787090249.8342133, "loss": 3.519282913208008, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 36.8907287530601, "train/time_per_step_avg": 0.02135816905647516, "train/epoch_time_elapsed": 226.53556845709682, "train/estimated_remaining_minutes": 0.2788252754591752}
|
| 104 |
-
{"step": 1740, "epoch": 0.05864509605662285, "timestamp": 1787090250.8161082, "loss": 3.5207279205322264, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 37.33156434074044, "train/time_per_step_avg": 0.021530726216733454, "train/epoch_time_elapsed": 227.51746324449778, "train/estimated_remaining_minutes": 0.2717623457755051}
|
| 105 |
-
{"step": 1760, "epoch": 0.05931917762049208, "timestamp": 1787090251.812636, "loss": 3.5107025146484374, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 37.78914465382695, "train/time_per_step_avg": 0.02189298763871193, "train/epoch_time_elapsed": 228.51399037614465, "train/estimated_remaining_minutes": 0.2648102939756813}
|
| 106 |
-
{"step": 1780, "epoch": 0.05999325918436131, "timestamp": 1787090252.737064, "loss": 3.5151645660400392, "grad_norm": 1.4921875, "learning_rate": 0.0003, "train/total_time_seconds": 38.21554037556052, "train/time_per_step_avg": 0.02193896032869816, "train/epoch_time_elapsed": 229.43841843679547, "train/estimated_remaining_minutes": 0.25763285646445294}
|
| 107 |
-
{"step": 1800, "epoch": 0.06066734074823053, "timestamp": 1787090253.6519675, "loss": 3.506744384765625, "grad_norm": 1.34375, "learning_rate": 0.0003, "train/total_time_seconds": 38.63394458219409, "train/time_per_step_avg": 0.021864305175840856, "train/epoch_time_elapsed": 230.35332256928086, "train/estimated_remaining_minutes": 0.2504051963660728}
|
| 108 |
-
{"step": 1800, "epoch": 0.06066734074823053, "timestamp": 1787090262.13157, "eval_loss": 3.510505437850952, "eval_runtime": 8.4781, "eval_samples_per_second": 1123.716, "eval_steps_per_second": 0.826, "train/total_time_seconds": 38.63394458219409, "train/time_per_step_avg": 0.021864305175840856, "train/epoch_time_elapsed": 238.83292232453823, "train/estimated_remaining_minutes": 0.2504051963660728}
|
| 109 |
-
{"step": 1820, "epoch": 0.06134142231209976, "timestamp": 1787090263.1293309, "loss": 3.4982723236083983, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 39.06876355037093, "train/time_per_step_avg": 0.02178034797310829, "train/epoch_time_elapsed": 239.83068535104394, "train/estimated_remaining_minutes": 0.24328534078985564}
|
| 110 |
-
{"step": 1840, "epoch": 0.06201550387596899, "timestamp": 1787090264.0554924, "loss": 3.4931972503662108, "grad_norm": 1.6015625, "learning_rate": 0.0003, "train/total_time_seconds": 39.49600052088499, "train/time_per_step_avg": 0.021644361801445484, "train/epoch_time_elapsed": 240.75684678554535, "train/estimated_remaining_minutes": 0.23611739441833418}
|
| 111 |
-
{"step": 1860, "epoch": 0.06268958543983821, "timestamp": 1787090264.9857557, "loss": 3.5061672210693358, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 39.91950794309378, "train/time_per_step_avg": 0.021303632892668248, "train/epoch_time_elapsed": 241.68710988014936, "train/estimated_remaining_minutes": 0.2289290778098568}
|
| 112 |
-
{"step": 1880, "epoch": 0.06336366700370745, "timestamp": 1787090265.9066734, "loss": 3.4790119171142577, "grad_norm": 1.7421875, "learning_rate": 0.0003, "train/total_time_seconds": 40.34255151450634, "train/time_per_step_avg": 0.02127011138945818, "train/epoch_time_elapsed": 242.60802837461233, "train/estimated_remaining_minutes": 0.22174097463647102}
|
| 113 |
-
{"step": 1900, "epoch": 0.06403774856757667, "timestamp": 1787090266.8239949, "loss": 3.4856170654296874, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 40.76226783171296, "train/time_per_step_avg": 0.021283232495188712, "train/epoch_time_elapsed": 243.5253492332995, "train/estimated_remaining_minutes": 0.2145382517458577}
|
| 114 |
-
{"step": 1900, "epoch": 0.06403774856757667, "timestamp": 1787090275.2962322, "eval_loss": 3.486959457397461, "eval_runtime": 8.4707, "eval_samples_per_second": 1124.702, "eval_steps_per_second": 0.826, "train/total_time_seconds": 40.76226783171296, "train/time_per_step_avg": 0.021283232495188712, "train/epoch_time_elapsed": 251.99758548289537, "train/estimated_remaining_minutes": 0.2145382517458577}
|
| 115 |
-
{"step": 1920, "epoch": 0.06471183013144591, "timestamp": 1787090276.2503629, "loss": 3.4741825103759765, "grad_norm": 1.625, "learning_rate": 0.0003, "train/total_time_seconds": 41.185888312757015, "train/time_per_step_avg": 0.021171247623860835, "train/epoch_time_elapsed": 252.95171785727143, "train/estimated_remaining_minutes": 0.20735950713020024}
|
| 116 |
-
{"step": 1940, "epoch": 0.06538591169531513, "timestamp": 1787090277.180064, "loss": 3.459843063354492, "grad_norm": 1.484375, "learning_rate": 0.0003, "train/total_time_seconds": 41.61065972223878, "train/time_per_step_avg": 0.021146592013537885, "train/epoch_time_elapsed": 253.8814189210534, "train/estimated_remaining_minutes": 0.20018874093173297}
|
| 117 |
-
{"step": 1960, "epoch": 0.06605999325918437, "timestamp": 1787090278.1718454, "loss": 3.482832336425781, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 42.0616734996438, "train/time_per_step_avg": 0.02142165556550026, "train/epoch_time_elapsed": 254.873200006783, "train/estimated_remaining_minutes": 0.1931403374983644}
|
| 118 |
-
{"step": 1980, "epoch": 0.06673407482305359, "timestamp": 1787090279.1152725, "loss": 3.490843963623047, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 42.502584993839264, "train/time_per_step_avg": 0.021600334793329238, "train/epoch_time_elapsed": 255.81662721559405, "train/estimated_remaining_minutes": 0.18603825081478467}
|
| 119 |
-
{"step": 2000, "epoch": 0.06740815638692282, "timestamp": 1787090280.0470285, "loss": 3.471944046020508, "grad_norm": 1.609375, "learning_rate": 0.0003, "train/total_time_seconds": 42.93809688463807, "train/time_per_step_avg": 0.021758290529251097, "train/epoch_time_elapsed": 256.7483834400773, "train/estimated_remaining_minutes": 0.1789087370193253}
|
| 120 |
-
{"step": 2000, "epoch": 0.06740815638692282, "timestamp": 1787090288.5251565, "eval_loss": 3.4656903743743896, "eval_runtime": 8.4764, "eval_samples_per_second": 1123.949, "eval_steps_per_second": 0.826, "train/total_time_seconds": 42.93809688463807, "train/time_per_step_avg": 0.021758290529251097, "train/epoch_time_elapsed": 265.22650999203324, "train/estimated_remaining_minutes": 0.1789087370193253}
|
| 121 |
-
{"step": 2020, "epoch": 0.06808223795079205, "timestamp": 1787090289.4874835, "loss": 3.4606929779052735, "grad_norm": 1.8046875, "learning_rate": 0.0003, "train/total_time_seconds": 43.36348307877779, "train/time_per_step_avg": 0.02177594766020775, "train/epoch_time_elapsed": 266.188837967813, "train/estimated_remaining_minutes": 0.1717365666486249}
|
| 122 |
-
{"step": 2040, "epoch": 0.06875631951466127, "timestamp": 1787090290.4118824, "loss": 3.4594757080078127, "grad_norm": 1.515625, "learning_rate": 0.0003, "train/total_time_seconds": 43.79138084128499, "train/time_per_step_avg": 0.02180721119046211, "train/epoch_time_elapsed": 267.1132374033332, "train/estimated_remaining_minutes": 0.16457545087411027}
|
| 123 |
-
{"step": 2060, "epoch": 0.0694304010785305, "timestamp": 1787090291.3374932, "loss": 3.446879577636719, "grad_norm": 1.2265625, "learning_rate": 0.0003, "train/total_time_seconds": 44.2190666384995, "train/time_per_step_avg": 0.02157393138855696, "train/epoch_time_elapsed": 268.0388476587832, "train/estimated_remaining_minutes": 0.15741415308203704}
|
| 124 |
-
{"step": 2080, "epoch": 0.07010448264239973, "timestamp": 1787090292.3561628, "loss": 3.4532211303710936, "grad_norm": 1.34375, "learning_rate": 0.0003, "train/total_time_seconds": 44.67429492995143, "train/time_per_step_avg": 0.021717099361121654, "train/epoch_time_elapsed": 269.0575177781284, "train/estimated_remaining_minutes": 0.15034618486041346}
|
| 125 |
-
{"step": 2100, "epoch": 0.07077856420626896, "timestamp": 1787090293.307748, "loss": 3.445978546142578, "grad_norm": 1.21875, "learning_rate": 0.0003, "train/total_time_seconds": 45.11734665185213, "train/time_per_step_avg": 0.021792497672140598, "train/epoch_time_elapsed": 270.0091031193733, "train/estimated_remaining_minutes": 0.14322967191064168}
|
| 126 |
-
{"step": 2100, "epoch": 0.07077856420626896, "timestamp": 1787090301.7926538, "eval_loss": 3.4454257488250732, "eval_runtime": 8.4831, "eval_samples_per_second": 1123.053, "eval_steps_per_second": 0.825, "train/total_time_seconds": 45.11734665185213, "train/time_per_step_avg": 0.021792497672140598, "train/epoch_time_elapsed": 278.49400681629777, "train/estimated_remaining_minutes": 0.14322967191064168}
|
| 127 |
-
{"step": 2120, "epoch": 0.07145264577013818, "timestamp": 1787090302.752118, "loss": 3.446368408203125, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 45.54672980308533, "train/time_per_step_avg": 0.02183246724307537, "train/epoch_time_elapsed": 279.4534733183682, "train/estimated_remaining_minutes": 0.13606727456896558}
|
| 128 |
-
{"step": 2140, "epoch": 0.07212672733400742, "timestamp": 1787090303.6751964, "loss": 3.4467777252197265, "grad_norm": 1.9453125, "learning_rate": 0.0003, "train/total_time_seconds": 45.97253290563822, "train/time_per_step_avg": 0.021811520643532277, "train/epoch_time_elapsed": 280.37655137851834, "train/estimated_remaining_minutes": 0.12889495207188284}
|
| 129 |
-
{"step": 2160, "epoch": 0.07280080889787664, "timestamp": 1787090304.601299, "loss": 3.427969741821289, "grad_norm": 1.2421875, "learning_rate": 0.0003, "train/total_time_seconds": 46.40290119126439, "train/time_per_step_avg": 0.021838345527648927, "train/epoch_time_elapsed": 281.30265340954065, "train/estimated_remaining_minutes": 0.1217360062116504}
|
| 130 |
-
{"step": 2180, "epoch": 0.07347489046174586, "timestamp": 1787090305.5422475, "loss": 3.437174606323242, "grad_norm": 1.546875, "learning_rate": 0.0003, "train/total_time_seconds": 46.82929648458958, "train/time_per_step_avg": 0.021550015546381474, "train/epoch_time_elapsed": 282.2436023950577, "train/estimated_remaining_minutes": 0.11456708620083077}
|
| 131 |
-
{"step": 2200, "epoch": 0.0741489720256151, "timestamp": 1787090306.4651196, "loss": 3.437343215942383, "grad_norm": 1.5, "learning_rate": 0.0003, "train/total_time_seconds": 47.253708347678185, "train/time_per_step_avg": 0.021363616958260535, "train/epoch_time_elapsed": 283.16647424921393, "train/estimated_remaining_minutes": 0.1073947916992686}
|
| 132 |
-
{"step": 2200, "epoch": 0.0741489720256151, "timestamp": 1787090314.8796115, "eval_loss": 3.431370735168457, "eval_runtime": 8.4127, "eval_samples_per_second": 1132.457, "eval_steps_per_second": 0.832, "train/total_time_seconds": 47.253708347678185, "train/time_per_step_avg": 0.021363616958260535, "train/epoch_time_elapsed": 291.5809638015926, "train/estimated_remaining_minutes": 0.1073947916992686}
|
| 133 |
-
{"step": 2220, "epoch": 0.07482305358948432, "timestamp": 1787090315.8496833, "loss": 3.3987106323242187, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 47.68398727849126, "train/time_per_step_avg": 0.021372574754059313, "train/epoch_time_elapsed": 292.55103803798556, "train/estimated_remaining_minutes": 0.10023660989472637}
|
| 134 |
-
{"step": 2240, "epoch": 0.07549713515335356, "timestamp": 1787090316.7768753, "loss": 3.4200679779052736, "grad_norm": 1.25, "learning_rate": 0.0003, "train/total_time_seconds": 48.1130493208766, "train/time_per_step_avg": 0.021405164152383804, "train/epoch_time_elapsed": 293.47823068127036, "train/estimated_remaining_minutes": 0.09307583946002912}
|
| 135 |
-
{"step": 2260, "epoch": 0.07617121671722278, "timestamp": 1787090317.712523, "loss": 3.391765594482422, "grad_norm": 1.46875, "learning_rate": 0.0003, "train/total_time_seconds": 48.5431542173028, "train/time_per_step_avg": 0.021402530260384082, "train/epoch_time_elapsed": 294.4138778038323, "train/estimated_remaining_minutes": 0.08591708711027043}
|
| 136 |
-
{"step": 2280, "epoch": 0.07684529828109202, "timestamp": 1787090318.6368556, "loss": 3.4147933959960937, "grad_norm": 1.546875, "learning_rate": 0.0003, "train/total_time_seconds": 48.96943225711584, "train/time_per_step_avg": 0.02140135772526264, "train/epoch_time_elapsed": 295.3382097110152, "train/estimated_remaining_minutes": 0.07875201093980616}
|
| 137 |
-
{"step": 2300, "epoch": 0.07751937984496124, "timestamp": 1787090319.5716066, "loss": 3.411570358276367, "grad_norm": 1.4609375, "learning_rate": 0.0003, "train/total_time_seconds": 49.40087690204382, "train/time_per_step_avg": 0.02147168554365635, "train/epoch_time_elapsed": 296.272961769253, "train/estimated_remaining_minutes": 0.071595473771078}
|
| 138 |
-
{"step": 2300, "epoch": 0.07751937984496124, "timestamp": 1787090327.969581, "eval_loss": 3.4143009185791016, "eval_runtime": 8.3964, "eval_samples_per_second": 1134.65, "eval_steps_per_second": 0.834, "train/total_time_seconds": 49.40087690204382, "train/time_per_step_avg": 0.02147168554365635, "train/epoch_time_elapsed": 304.6709332689643, "train/estimated_remaining_minutes": 0.071595473771078}
|
| 139 |
-
{"step": 2320, "epoch": 0.07819346140883048, "timestamp": 1787090328.9620771, "loss": 3.408540725708008, "grad_norm": 1.4765625, "learning_rate": 0.0003, "train/total_time_seconds": 49.833336658775806, "train/time_per_step_avg": 0.021493493802845477, "train/epoch_time_elapsed": 305.66343190521, "train/estimated_remaining_minutes": 0.06443965947255492}
|
| 140 |
-
{"step": 2340, "epoch": 0.0788675429726997, "timestamp": 1787090329.8789449, "loss": 3.393010711669922, "grad_norm": 1.46875, "learning_rate": 0.0003, "train/total_time_seconds": 50.2537716627121, "train/time_per_step_avg": 0.02140722341835499, "train/epoch_time_elapsed": 306.58029908686876, "train/estimated_remaining_minutes": 0.05726925545608216}
|
| 141 |
-
{"step": 2360, "epoch": 0.07954162453656892, "timestamp": 1787090330.7985592, "loss": 3.4179660797119142, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 50.677648298442364, "train/time_per_step_avg": 0.021344940811395645, "train/epoch_time_elapsed": 307.4999114535749, "train/estimated_remaining_minutes": 0.050105019504109685}
|
| 142 |
-
{"step": 2380, "epoch": 0.08021570610043816, "timestamp": 1787090331.7210913, "loss": 3.4083057403564454, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 51.10111549496651, "train/time_per_step_avg": 0.02131683237850666, "train/epoch_time_elapsed": 308.42244643718004, "train/estimated_remaining_minutes": 0.04294211386131639}
|
| 143 |
-
{"step": 2400, "epoch": 0.08088978766430738, "timestamp": 1787090332.637041, "loss": 3.405569839477539, "grad_norm": 1.6640625, "learning_rate": 0.0003, "train/total_time_seconds": 51.520632427185774, "train/time_per_step_avg": 0.021197555251419545, "train/epoch_time_elapsed": 309.3383958786726, "train/estimated_remaining_minutes": 0.03577821696332346}
|
| 144 |
-
{"step": 2400, "epoch": 0.08088978766430738, "timestamp": 1787090341.087288, "eval_loss": 3.405479907989502, "eval_runtime": 8.4486, "eval_samples_per_second": 1127.639, "eval_steps_per_second": 0.829, "train/total_time_seconds": 51.520632427185774, "train/time_per_step_avg": 0.021197555251419545, "train/epoch_time_elapsed": 317.7886408865452, "train/estimated_remaining_minutes": 0.03577821696332346}
|
| 145 |
-
{"step": 2420, "epoch": 0.08156386922817661, "timestamp": 1787090342.070789, "loss": 3.422397232055664, "grad_norm": 1.359375, "learning_rate": 0.0003, "train/total_time_seconds": 51.95656028389931, "train/time_per_step_avg": 0.021232236251235007, "train/epoch_time_elapsed": 318.77214378118515, "train/estimated_remaining_minutes": 0.028626204013167664}
|
| 146 |
-
{"step": 2440, "epoch": 0.08223795079204584, "timestamp": 1787090343.04529, "loss": 3.372893524169922, "grad_norm": 1.5234375, "learning_rate": 0.0003, "train/total_time_seconds": 52.38708958029747, "train/time_per_step_avg": 0.021333179175853728, "train/epoch_time_elapsed": 319.74664357304573, "train/estimated_remaining_minutes": 0.021470118680449783}
|
| 147 |
-
{"step": 2460, "epoch": 0.08291203235591507, "timestamp": 1787090344.0064027, "loss": 3.4056819915771483, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 52.83087908104062, "train/time_per_step_avg": 0.02153230782598257, "train/epoch_time_elapsed": 320.7077578008175, "train/estimated_remaining_minutes": 0.01431731140407605}
|
| 148 |
-
{"step": 2480, "epoch": 0.0835861139197843, "timestamp": 1787090344.98779, "loss": 3.3840850830078124, "grad_norm": 1.6328125, "learning_rate": 0.0003, "train/total_time_seconds": 53.28470703586936, "train/time_per_step_avg": 0.02183591540902853, "train/epoch_time_elapsed": 321.68914468213916, "train/estimated_remaining_minutes": 0.007161922988692117}
|
| 149 |
-
{"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090345.914262, "loss": 3.3768421173095704, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 322.6156167127192, "train/estimated_remaining_minutes": 0.0}
|
| 150 |
-
{"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090354.314014, "eval_loss": 3.386671304702759, "eval_runtime": 8.3982, "eval_samples_per_second": 1134.413, "eval_steps_per_second": 0.834, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 331.015366576612, "train/estimated_remaining_minutes": 0.0}
|
| 151 |
-
{"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090354.3694549, "train_runtime": 332.0558, "train_samples_per_second": 240.923, "train_steps_per_second": 7.529, "total_flos": 362985553920000.0, "train_loss": 4.1883163818359375, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 331.0708075389266, "train/estimated_remaining_minutes": 0.0}
|
| 152 |
-
{"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090363.1779864, "eval_loss": 3.386671304702759, "eval_runtime": 8.8055, "eval_samples_per_second": 1081.938, "eval_steps_per_second": 0.795, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 339.8793397396803, "train/estimated_remaining_minutes": 0.0}
|
|
|
|
| 1 |
+
{"step": 20, "epoch": 0.0016852039096730705, "timestamp": 1787159744.744951, "loss": 8.200721740722656, "grad_norm": 1.2421875, "learning_rate": 9.5e-05, "train/total_time_seconds": 1.1820026859641075, "train/time_per_step_avg": 0.059100134298205376, "train/epoch_time_elapsed": 2.652187306433916, "train/estimated_remaining_minutes": 0.9653021935373545}
|
| 2 |
+
{"step": 40, "epoch": 0.003370407819346141, "timestamp": 1787159746.6767337, "loss": 7.764720153808594, "grad_norm": 1.2421875, "learning_rate": 0.00019500000000000002, "train/total_time_seconds": 1.8186652921140194, "train/time_per_step_avg": 0.04546663230285049, "train/epoch_time_elapsed": 4.583965107798576, "train/estimated_remaining_minutes": 0.7274661168456078}
|
| 3 |
+
{"step": 60, "epoch": 0.005055611729019211, "timestamp": 1787159748.5245914, "loss": 7.160160064697266, "grad_norm": 1.2265625, "learning_rate": 0.000295, "train/total_time_seconds": 2.4542956836521626, "train/time_per_step_avg": 0.04090492806086938, "train/epoch_time_elapsed": 6.4318275190889835, "train/estimated_remaining_minutes": 0.6408438729536203}
|
| 4 |
+
{"step": 80, "epoch": 0.006740815638692282, "timestamp": 1787159750.3386374, "loss": 6.480591583251953, "grad_norm": 1.2109375, "learning_rate": 0.000395, "train/total_time_seconds": 3.086422026157379, "train/time_per_step_avg": 0.03858027532696724, "train/epoch_time_elapsed": 8.245873432606459, "train/estimated_remaining_minutes": 0.5915642216801643}
|
| 5 |
+
{"step": 100, "epoch": 0.008426019548365353, "timestamp": 1787159752.2435765, "loss": 5.919136428833008, "grad_norm": 1.0078125, "learning_rate": 0.000495, "train/total_time_seconds": 3.722423255443573, "train/time_per_step_avg": 0.03722423255443573, "train/epoch_time_elapsed": 10.150811605155468, "train/estimated_remaining_minutes": 0.5583634883165359}
|
| 6 |
+
{"step": 100, "epoch": 0.008426019548365353, "timestamp": 1787159760.215479, "eval_loss": 5.634258270263672, "eval_runtime": 7.97, "eval_samples_per_second": 1195.356, "eval_steps_per_second": 0.878, "train/total_time_seconds": 3.722423255443573, "train/time_per_step_avg": 0.03722423255443573, "train/epoch_time_elapsed": 18.122713901102543, "train/estimated_remaining_minutes": 0.5583634883165359}
|
| 7 |
+
{"step": 120, "epoch": 0.010111223458038422, "timestamp": 1787159762.0986185, "loss": 5.412916946411133, "grad_norm": 1.171875, "learning_rate": 0.0005949999999999999, "train/total_time_seconds": 4.363878160715103, "train/time_per_step_avg": 0.031818754747509954, "train/epoch_time_elapsed": 20.005853559821844, "train/estimated_remaining_minutes": 0.5333628863096237}
|
| 8 |
+
{"step": 140, "epoch": 0.011796427367711493, "timestamp": 1787159763.9296083, "loss": 5.0327880859375, "grad_norm": 1.515625, "learning_rate": 0.000695, "train/total_time_seconds": 5.001869980245829, "train/time_per_step_avg": 0.03183204688131809, "train/epoch_time_elapsed": 21.836841892451048, "train/estimated_remaining_minutes": 0.5120962122632634}
|
| 9 |
+
{"step": 160, "epoch": 0.013481631277384564, "timestamp": 1787159765.7692258, "loss": 4.69476089477539, "grad_norm": 0.9765625, "learning_rate": 0.000795, "train/total_time_seconds": 5.6416174322366714, "train/time_per_step_avg": 0.03187321748584509, "train/epoch_time_elapsed": 23.676462575793266, "train/estimated_remaining_minutes": 0.4936415253207088}
|
| 10 |
+
{"step": 180, "epoch": 0.015166835187057633, "timestamp": 1787159767.6319602, "loss": 4.4246673583984375, "grad_norm": 0.8828125, "learning_rate": 0.0008950000000000001, "train/total_time_seconds": 6.286054410040379, "train/time_per_step_avg": 0.03199632383882999, "train/epoch_time_elapsed": 25.539197202771902, "train/estimated_remaining_minutes": 0.4772745015030658}
|
| 11 |
+
{"step": 200, "epoch": 0.016852039096730706, "timestamp": 1787159769.4737778, "loss": 4.204695129394532, "grad_norm": 0.87890625, "learning_rate": 0.000995, "train/total_time_seconds": 6.9344093687832355, "train/time_per_step_avg": 0.03211986113339663, "train/epoch_time_elapsed": 27.38101428002119, "train/estimated_remaining_minutes": 0.4622939579188824}
|
| 12 |
+
{"step": 200, "epoch": 0.016852039096730706, "timestamp": 1787159777.4086134, "eval_loss": 4.133424282073975, "eval_runtime": 7.9329, "eval_samples_per_second": 1200.942, "eval_steps_per_second": 0.882, "train/total_time_seconds": 6.9344093687832355, "train/time_per_step_avg": 0.03211986113339663, "train/epoch_time_elapsed": 35.31584985554218, "train/estimated_remaining_minutes": 0.4622939579188824}
|
| 13 |
+
{"step": 220, "epoch": 0.018537243006403775, "timestamp": 1787159779.2836711, "loss": 4.048733520507812, "grad_norm": 0.6875, "learning_rate": 0.001, "train/total_time_seconds": 7.581706669181585, "train/time_per_step_avg": 0.032178285084664825, "train/epoch_time_elapsed": 37.19090821221471, "train/estimated_remaining_minutes": 0.4480099395425482}
|
| 14 |
+
{"step": 240, "epoch": 0.020222446916076844, "timestamp": 1787159781.146838, "loss": 3.9190834045410154, "grad_norm": 0.73828125, "learning_rate": 0.001, "train/total_time_seconds": 8.228051170706749, "train/time_per_step_avg": 0.0322618119046092, "train/epoch_time_elapsed": 39.054074600338936, "train/estimated_remaining_minutes": 0.434258256231745}
|
| 15 |
+
{"step": 260, "epoch": 0.021907650825749917, "timestamp": 1787159782.9797618, "loss": 3.7833847045898437, "grad_norm": 0.68359375, "learning_rate": 0.001, "train/total_time_seconds": 8.87311216443777, "train/time_per_step_avg": 0.03231494732201099, "train/epoch_time_elapsed": 40.886998776346445, "train/estimated_remaining_minutes": 0.420904038569484}
|
| 16 |
+
{"step": 280, "epoch": 0.023592854735422986, "timestamp": 1787159784.913182, "loss": 3.700960159301758, "grad_norm": 0.82421875, "learning_rate": 0.001, "train/total_time_seconds": 9.516451723873615, "train/time_per_step_avg": 0.032303973138332366, "train/epoch_time_elapsed": 42.82041900604963, "train/estimated_remaining_minutes": 0.4078479310231549}
|
| 17 |
+
{"step": 300, "epoch": 0.025278058645096056, "timestamp": 1787159786.7454824, "loss": 3.6359439849853517, "grad_norm": 0.984375, "learning_rate": 0.001, "train/total_time_seconds": 10.157686203718185, "train/time_per_step_avg": 0.032232768349349496, "train/epoch_time_elapsed": 44.65271816775203, "train/estimated_remaining_minutes": 0.3950211301445961}
|
| 18 |
+
{"step": 300, "epoch": 0.025278058645096056, "timestamp": 1787159794.7201555, "eval_loss": 3.5975606441497803, "eval_runtime": 7.973, "eval_samples_per_second": 1194.91, "eval_steps_per_second": 0.878, "train/total_time_seconds": 10.157686203718185, "train/time_per_step_avg": 0.032232768349349496, "train/epoch_time_elapsed": 52.627391546964645, "train/estimated_remaining_minutes": 0.3950211301445961}
|
| 19 |
+
{"step": 320, "epoch": 0.026963262554769128, "timestamp": 1787159796.601689, "loss": 3.5393722534179686, "grad_norm": 0.80859375, "learning_rate": 0.001, "train/total_time_seconds": 10.796786732971668, "train/time_per_step_avg": 0.03215080063790083, "train/epoch_time_elapsed": 54.50892523676157, "train/estimated_remaining_minutes": 0.3823861967927466}
|
| 20 |
+
{"step": 340, "epoch": 0.028648466464442197, "timestamp": 1787159798.453831, "loss": 3.492219924926758, "grad_norm": 0.72265625, "learning_rate": 0.001, "train/total_time_seconds": 11.440050598233938, "train/time_per_step_avg": 0.032119994275271894, "train/epoch_time_elapsed": 56.36106700450182, "train/estimated_remaining_minutes": 0.3701192840605098}
|
| 21 |
+
{"step": 360, "epoch": 0.030333670374115267, "timestamp": 1787159800.2839177, "loss": 3.4430694580078125, "grad_norm": 0.8203125, "learning_rate": 0.001, "train/total_time_seconds": 12.078172758221626, "train/time_per_step_avg": 0.03205060593783855, "train/epoch_time_elapsed": 58.19115437567234, "train/estimated_remaining_minutes": 0.35787178542878895}
|
| 22 |
+
{"step": 380, "epoch": 0.032018874283788336, "timestamp": 1787159802.1708894, "loss": 3.3989883422851563, "grad_norm": 0.8203125, "learning_rate": 0.001, "train/total_time_seconds": 12.717796016484499, "train/time_per_step_avg": 0.03201344292610884, "train/epoch_time_elapsed": 60.07812675833702, "train/estimated_remaining_minutes": 0.34583480395703464}
|
| 23 |
+
{"step": 400, "epoch": 0.03370407819346141, "timestamp": 1787159804.0363617, "loss": 3.341664123535156, "grad_norm": 0.6484375, "learning_rate": 0.001, "train/total_time_seconds": 13.358520355075598, "train/time_per_step_avg": 0.03200834151357412, "train/epoch_time_elapsed": 61.94359891861677, "train/estimated_remaining_minutes": 0.33396300887688996}
|
| 24 |
+
{"step": 400, "epoch": 0.03370407819346141, "timestamp": 1787159812.2065763, "eval_loss": 3.3295202255249023, "eval_runtime": 8.1687, "eval_samples_per_second": 1166.287, "eval_steps_per_second": 0.857, "train/total_time_seconds": 13.358520355075598, "train/time_per_step_avg": 0.03200834151357412, "train/epoch_time_elapsed": 70.11381255835295, "train/estimated_remaining_minutes": 0.33396300887688996}
|
| 25 |
+
{"step": 420, "epoch": 0.03538928210313448, "timestamp": 1787159814.10439, "loss": 3.3047794342041015, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 14.000833839178085, "train/time_per_step_avg": 0.032040471062064174, "train/epoch_time_elapsed": 72.01162526011467, "train/estimated_remaining_minutes": 0.32224141375886073}
|
| 26 |
+
{"step": 440, "epoch": 0.03707448601280755, "timestamp": 1787159816.047692, "loss": 3.258700942993164, "grad_norm": 0.71875, "learning_rate": 0.001, "train/total_time_seconds": 14.651910278946161, "train/time_per_step_avg": 0.03211859680712223, "train/epoch_time_elapsed": 73.95492927730083, "train/estimated_remaining_minutes": 0.31079809682613063}
|
| 27 |
+
{"step": 460, "epoch": 0.03875968992248062, "timestamp": 1787159817.9195716, "loss": 3.229313278198242, "grad_norm": 0.73046875, "learning_rate": 0.001, "train/total_time_seconds": 15.297841779887676, "train/time_per_step_avg": 0.0321966902166605, "train/epoch_time_elapsed": 75.82680872827768, "train/estimated_remaining_minutes": 0.29930560004128065}
|
| 28 |
+
{"step": 480, "epoch": 0.04044489383215369, "timestamp": 1787159819.7689188, "loss": 3.202067565917969, "grad_norm": 0.703125, "learning_rate": 0.001, "train/total_time_seconds": 15.936651539057493, "train/time_per_step_avg": 0.03218855522572994, "train/epoch_time_elapsed": 77.67615579813719, "train/estimated_remaining_minutes": 0.2877450972329825}
|
| 29 |
+
{"step": 500, "epoch": 0.04213009774182676, "timestamp": 1787159821.6968458, "loss": 3.163772201538086, "grad_norm": 0.7265625, "learning_rate": 0.001, "train/total_time_seconds": 16.570955902338028, "train/time_per_step_avg": 0.0321243554726243, "train/epoch_time_elapsed": 79.60408221185207, "train/estimated_remaining_minutes": 0.27618259837230047}
|
| 30 |
+
{"step": 500, "epoch": 0.04213009774182676, "timestamp": 1787159829.648704, "eval_loss": 3.157355308532715, "eval_runtime": 7.9503, "eval_samples_per_second": 1198.315, "eval_steps_per_second": 0.88, "train/total_time_seconds": 16.570955902338028, "train/time_per_step_avg": 0.0321243554726243, "train/epoch_time_elapsed": 87.5559413023293, "train/estimated_remaining_minutes": 0.27618259837230047}
|
| 31 |
+
{"step": 520, "epoch": 0.043815301651499834, "timestamp": 1787159831.4997501, "loss": 3.1448848724365233, "grad_norm": 0.6953125, "learning_rate": 0.001, "train/total_time_seconds": 17.204289257526398, "train/time_per_step_avg": 0.032034554183483124, "train/epoch_time_elapsed": 89.40698843076825, "train/estimated_remaining_minutes": 0.2646813731927138}
|
| 32 |
+
{"step": 540, "epoch": 0.0455005055611729, "timestamp": 1787159833.3223414, "loss": 3.103527069091797, "grad_norm": 0.828125, "learning_rate": 0.001, "train/total_time_seconds": 17.837791569530964, "train/time_per_step_avg": 0.03185881290584803, "train/epoch_time_elapsed": 91.22957941517234, "train/estimated_remaining_minutes": 0.2532525963575384}
|
| 33 |
+
{"step": 560, "epoch": 0.04718570947084597, "timestamp": 1787159835.1489348, "loss": 3.08404541015625, "grad_norm": 0.7578125, "learning_rate": 0.001, "train/total_time_seconds": 18.47138799726963, "train/time_per_step_avg": 0.03173546217381954, "train/epoch_time_elapsed": 93.05617316067219, "train/estimated_remaining_minutes": 0.241887223773769}
|
| 34 |
+
{"step": 580, "epoch": 0.04887091338051904, "timestamp": 1787159836.9735324, "loss": 3.0501741409301757, "grad_norm": 0.76953125, "learning_rate": 0.001, "train/total_time_seconds": 19.10541071370244, "train/time_per_step_avg": 0.03168759174644947, "train/epoch_time_elapsed": 94.88077012076974, "train/estimated_remaining_minutes": 0.23058254309640874}
|
| 35 |
+
{"step": 600, "epoch": 0.05055611729019211, "timestamp": 1787159839.033019, "loss": 3.0376760482788088, "grad_norm": 0.96875, "learning_rate": 0.001, "train/total_time_seconds": 19.750634588301182, "train/time_per_step_avg": 0.03179678685963154, "train/epoch_time_elapsed": 96.94025699794292, "train/estimated_remaining_minutes": 0.2194514954255687}
|
| 36 |
+
{"step": 600, "epoch": 0.05055611729019211, "timestamp": 1787159847.154113, "eval_loss": 3.0310890674591064, "eval_runtime": 8.1195, "eval_samples_per_second": 1173.346, "eval_steps_per_second": 0.862, "train/total_time_seconds": 19.750634588301182, "train/time_per_step_avg": 0.03179678685963154, "train/epoch_time_elapsed": 105.06134972721338, "train/estimated_remaining_minutes": 0.2194514954255687}
|
| 37 |
+
{"step": 620, "epoch": 0.05224132119986518, "timestamp": 1787159849.0143704, "loss": 3.0314205169677733, "grad_norm": 0.82421875, "learning_rate": 0.001, "train/total_time_seconds": 20.382843017578125, "train/time_per_step_avg": 0.03178553760051727, "train/epoch_time_elapsed": 106.92160803079605, "train/estimated_remaining_minutes": 0.2082118372763357}
|
| 38 |
+
{"step": 640, "epoch": 0.053926525109538256, "timestamp": 1787159850.828231, "loss": 3.0014928817749023, "grad_norm": 0.640625, "learning_rate": 0.001, "train/total_time_seconds": 21.01229700446129, "train/time_per_step_avg": 0.03174505434930325, "train/epoch_time_elapsed": 108.73546917364001, "train/estimated_remaining_minutes": 0.19699028441682456}
|
| 39 |
+
{"step": 660, "epoch": 0.055611729019211326, "timestamp": 1787159852.7167454, "loss": 2.9963293075561523, "grad_norm": 0.70703125, "learning_rate": 0.001, "train/total_time_seconds": 21.644355565309525, "train/time_per_step_avg": 0.031729675680398944, "train/epoch_time_elapsed": 110.62398328632116, "train/estimated_remaining_minutes": 0.18583537606578884}
|
| 40 |
+
{"step": 680, "epoch": 0.057296932928884395, "timestamp": 1787159854.5207703, "loss": 2.9568761825561523, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 22.27377400547266, "train/time_per_step_avg": 0.031683632917702195, "train/epoch_time_elapsed": 112.42800731211901, "train/estimated_remaining_minutes": 0.1746962667095895}
|
| 41 |
+
{"step": 700, "epoch": 0.058982136838557464, "timestamp": 1787159856.334935, "loss": 2.9395275115966797, "grad_norm": 0.69921875, "learning_rate": 0.001, "train/total_time_seconds": 22.903755206614733, "train/time_per_step_avg": 0.03153120618313551, "train/epoch_time_elapsed": 114.24217312037945, "train/estimated_remaining_minutes": 0.16359825147581952}
|
| 42 |
+
{"step": 700, "epoch": 0.058982136838557464, "timestamp": 1787159864.3117049, "eval_loss": 2.939429759979248, "eval_runtime": 7.975, "eval_samples_per_second": 1194.602, "eval_steps_per_second": 0.878, "train/total_time_seconds": 22.903755206614733, "train/time_per_step_avg": 0.03153120618313551, "train/epoch_time_elapsed": 122.21894185245037, "train/estimated_remaining_minutes": 0.16359825147581952}
|
| 43 |
+
{"step": 720, "epoch": 0.06066734074823053, "timestamp": 1787159866.154933, "loss": 2.922671890258789, "grad_norm": 0.6484375, "learning_rate": 0.001, "train/total_time_seconds": 23.537149403244257, "train/time_per_step_avg": 0.03154306385666132, "train/epoch_time_elapsed": 124.06217032298446, "train/estimated_remaining_minutes": 0.15255559798399052}
|
| 44 |
+
{"step": 740, "epoch": 0.06235254465790361, "timestamp": 1787159867.9758906, "loss": 2.9145633697509767, "grad_norm": 0.95703125, "learning_rate": 0.001, "train/total_time_seconds": 24.17463242635131, "train/time_per_step_avg": 0.0316233542189002, "train/epoch_time_elapsed": 125.88312843069434, "train/estimated_remaining_minutes": 0.14156316285701215}
|
| 45 |
+
{"step": 760, "epoch": 0.06403774856757667, "timestamp": 1787159869.9564908, "loss": 2.8876668930053713, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 24.810363072901964, "train/time_per_step_avg": 0.031660075075924395, "train/epoch_time_elapsed": 127.86372835934162, "train/estimated_remaining_minutes": 0.13058085827843138}
|
| 46 |
+
{"step": 780, "epoch": 0.06572295247724974, "timestamp": 1787159871.839654, "loss": 2.8692466735839846, "grad_norm": 0.69140625, "learning_rate": 0.001, "train/total_time_seconds": 25.449912142008543, "train/time_per_step_avg": 0.03176138136535883, "train/epoch_time_elapsed": 129.74689135327935, "train/estimated_remaining_minutes": 0.11963633912909998}
|
| 47 |
+
{"step": 800, "epoch": 0.06740815638692282, "timestamp": 1787159873.668295, "loss": 2.8788633346557617, "grad_norm": 0.68359375, "learning_rate": 0.001, "train/total_time_seconds": 26.088915783911943, "train/time_per_step_avg": 0.03185160577297211, "train/epoch_time_elapsed": 131.57553120702505, "train/estimated_remaining_minutes": 0.10870381576629977}
|
| 48 |
+
{"step": 800, "epoch": 0.06740815638692282, "timestamp": 1787159881.9473228, "eval_loss": 2.8632190227508545, "eval_runtime": 8.2773, "eval_samples_per_second": 1150.973, "eval_steps_per_second": 0.846, "train/total_time_seconds": 26.088915783911943, "train/time_per_step_avg": 0.03185160577297211, "train/epoch_time_elapsed": 139.8545593805611, "train/estimated_remaining_minutes": 0.10870381576629977}
|
| 49 |
+
{"step": 820, "epoch": 0.0690933602965959, "timestamp": 1787159883.8665261, "loss": 2.8496471405029298, "grad_norm": 0.6796875, "learning_rate": 0.001, "train/total_time_seconds": 26.722276385873556, "train/time_per_step_avg": 0.031851269826292994, "train/epoch_time_elapsed": 141.77376406639814, "train/estimated_remaining_minutes": 0.09776442580197643}
|
| 50 |
+
{"step": 840, "epoch": 0.07077856420626896, "timestamp": 1787159885.6946013, "loss": 2.847636604309082, "grad_norm": 0.62109375, "learning_rate": 0.001, "train/total_time_seconds": 27.35644706711173, "train/time_per_step_avg": 0.03181814640760422, "train/epoch_time_elapsed": 143.60183906927705, "train/estimated_remaining_minutes": 0.0868458637051166}
|
| 51 |
+
{"step": 860, "epoch": 0.07246376811594203, "timestamp": 1787159887.564113, "loss": 2.8339916229248048, "grad_norm": 0.73046875, "learning_rate": 0.001, "train/total_time_seconds": 27.989167381078005, "train/time_per_step_avg": 0.031788043081760406, "train/epoch_time_elapsed": 145.47135097533464, "train/estimated_remaining_minutes": 0.07593960142152947}
|
| 52 |
+
{"step": 880, "epoch": 0.0741489720256151, "timestamp": 1787159889.387239, "loss": 2.8253028869628904, "grad_norm": 0.69921875, "learning_rate": 0.001, "train/total_time_seconds": 28.62036318704486, "train/time_per_step_avg": 0.03170451045036316, "train/epoch_time_elapsed": 147.29447646439075, "train/estimated_remaining_minutes": 0.0650462799705565}
|
| 53 |
+
{"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159891.2191205, "loss": 2.7853120803833007, "grad_norm": 0.6875, "learning_rate": 0.001, "train/total_time_seconds": 29.254599027335644, "train/time_per_step_avg": 0.031656832434237, "train/epoch_time_elapsed": 149.1263581365347, "train/estimated_remaining_minutes": 0.0541751833839549}
|
| 54 |
+
{"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159899.179946, "eval_loss": 2.8034849166870117, "eval_runtime": 7.9593, "eval_samples_per_second": 1196.967, "eval_steps_per_second": 0.879, "train/total_time_seconds": 29.254599027335644, "train/time_per_step_avg": 0.031656832434237, "train/epoch_time_elapsed": 157.0871831253171, "train/estimated_remaining_minutes": 0.0541751833839549}
|
| 55 |
+
{"step": 920, "epoch": 0.07751937984496124, "timestamp": 1787159901.2376366, "loss": 2.794095993041992, "grad_norm": 0.609375, "learning_rate": 0.001, "train/total_time_seconds": 29.887034360319376, "train/time_per_step_avg": 0.0316475797444582, "train/epoch_time_elapsed": 159.14487295597792, "train/estimated_remaining_minutes": 0.0433145425511875}
|
| 56 |
+
{"step": 940, "epoch": 0.07920458375463431, "timestamp": 1787159903.067668, "loss": 2.7890634536743164, "grad_norm": 0.671875, "learning_rate": 0.001, "train/total_time_seconds": 30.5212767906487, "train/time_per_step_avg": 0.03164829723536968, "train/epoch_time_elapsed": 160.9749058149755, "train/estimated_remaining_minutes": 0.03246944339430713}
|
| 57 |
+
{"step": 960, "epoch": 0.08088978766430738, "timestamp": 1787159904.9205842, "loss": 2.784823989868164, "grad_norm": 0.58203125, "learning_rate": 0.001, "train/total_time_seconds": 31.156463339924812, "train/time_per_step_avg": 0.03167295958846807, "train/epoch_time_elapsed": 162.82782202214003, "train/estimated_remaining_minutes": 0.021636432874947785}
|
| 58 |
+
{"step": 980, "epoch": 0.08257499157398045, "timestamp": 1787159906.7382162, "loss": 2.775278663635254, "grad_norm": 0.60546875, "learning_rate": 0.001, "train/total_time_seconds": 31.790426205843687, "train/time_per_step_avg": 0.03170063018798828, "train/epoch_time_elapsed": 164.64545308053493, "train/estimated_remaining_minutes": 0.010813070138042068}
|
| 59 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159908.6763556, "loss": 2.753749656677246, "grad_norm": 0.78125, "learning_rate": 0.001, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 166.58359253406525, "train/estimated_remaining_minutes": 0.0}
|
| 60 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159916.6100245, "eval_loss": 2.7597591876983643, "eval_runtime": 7.9273, "eval_samples_per_second": 1201.797, "eval_steps_per_second": 0.883, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 174.5172589495778, "train/estimated_remaining_minutes": 0.0}
|
| 61 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159916.6630983, "train_runtime": 175.4068, "train_samples_per_second": 456.083, "train_steps_per_second": 5.701, "total_flos": 362985553920000.0, "train_loss": 3.6923015975952147, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 174.57033431902528, "train/estimated_remaining_minutes": 0.0}
|
| 62 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159924.561638, "eval_loss": 2.7597591876983643, "eval_runtime": 7.8957, "eval_samples_per_second": 1206.607, "eval_steps_per_second": 0.887, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 182.46887450292706, "train/estimated_remaining_minutes": 0.0}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
zain/Activation/out/sweep_summary.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
-
"variant": "mlp-linear-
|
| 4 |
-
"eval_loss": 2.
|
| 5 |
-
"out": "out/mlp-linear-
|
| 6 |
-
"run_name": "LM-mlp-linear-
|
| 7 |
"status": "success"
|
| 8 |
}
|
| 9 |
]
|
|
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
+
"variant": "mlp-linear-9L",
|
| 4 |
+
"eval_loss": 2.7597591876983643,
|
| 5 |
+
"out": "out/mlp-linear-9L_run",
|
| 6 |
+
"run_name": "LM-mlp-linear-9L-2.0M-20260819-171540",
|
| 7 |
"status": "success"
|
| 8 |
}
|
| 9 |
]
|
zain/Activation/wandb/debug-internal.log
CHANGED
|
@@ -7,3 +7,35 @@
|
|
| 7 |
{"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
|
| 8 |
{"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
|
| 9 |
{"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
{"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
|
| 8 |
{"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
|
| 9 |
{"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 10 |
+
{"time":"2026-08-19T17:15:57.093327808Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":5,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":9,"uploaded_len":2}
|
| 11 |
+
{"time":"2026-08-19T17:15:57.208750755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 12 |
+
{"time":"2026-08-19T17:16:12.09424474Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":6,"events_offset":1,"events_lines":2,"console_offset":2,"console_lines":1}
|
| 13 |
+
{"time":"2026-08-19T17:16:12.254954463Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 14 |
+
{"time":"2026-08-19T17:16:27.093741092Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":6,"events_offset":3,"events_lines":2,"console_offset":8,"console_lines":22}
|
| 15 |
+
{"time":"2026-08-19T17:16:27.218480793Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 16 |
+
{"time":"2026-08-19T17:16:42.09555922Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":4,"events_offset":5,"events_lines":2,"console_offset":24,"console_lines":1}
|
| 17 |
+
{"time":"2026-08-19T17:16:42.264597826Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 18 |
+
{"time":"2026-08-19T17:16:57.09386987Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":5,"events_offset":7,"events_lines":2,"console_offset":30,"console_lines":19}
|
| 19 |
+
{"time":"2026-08-19T17:16:57.209800875Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 20 |
+
{"time":"2026-08-19T17:17:12.094049325Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":5,"events_offset":9,"events_lines":2,"console_offset":46,"console_lines":1}
|
| 21 |
+
{"time":"2026-08-19T17:17:12.252024387Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 22 |
+
{"time":"2026-08-19T17:17:27.093470433Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":4,"events_offset":11,"events_lines":2,"console_offset":49,"console_lines":15}
|
| 23 |
+
{"time":"2026-08-19T17:17:27.202702289Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
+
{"time":"2026-08-19T17:17:42.093533111Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":57,"console_lines":1}
|
| 25 |
+
{"time":"2026-08-19T17:17:42.240909652Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 26 |
+
{"time":"2026-08-19T17:17:57.093850562Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":63,"console_lines":23}
|
| 27 |
+
{"time":"2026-08-19T17:17:57.20571167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 28 |
+
{"time":"2026-08-19T17:18:12.094059821Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":17,"events_lines":2,"console_offset":79,"console_lines":1}
|
| 29 |
+
{"time":"2026-08-19T17:18:12.283285285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 30 |
+
{"time":"2026-08-19T17:18:27.0939628Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":5,"events_offset":19,"events_lines":2,"console_offset":85,"console_lines":21}
|
| 31 |
+
{"time":"2026-08-19T17:18:27.220681676Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 32 |
+
{"time":"2026-08-19T17:18:42.09374958Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":3,"events_offset":21,"events_lines":2,"console_offset":101,"console_lines":1}
|
| 33 |
+
{"time":"2026-08-19T17:18:42.218715049Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 34 |
+
{"time":"2026-08-19T17:18:45.014142777Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 35 |
+
{"time":"2026-08-19T17:18:45.014366349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":23,"events_lines":1,"console_offset":106,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
|
| 36 |
+
{"time":"2026-08-19T17:18:45.153108654Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 37 |
+
{"time":"2026-08-19T17:18:45.154305649Z","level":"INFO","msg":"handler: operation stats","stats":{}}
|
| 38 |
+
{"time":"2026-08-19T17:18:45.157232952Z","level":"INFO","msg":"stream: finishing up"}
|
| 39 |
+
{"time":"2026-08-19T17:18:45.157265556Z","level":"INFO","msg":"handler: closed"}
|
| 40 |
+
{"time":"2026-08-19T17:18:45.161743858Z","level":"INFO","msg":"sender: closed"}
|
| 41 |
+
{"time":"2026-08-19T17:18:45.161763043Z","level":"INFO","msg":"stream: all finished"}
|
zain/Activation/wandb/debug.log
CHANGED
|
@@ -21,3 +21,8 @@ config: {'_wandb': {}}
|
|
| 21 |
2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
|
| 23 |
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 21 |
2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
|
| 23 |
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
|
| 24 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate-checking/bhq23mh5
|
| 25 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
|
| 26 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2570] restore
|
| 27 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2576] restore done
|
| 28 |
+
2026-08-19 17:18:45,156 INFO MainThread:3953242 [wandb_run.py:_footer_sync_info():3993] logging synced files
|
zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log
CHANGED
|
@@ -65,3 +65,12 @@
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log
CHANGED
|
@@ -65,3 +65,12 @@
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log
CHANGED
|
@@ -65,3 +65,12 @@
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log
CHANGED
|
@@ -65,3 +65,12 @@
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_171215-2nil4rqn/logs/debug-core.log
CHANGED
|
@@ -62,3 +62,15 @@
|
|
| 62 |
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
| 65 |
+
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
+
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
+
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/config.yaml
ADDED
|
@@ -0,0 +1,435 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_name_or_path:
|
| 2 |
+
value: ""
|
| 3 |
+
_wandb:
|
| 4 |
+
value:
|
| 5 |
+
cli_version: 0.28.1
|
| 6 |
+
e:
|
| 7 |
+
2l4ndnmtei7frfwveku3d762xxak1bdj:
|
| 8 |
+
args:
|
| 9 |
+
- --config
|
| 10 |
+
- /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml
|
| 11 |
+
- --variants
|
| 12 |
+
- mlp-linear-9L
|
| 13 |
+
codePath: sweep.py
|
| 14 |
+
codePathLocal: sweep.py
|
| 15 |
+
cpu_count: 112
|
| 16 |
+
cpu_count_logical: 224
|
| 17 |
+
cudaVersion: "12.4"
|
| 18 |
+
disk:
|
| 19 |
+
/:
|
| 20 |
+
total: "1560765693952"
|
| 21 |
+
used: "708485218304"
|
| 22 |
+
email: deepnevro@gmail.com
|
| 23 |
+
executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python
|
| 24 |
+
git:
|
| 25 |
+
commit: 463c9961366755fc55f02df9a0d471b3cbcb025e
|
| 26 |
+
remote: https://github.com/w-ahmad1a10/Activation.git
|
| 27 |
+
gpu: NVIDIA H100 80GB HBM3
|
| 28 |
+
gpu_count: 8
|
| 29 |
+
gpu_nvidia:
|
| 30 |
+
- architecture: Hopper
|
| 31 |
+
cudaCores: 16896
|
| 32 |
+
memoryTotal: "85520809984"
|
| 33 |
+
name: NVIDIA H100 80GB HBM3
|
| 34 |
+
uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae
|
| 35 |
+
- architecture: Hopper
|
| 36 |
+
cudaCores: 16896
|
| 37 |
+
memoryTotal: "85520809984"
|
| 38 |
+
name: NVIDIA H100 80GB HBM3
|
| 39 |
+
uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3
|
| 40 |
+
- architecture: Hopper
|
| 41 |
+
cudaCores: 16896
|
| 42 |
+
memoryTotal: "85520809984"
|
| 43 |
+
name: NVIDIA H100 80GB HBM3
|
| 44 |
+
uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab
|
| 45 |
+
- architecture: Hopper
|
| 46 |
+
cudaCores: 16896
|
| 47 |
+
memoryTotal: "85520809984"
|
| 48 |
+
name: NVIDIA H100 80GB HBM3
|
| 49 |
+
uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864
|
| 50 |
+
- architecture: Hopper
|
| 51 |
+
cudaCores: 16896
|
| 52 |
+
memoryTotal: "85520809984"
|
| 53 |
+
name: NVIDIA H100 80GB HBM3
|
| 54 |
+
uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef
|
| 55 |
+
- architecture: Hopper
|
| 56 |
+
cudaCores: 16896
|
| 57 |
+
memoryTotal: "85520809984"
|
| 58 |
+
name: NVIDIA H100 80GB HBM3
|
| 59 |
+
uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54
|
| 60 |
+
- architecture: Hopper
|
| 61 |
+
cudaCores: 16896
|
| 62 |
+
memoryTotal: "85520809984"
|
| 63 |
+
name: NVIDIA H100 80GB HBM3
|
| 64 |
+
uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9
|
| 65 |
+
- architecture: Hopper
|
| 66 |
+
cudaCores: 16896
|
| 67 |
+
memoryTotal: "85520809984"
|
| 68 |
+
name: NVIDIA H100 80GB HBM3
|
| 69 |
+
uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea
|
| 70 |
+
host: deeplens-k3s-node1
|
| 71 |
+
memory:
|
| 72 |
+
total: "2164089937920"
|
| 73 |
+
os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35
|
| 74 |
+
program: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py
|
| 75 |
+
python: CPython 3.11.15
|
| 76 |
+
root: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation
|
| 77 |
+
startedAt: "2026-08-19T17:15:41.408834Z"
|
| 78 |
+
writerId: 2l4ndnmtei7frfwveku3d762xxak1bdj
|
| 79 |
+
m:
|
| 80 |
+
- "1": train/global_step
|
| 81 |
+
"6":
|
| 82 |
+
- 3
|
| 83 |
+
"7": []
|
| 84 |
+
- "2": '*'
|
| 85 |
+
"5": 1
|
| 86 |
+
"6":
|
| 87 |
+
- 1
|
| 88 |
+
"7": []
|
| 89 |
+
python_version: 3.11.15
|
| 90 |
+
t:
|
| 91 |
+
"1":
|
| 92 |
+
- 1
|
| 93 |
+
- 5
|
| 94 |
+
- 11
|
| 95 |
+
- 41
|
| 96 |
+
- 49
|
| 97 |
+
- 51
|
| 98 |
+
- 53
|
| 99 |
+
- 71
|
| 100 |
+
"2":
|
| 101 |
+
- 1
|
| 102 |
+
- 5
|
| 103 |
+
- 11
|
| 104 |
+
- 41
|
| 105 |
+
- 49
|
| 106 |
+
- 51
|
| 107 |
+
- 53
|
| 108 |
+
- 71
|
| 109 |
+
"3":
|
| 110 |
+
- 2
|
| 111 |
+
- 7
|
| 112 |
+
- 13
|
| 113 |
+
- 19
|
| 114 |
+
- 62
|
| 115 |
+
- 66
|
| 116 |
+
"4": 3.11.15
|
| 117 |
+
"5": 0.28.1
|
| 118 |
+
"6": 5.16.0.dev0
|
| 119 |
+
"9":
|
| 120 |
+
"1": transformers_trainer
|
| 121 |
+
"12": 0.28.1
|
| 122 |
+
"13": linux-x86_64
|
| 123 |
+
accelerator_config:
|
| 124 |
+
value:
|
| 125 |
+
dispatch_batches: null
|
| 126 |
+
even_batches: true
|
| 127 |
+
gradient_accumulation_kwargs: null
|
| 128 |
+
non_blocking: false
|
| 129 |
+
split_batches: false
|
| 130 |
+
use_seedable_sampler: true
|
| 131 |
+
activation:
|
| 132 |
+
value: linear
|
| 133 |
+
adam_beta1:
|
| 134 |
+
value: 0.9
|
| 135 |
+
adam_beta2:
|
| 136 |
+
value: 0.999
|
| 137 |
+
adam_epsilon:
|
| 138 |
+
value: 1e-08
|
| 139 |
+
architectures:
|
| 140 |
+
value: null
|
| 141 |
+
attention_bias:
|
| 142 |
+
value: false
|
| 143 |
+
attention_dropout:
|
| 144 |
+
value: 0
|
| 145 |
+
auto_find_batch_size:
|
| 146 |
+
value: false
|
| 147 |
+
average_tokens_across_devices:
|
| 148 |
+
value: true
|
| 149 |
+
batch_eval_metrics:
|
| 150 |
+
value: false
|
| 151 |
+
bf16:
|
| 152 |
+
value: true
|
| 153 |
+
bf16_full_eval:
|
| 154 |
+
value: false
|
| 155 |
+
bos_token_id:
|
| 156 |
+
value: 1
|
| 157 |
+
chunk_size_feed_forward:
|
| 158 |
+
value: 0
|
| 159 |
+
data_seed:
|
| 160 |
+
value: 42
|
| 161 |
+
dataloader_drop_last:
|
| 162 |
+
value: false
|
| 163 |
+
dataloader_in_order:
|
| 164 |
+
value: true
|
| 165 |
+
dataloader_multiprocessing_context:
|
| 166 |
+
value: null
|
| 167 |
+
dataloader_num_workers:
|
| 168 |
+
value: 0
|
| 169 |
+
dataloader_persistent_workers:
|
| 170 |
+
value: false
|
| 171 |
+
dataloader_pin_memory:
|
| 172 |
+
value: true
|
| 173 |
+
dataloader_prefetch_factor:
|
| 174 |
+
value: null
|
| 175 |
+
ddp_backend:
|
| 176 |
+
value: null
|
| 177 |
+
ddp_broadcast_buffers:
|
| 178 |
+
value: null
|
| 179 |
+
ddp_bucket_cap_mb:
|
| 180 |
+
value: null
|
| 181 |
+
ddp_find_unused_parameters:
|
| 182 |
+
value: null
|
| 183 |
+
ddp_static_graph:
|
| 184 |
+
value: null
|
| 185 |
+
ddp_timeout:
|
| 186 |
+
value: 1800
|
| 187 |
+
debug:
|
| 188 |
+
value: []
|
| 189 |
+
deepspeed:
|
| 190 |
+
value: null
|
| 191 |
+
disable_tqdm:
|
| 192 |
+
value: false
|
| 193 |
+
do_eval:
|
| 194 |
+
value: true
|
| 195 |
+
do_predict:
|
| 196 |
+
value: false
|
| 197 |
+
do_train:
|
| 198 |
+
value: false
|
| 199 |
+
dtype:
|
| 200 |
+
value: null
|
| 201 |
+
enable_jit_checkpoint:
|
| 202 |
+
value: false
|
| 203 |
+
eos_token_id:
|
| 204 |
+
value: 2
|
| 205 |
+
eval_accumulation_steps:
|
| 206 |
+
value: null
|
| 207 |
+
eval_delay:
|
| 208 |
+
value: 0
|
| 209 |
+
eval_do_concat_batches:
|
| 210 |
+
value: true
|
| 211 |
+
eval_on_start:
|
| 212 |
+
value: false
|
| 213 |
+
eval_steps:
|
| 214 |
+
value: 100
|
| 215 |
+
eval_strategy:
|
| 216 |
+
value: steps
|
| 217 |
+
eval_use_gather_object:
|
| 218 |
+
value: false
|
| 219 |
+
fp16:
|
| 220 |
+
value: false
|
| 221 |
+
fp16_full_eval:
|
| 222 |
+
value: false
|
| 223 |
+
fsdp:
|
| 224 |
+
value: null
|
| 225 |
+
fsdp_config:
|
| 226 |
+
value: null
|
| 227 |
+
full_determinism:
|
| 228 |
+
value: false
|
| 229 |
+
gradient_accumulation_steps:
|
| 230 |
+
value: 1
|
| 231 |
+
gradient_checkpointing:
|
| 232 |
+
value: false
|
| 233 |
+
gradient_checkpointing_kwargs:
|
| 234 |
+
value: null
|
| 235 |
+
greater_is_better:
|
| 236 |
+
value: null
|
| 237 |
+
head_dim:
|
| 238 |
+
value: 32
|
| 239 |
+
hidden_act:
|
| 240 |
+
value: silu
|
| 241 |
+
hidden_size:
|
| 242 |
+
value: 128
|
| 243 |
+
hub_always_push:
|
| 244 |
+
value: false
|
| 245 |
+
hub_model_id:
|
| 246 |
+
value: w-ahmad/6L-mlp-linear-9L
|
| 247 |
+
hub_private_repo:
|
| 248 |
+
value: null
|
| 249 |
+
hub_revision:
|
| 250 |
+
value: null
|
| 251 |
+
hub_strategy:
|
| 252 |
+
value: every_save
|
| 253 |
+
hub_token:
|
| 254 |
+
value: <HUB_TOKEN>
|
| 255 |
+
id2label:
|
| 256 |
+
value:
|
| 257 |
+
"0": LABEL_0
|
| 258 |
+
"1": LABEL_1
|
| 259 |
+
ignore_data_skip:
|
| 260 |
+
value: false
|
| 261 |
+
include_for_metrics:
|
| 262 |
+
value: []
|
| 263 |
+
include_num_input_tokens_seen:
|
| 264 |
+
value: "no"
|
| 265 |
+
initializer_range:
|
| 266 |
+
value: 0.02
|
| 267 |
+
intermediate_size:
|
| 268 |
+
value: 256
|
| 269 |
+
is_encoder_decoder:
|
| 270 |
+
value: false
|
| 271 |
+
label_names:
|
| 272 |
+
value: null
|
| 273 |
+
label_smoothing_factor:
|
| 274 |
+
value: 0
|
| 275 |
+
label2id:
|
| 276 |
+
value:
|
| 277 |
+
LABEL_0: 0
|
| 278 |
+
LABEL_1: 1
|
| 279 |
+
learning_rate:
|
| 280 |
+
value: 0.001
|
| 281 |
+
length_column_name:
|
| 282 |
+
value: length
|
| 283 |
+
liger_kernel_config:
|
| 284 |
+
value: null
|
| 285 |
+
load_best_model_at_end:
|
| 286 |
+
value: false
|
| 287 |
+
local_rank:
|
| 288 |
+
value: -1
|
| 289 |
+
log_level:
|
| 290 |
+
value: passive
|
| 291 |
+
log_level_replica:
|
| 292 |
+
value: warning
|
| 293 |
+
log_on_each_node:
|
| 294 |
+
value: true
|
| 295 |
+
logging_first_step:
|
| 296 |
+
value: false
|
| 297 |
+
logging_nan_inf_filter:
|
| 298 |
+
value: true
|
| 299 |
+
logging_steps:
|
| 300 |
+
value: 20
|
| 301 |
+
logging_strategy:
|
| 302 |
+
value: steps
|
| 303 |
+
lr_scheduler_kwargs:
|
| 304 |
+
value: null
|
| 305 |
+
lr_scheduler_type:
|
| 306 |
+
value: constant_with_warmup
|
| 307 |
+
max_grad_norm:
|
| 308 |
+
value: 1
|
| 309 |
+
max_position_embeddings:
|
| 310 |
+
value: 512
|
| 311 |
+
max_steps:
|
| 312 |
+
value: 1000
|
| 313 |
+
metric_for_best_model:
|
| 314 |
+
value: null
|
| 315 |
+
mlp_bias:
|
| 316 |
+
value: false
|
| 317 |
+
mlp_type:
|
| 318 |
+
value: mlp
|
| 319 |
+
model/num_parameters:
|
| 320 |
+
value: 2001280
|
| 321 |
+
model_type:
|
| 322 |
+
value: tiny_llama
|
| 323 |
+
neftune_noise_alpha:
|
| 324 |
+
value: null
|
| 325 |
+
num_attention_heads:
|
| 326 |
+
value: 4
|
| 327 |
+
num_hidden_layers:
|
| 328 |
+
value: 9
|
| 329 |
+
num_key_value_heads:
|
| 330 |
+
value: 4
|
| 331 |
+
num_train_epochs:
|
| 332 |
+
value: 1
|
| 333 |
+
optim:
|
| 334 |
+
value: adamw_torch_fused
|
| 335 |
+
optim_args:
|
| 336 |
+
value: null
|
| 337 |
+
optim_target_modules:
|
| 338 |
+
value: null
|
| 339 |
+
output_attentions:
|
| 340 |
+
value: false
|
| 341 |
+
output_dir:
|
| 342 |
+
value: out/mlp-linear-9L_run
|
| 343 |
+
output_hidden_states:
|
| 344 |
+
value: false
|
| 345 |
+
pad_token_id:
|
| 346 |
+
value: 0
|
| 347 |
+
parallelism_config:
|
| 348 |
+
value: null
|
| 349 |
+
per_device_eval_batch_size:
|
| 350 |
+
value: 1500
|
| 351 |
+
per_device_train_batch_size:
|
| 352 |
+
value: 80
|
| 353 |
+
powlu_m:
|
| 354 |
+
value: 3
|
| 355 |
+
prediction_loss_only:
|
| 356 |
+
value: false
|
| 357 |
+
pretraining_tp:
|
| 358 |
+
value: 1
|
| 359 |
+
problem_type:
|
| 360 |
+
value: null
|
| 361 |
+
project:
|
| 362 |
+
value: huggingface
|
| 363 |
+
push_to_hub:
|
| 364 |
+
value: false
|
| 365 |
+
remove_unused_columns:
|
| 366 |
+
value: false
|
| 367 |
+
report_to:
|
| 368 |
+
value:
|
| 369 |
+
- wandb
|
| 370 |
+
restore_callback_states_from_checkpoint:
|
| 371 |
+
value: false
|
| 372 |
+
resume_from_checkpoint:
|
| 373 |
+
value: null
|
| 374 |
+
return_dict:
|
| 375 |
+
value: true
|
| 376 |
+
rms_norm_eps:
|
| 377 |
+
value: 1e-06
|
| 378 |
+
rope_parameters:
|
| 379 |
+
value:
|
| 380 |
+
rope_theta: 10000
|
| 381 |
+
rope_type: default
|
| 382 |
+
run_name:
|
| 383 |
+
value: LM-mlp-linear-9L-2.0M-20260819-171540
|
| 384 |
+
save_on_each_node:
|
| 385 |
+
value: false
|
| 386 |
+
save_only_model:
|
| 387 |
+
value: false
|
| 388 |
+
save_steps:
|
| 389 |
+
value: 100
|
| 390 |
+
save_strategy:
|
| 391 |
+
value: steps
|
| 392 |
+
save_total_limit:
|
| 393 |
+
value: null
|
| 394 |
+
seed:
|
| 395 |
+
value: 42
|
| 396 |
+
skip_memory_metrics:
|
| 397 |
+
value: true
|
| 398 |
+
tf32:
|
| 399 |
+
value: null
|
| 400 |
+
tie_word_embeddings:
|
| 401 |
+
value: true
|
| 402 |
+
tokenizer_name:
|
| 403 |
+
value: w-ahmad/tiny-stories-tokenizer
|
| 404 |
+
torch_compile:
|
| 405 |
+
value: false
|
| 406 |
+
torch_compile_backend:
|
| 407 |
+
value: null
|
| 408 |
+
torch_compile_mode:
|
| 409 |
+
value: null
|
| 410 |
+
torch_empty_cache_steps:
|
| 411 |
+
value: null
|
| 412 |
+
trackio_bucket_id:
|
| 413 |
+
value: null
|
| 414 |
+
trackio_space_id:
|
| 415 |
+
value: null
|
| 416 |
+
trackio_static_space_id:
|
| 417 |
+
value: null
|
| 418 |
+
train_sampling_strategy:
|
| 419 |
+
value: random
|
| 420 |
+
transformers_version:
|
| 421 |
+
value: 5.16.0.dev0
|
| 422 |
+
use_cache:
|
| 423 |
+
value: false
|
| 424 |
+
use_cpu:
|
| 425 |
+
value: false
|
| 426 |
+
use_liger_kernel:
|
| 427 |
+
value: false
|
| 428 |
+
vocab_size:
|
| 429 |
+
value: 4096
|
| 430 |
+
waleed_beta:
|
| 431 |
+
value: 10
|
| 432 |
+
warmup_steps:
|
| 433 |
+
value: 200
|
| 434 |
+
weight_decay:
|
| 435 |
+
value: 0
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/output.log
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
0%| | 0/1000 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
|
| 2 |
+
[INFO] Causal mask (float with -inf) applied to all attention layers.
|
| 3 |
+
10%|█ | 100/1000 [00:18<01:34, 9.55[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 4 |
+
{'loss': '8.201', 'grad_norm': '1.242', 'learning_rate': '9.5e-05', 'epoch': '0.001685', 'train/total_time_seconds': '1.182', 'train/time_per_step_avg': '0.0591', 'train/epoch_time_elapsed': '2.652', 'train/estimated_remaining_minutes': '0.9653'}
|
| 5 |
+
{'loss': '7.765', 'grad_norm': '1.242', 'learning_rate': '0.000195', 'epoch': '0.00337', 'train/total_time_seconds': '1.819', 'train/time_per_step_avg': '0.04547', 'train/epoch_time_elapsed': '4.584', 'train/estimated_remaining_minutes': '0.7275'}
|
| 6 |
+
{'loss': '7.16', 'grad_norm': '1.227', 'learning_rate': '0.000295', 'epoch': '0.005056', 'train/total_time_seconds': '2.454', 'train/time_per_step_avg': '0.0409', 'train/epoch_time_elapsed': '6.432', 'train/estimated_remaining_minutes': '0.6408'}
|
| 7 |
+
{'loss': '6.481', 'grad_norm': '1.211', 'learning_rate': '0.000395', 'epoch': '0.006741', 'train/total_time_seconds': '3.086', 'train/time_per_step_avg': '0.03858', 'train/epoch_time_elapsed': '8.246', 'train/estimated_remaining_minutes': '0.5916'}
|
| 8 |
+
{'loss': '5.919', 'grad_norm': '1.008', 'learning_rate': '0.000495', 'epoch': '0.008426', 'train/total_time_seconds': '3.722', 'train/time_per_step_avg': '0.03722', 'train/epoch_time_elapsed': '10.15', 'train/estimated_remaining_minutes': '0.5584'}
|
| 9 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 10 |
+
{'eval_loss': '5.634', 'eval_runtime': '7.97', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.008426', 'train/total_time_seconds': '3.722', 'train/time_per_step_avg': '0.03722', 'train/epoch_time_elapsed': '18.12', 'train/estimated_remaining_minutes': '0.5584'}
|
| 11 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 12 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 13 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 93.80it/s]
|
| 14 |
+
20%|██ | 200/1000 [00:35<01:13, 10.84[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 15 |
+
{'loss': '5.413', 'grad_norm': '1.172', 'learning_rate': '0.000595', 'epoch': '0.01011', 'train/total_time_seconds': '4.364', 'train/time_per_step_avg': '0.03182', 'train/epoch_time_elapsed': '20.01', 'train/estimated_remaining_minutes': '0.5334'}
|
| 16 |
+
{'loss': '5.033', 'grad_norm': '1.516', 'learning_rate': '0.000695', 'epoch': '0.0118', 'train/total_time_seconds': '5.002', 'train/time_per_step_avg': '0.03183', 'train/epoch_time_elapsed': '21.84', 'train/estimated_remaining_minutes': '0.5121'}
|
| 17 |
+
{'loss': '4.695', 'grad_norm': '0.9766', 'learning_rate': '0.000795', 'epoch': '0.01348', 'train/total_time_seconds': '5.642', 'train/time_per_step_avg': '0.03187', 'train/epoch_time_elapsed': '23.68', 'train/estimated_remaining_minutes': '0.4936'}
|
| 18 |
+
{'loss': '4.425', 'grad_norm': '0.8828', 'learning_rate': '0.000895', 'epoch': '0.01517', 'train/total_time_seconds': '6.286', 'train/time_per_step_avg': '0.032', 'train/epoch_time_elapsed': '25.54', 'train/estimated_remaining_minutes': '0.4773'}
|
| 19 |
+
{'loss': '4.205', 'grad_norm': '0.8789', 'learning_rate': '0.000995', 'epoch': '0.01685', 'train/total_time_seconds': '6.934', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '27.38', 'train/estimated_remaining_minutes': '0.4623'}
|
| 20 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 21 |
+
{'eval_loss': '4.133', 'eval_runtime': '7.933', 'eval_samples_per_second': '1201', 'eval_steps_per_second': '0.882', 'epoch': '0.01685', 'train/total_time_seconds': '6.934', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '35.32', 'train/estimated_remaining_minutes': '0.4623'}
|
| 22 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 23 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 24 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 114.05it/s]
|
| 25 |
+
30%|███ | 300/1000 [00:52<01:04, 10.91[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 26 |
+
{'loss': '4.049', 'grad_norm': '0.6875', 'learning_rate': '0.001', 'epoch': '0.01854', 'train/total_time_seconds': '7.582', 'train/time_per_step_avg': '0.03218', 'train/epoch_time_elapsed': '37.19', 'train/estimated_remaining_minutes': '0.448'}
|
| 27 |
+
{'loss': '3.919', 'grad_norm': '0.7383', 'learning_rate': '0.001', 'epoch': '0.02022', 'train/total_time_seconds': '8.228', 'train/time_per_step_avg': '0.03226', 'train/epoch_time_elapsed': '39.05', 'train/estimated_remaining_minutes': '0.4343'}
|
| 28 |
+
{'loss': '3.783', 'grad_norm': '0.6836', 'learning_rate': '0.001', 'epoch': '0.02191', 'train/total_time_seconds': '8.873', 'train/time_per_step_avg': '0.03231', 'train/epoch_time_elapsed': '40.89', 'train/estimated_remaining_minutes': '0.4209'}
|
| 29 |
+
{'loss': '3.701', 'grad_norm': '0.8242', 'learning_rate': '0.001', 'epoch': '0.02359', 'train/total_time_seconds': '9.516', 'train/time_per_step_avg': '0.0323', 'train/epoch_time_elapsed': '42.82', 'train/estimated_remaining_minutes': '0.4078'}
|
| 30 |
+
{'loss': '3.636', 'grad_norm': '0.9844', 'learning_rate': '0.001', 'epoch': '0.02528', 'train/total_time_seconds': '10.16', 'train/time_per_step_avg': '0.03223', 'train/epoch_time_elapsed': '44.65', 'train/estimated_remaining_minutes': '0.395'}
|
| 31 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 32 |
+
{'eval_loss': '3.598', 'eval_runtime': '7.973', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.02528', 'train/total_time_seconds': '10.16', 'train/time_per_step_avg': '0.03223', 'train/epoch_time_elapsed': '52.63', 'train/estimated_remaining_minutes': '0.395'}
|
| 33 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 34 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 35 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 170.80it/s]
|
| 36 |
+
40%|████ | 400/1000 [01:10<00:56, 10.56[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 37 |
+
{'loss': '3.539', 'grad_norm': '0.8086', 'learning_rate': '0.001', 'epoch': '0.02696', 'train/total_time_seconds': '10.8', 'train/time_per_step_avg': '0.03215', 'train/epoch_time_elapsed': '54.51', 'train/estimated_remaining_minutes': '0.3824'}
|
| 38 |
+
{'loss': '3.492', 'grad_norm': '0.7227', 'learning_rate': '0.001', 'epoch': '0.02865', 'train/total_time_seconds': '11.44', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '56.36', 'train/estimated_remaining_minutes': '0.3701'}
|
| 39 |
+
{'loss': '3.443', 'grad_norm': '0.8203', 'learning_rate': '0.001', 'epoch': '0.03033', 'train/total_time_seconds': '12.08', 'train/time_per_step_avg': '0.03205', 'train/epoch_time_elapsed': '58.19', 'train/estimated_remaining_minutes': '0.3579'}
|
| 40 |
+
{'loss': '3.399', 'grad_norm': '0.8203', 'learning_rate': '0.001', 'epoch': '0.03202', 'train/total_time_seconds': '12.72', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '60.08', 'train/estimated_remaining_minutes': '0.3458'}
|
| 41 |
+
{'loss': '3.342', 'grad_norm': '0.6484', 'learning_rate': '0.001', 'epoch': '0.0337', 'train/total_time_seconds': '13.36', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '61.94', 'train/estimated_remaining_minutes': '0.334'}
|
| 42 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 43 |
+
{'eval_loss': '3.33', 'eval_runtime': '8.169', 'eval_samples_per_second': '1166', 'eval_steps_per_second': '0.857', 'epoch': '0.0337', 'train/total_time_seconds': '13.36', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '70.11', 'train/estimated_remaining_minutes': '0.334'}
|
| 44 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 45 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 46 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 115.56it/s]
|
| 47 |
+
50%|█████ | 500/1000 [01:27<00:46, 10.76[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 48 |
+
{'loss': '3.305', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.03539', 'train/total_time_seconds': '14', 'train/time_per_step_avg': '0.03204', 'train/epoch_time_elapsed': '72.01', 'train/estimated_remaining_minutes': '0.3222'}
|
| 49 |
+
{'loss': '3.259', 'grad_norm': '0.7188', 'learning_rate': '0.001', 'epoch': '0.03707', 'train/total_time_seconds': '14.65', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '73.95', 'train/estimated_remaining_minutes': '0.3108'}
|
| 50 |
+
{'loss': '3.229', 'grad_norm': '0.7305', 'learning_rate': '0.001', 'epoch': '0.03876', 'train/total_time_seconds': '15.3', 'train/time_per_step_avg': '0.0322', 'train/epoch_time_elapsed': '75.83', 'train/estimated_remaining_minutes': '0.2993'}
|
| 51 |
+
{'loss': '3.202', 'grad_norm': '0.7031', 'learning_rate': '0.001', 'epoch': '0.04044', 'train/total_time_seconds': '15.94', 'train/time_per_step_avg': '0.03219', 'train/epoch_time_elapsed': '77.68', 'train/estimated_remaining_minutes': '0.2877'}
|
| 52 |
+
{'loss': '3.164', 'grad_norm': '0.7266', 'learning_rate': '0.001', 'epoch': '0.04213', 'train/total_time_seconds': '16.57', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '79.6', 'train/estimated_remaining_minutes': '0.2762'}
|
| 53 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 54 |
+
{'eval_loss': '3.157', 'eval_runtime': '7.95', 'eval_samples_per_second': '1198', 'eval_steps_per_second': '0.88', 'epoch': '0.04213', 'train/total_time_seconds': '16.57', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '87.56', 'train/estimated_remaining_minutes': '0.2762'}
|
| 55 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 56 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 57 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 173.21it/s]
|
| 58 |
+
60%|██████ | 600/1000 [01:45<00:39, 10.00[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 59 |
+
{'loss': '3.145', 'grad_norm': '0.6953', 'learning_rate': '0.001', 'epoch': '0.04382', 'train/total_time_seconds': '17.2', 'train/time_per_step_avg': '0.03203', 'train/epoch_time_elapsed': '89.41', 'train/estimated_remaining_minutes': '0.2647'}
|
| 60 |
+
{'loss': '3.104', 'grad_norm': '0.8281', 'learning_rate': '0.001', 'epoch': '0.0455', 'train/total_time_seconds': '17.84', 'train/time_per_step_avg': '0.03186', 'train/epoch_time_elapsed': '91.23', 'train/estimated_remaining_minutes': '0.2533'}
|
| 61 |
+
{'loss': '3.084', 'grad_norm': '0.7578', 'learning_rate': '0.001', 'epoch': '0.04719', 'train/total_time_seconds': '18.47', 'train/time_per_step_avg': '0.03174', 'train/epoch_time_elapsed': '93.06', 'train/estimated_remaining_minutes': '0.2419'}
|
| 62 |
+
{'loss': '3.05', 'grad_norm': '0.7695', 'learning_rate': '0.001', 'epoch': '0.04887', 'train/total_time_seconds': '19.11', 'train/time_per_step_avg': '0.03169', 'train/epoch_time_elapsed': '94.88', 'train/estimated_remaining_minutes': '0.2306'}
|
| 63 |
+
{'loss': '3.038', 'grad_norm': '0.9688', 'learning_rate': '0.001', 'epoch': '0.05056', 'train/total_time_seconds': '19.75', 'train/time_per_step_avg': '0.0318', 'train/epoch_time_elapsed': '96.94', 'train/estimated_remaining_minutes': '0.2195'}
|
| 64 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 65 |
+
{'eval_loss': '3.031', 'eval_runtime': '8.12', 'eval_samples_per_second': '1173', 'eval_steps_per_second': '0.862', 'epoch': '0.05056', 'train/total_time_seconds': '19.75', 'train/time_per_step_avg': '0.0318', 'train/epoch_time_elapsed': '105.1', 'train/estimated_remaining_minutes': '0.2195'}
|
| 66 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 67 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 68 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 172.53it/s]
|
| 69 |
+
70%|███████ | 700/1000 [02:02<00:27, 11.05[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 70 |
+
{'loss': '3.031', 'grad_norm': '0.8242', 'learning_rate': '0.001', 'epoch': '0.05224', 'train/total_time_seconds': '20.38', 'train/time_per_step_avg': '0.03179', 'train/epoch_time_elapsed': '106.9', 'train/estimated_remaining_minutes': '0.2082'}
|
| 71 |
+
{'loss': '3.001', 'grad_norm': '0.6406', 'learning_rate': '0.001', 'epoch': '0.05393', 'train/total_time_seconds': '21.01', 'train/time_per_step_avg': '0.03175', 'train/epoch_time_elapsed': '108.7', 'train/estimated_remaining_minutes': '0.197'}
|
| 72 |
+
{'loss': '2.996', 'grad_norm': '0.707', 'learning_rate': '0.001', 'epoch': '0.05561', 'train/total_time_seconds': '21.64', 'train/time_per_step_avg': '0.03173', 'train/epoch_time_elapsed': '110.6', 'train/estimated_remaining_minutes': '0.1858'}
|
| 73 |
+
{'loss': '2.957', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.0573', 'train/total_time_seconds': '22.27', 'train/time_per_step_avg': '0.03168', 'train/epoch_time_elapsed': '112.4', 'train/estimated_remaining_minutes': '0.1747'}
|
| 74 |
+
{'loss': '2.94', 'grad_norm': '0.6992', 'learning_rate': '0.001', 'epoch': '0.05898', 'train/total_time_seconds': '22.9', 'train/time_per_step_avg': '0.03153', 'train/epoch_time_elapsed': '114.2', 'train/estimated_remaining_minutes': '0.1636'}
|
| 75 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 76 |
+
{'eval_loss': '2.939', 'eval_runtime': '7.975', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.05898', 'train/total_time_seconds': '22.9', 'train/time_per_step_avg': '0.03153', 'train/epoch_time_elapsed': '122.2', 'train/estimated_remaining_minutes': '0.1636'}
|
| 77 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 78 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 79 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 171.02it/s]
|
| 80 |
+
80%|████████ | 800/1000 [02:19<00:18, 10.93[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 81 |
+
{'loss': '2.923', 'grad_norm': '0.6484', 'learning_rate': '0.001', 'epoch': '0.06067', 'train/total_time_seconds': '23.54', 'train/time_per_step_avg': '0.03154', 'train/epoch_time_elapsed': '124.1', 'train/estimated_remaining_minutes': '0.1526'}
|
| 82 |
+
{'loss': '2.915', 'grad_norm': '0.957', 'learning_rate': '0.001', 'epoch': '0.06235', 'train/total_time_seconds': '24.17', 'train/time_per_step_avg': '0.03162', 'train/epoch_time_elapsed': '125.9', 'train/estimated_remaining_minutes': '0.1416'}
|
| 83 |
+
{'loss': '2.888', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.06404', 'train/total_time_seconds': '24.81', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '127.9', 'train/estimated_remaining_minutes': '0.1306'}
|
| 84 |
+
{'loss': '2.869', 'grad_norm': '0.6914', 'learning_rate': '0.001', 'epoch': '0.06572', 'train/total_time_seconds': '25.45', 'train/time_per_step_avg': '0.03176', 'train/epoch_time_elapsed': '129.7', 'train/estimated_remaining_minutes': '0.1196'}
|
| 85 |
+
{'loss': '2.879', 'grad_norm': '0.6836', 'learning_rate': '0.001', 'epoch': '0.06741', 'train/total_time_seconds': '26.09', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '131.6', 'train/estimated_remaining_minutes': '0.1087'}
|
| 86 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 87 |
+
{'eval_loss': '2.863', 'eval_runtime': '8.277', 'eval_samples_per_second': '1151', 'eval_steps_per_second': '0.846', 'epoch': '0.06741', 'train/total_time_seconds': '26.09', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '139.9', 'train/estimated_remaining_minutes': '0.1087'}
|
| 88 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 89 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 90 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 118.50it/s]
|
| 91 |
+
90%|█████████ | 900/1000 [02:37<00:09, 10.89[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 92 |
+
{'loss': '2.85', 'grad_norm': '0.6797', 'learning_rate': '0.001', 'epoch': '0.06909', 'train/total_time_seconds': '26.72', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '141.8', 'train/estimated_remaining_minutes': '0.09776'}
|
| 93 |
+
{'loss': '2.848', 'grad_norm': '0.6211', 'learning_rate': '0.001', 'epoch': '0.07078', 'train/total_time_seconds': '27.36', 'train/time_per_step_avg': '0.03182', 'train/epoch_time_elapsed': '143.6', 'train/estimated_remaining_minutes': '0.08685'}
|
| 94 |
+
{'loss': '2.834', 'grad_norm': '0.7305', 'learning_rate': '0.001', 'epoch': '0.07246', 'train/total_time_seconds': '27.99', 'train/time_per_step_avg': '0.03179', 'train/epoch_time_elapsed': '145.5', 'train/estimated_remaining_minutes': '0.07594'}
|
| 95 |
+
{'loss': '2.825', 'grad_norm': '0.6992', 'learning_rate': '0.001', 'epoch': '0.07415', 'train/total_time_seconds': '28.62', 'train/time_per_step_avg': '0.0317', 'train/epoch_time_elapsed': '147.3', 'train/estimated_remaining_minutes': '0.06505'}
|
| 96 |
+
{'loss': '2.785', 'grad_norm': '0.6875', 'learning_rate': '0.001', 'epoch': '0.07583', 'train/total_time_seconds': '29.25', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '149.1', 'train/estimated_remaining_minutes': '0.05418'}
|
| 97 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 98 |
+
{'eval_loss': '2.803', 'eval_runtime': '7.959', 'eval_samples_per_second': '1197', 'eval_steps_per_second': '0.879', 'epoch': '0.07583', 'train/total_time_seconds': '29.25', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '157.1', 'train/estimated_remaining_minutes': '0.05418'}
|
| 99 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 100 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 101 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 84.10it/s]
|
| 102 |
+
100%|██████████| 1000/1000 [02:54<00:00, 10.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 103 |
+
{'loss': '2.794', 'grad_norm': '0.6094', 'learning_rate': '0.001', 'epoch': '0.07752', 'train/total_time_seconds': '29.89', 'train/time_per_step_avg': '0.03165', 'train/epoch_time_elapsed': '159.1', 'train/estimated_remaining_minutes': '0.04331'}
|
| 104 |
+
{'loss': '2.789', 'grad_norm': '0.6719', 'learning_rate': '0.001', 'epoch': '0.0792', 'train/total_time_seconds': '30.52', 'train/time_per_step_avg': '0.03165', 'train/epoch_time_elapsed': '161', 'train/estimated_remaining_minutes': '0.03247'}
|
| 105 |
+
{'loss': '2.785', 'grad_norm': '0.582', 'learning_rate': '0.001', 'epoch': '0.08089', 'train/total_time_seconds': '31.16', 'train/time_per_step_avg': '0.03167', 'train/epoch_time_elapsed': '162.8', 'train/estimated_remaining_minutes': '0.02164'}
|
| 106 |
+
{'loss': '2.775', 'grad_norm': '0.6055', 'learning_rate': '0.001', 'epoch': '0.08257', 'train/total_time_seconds': '31.79', 'train/time_per_step_avg': '0.0317', 'train/epoch_time_elapsed': '164.6', 'train/estimated_remaining_minutes': '0.01081'}
|
| 107 |
+
{'loss': '2.754', 'grad_norm': '0.7812', 'learning_rate': '0.001', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '166.6', 'train/estimated_remaining_minutes': '0'}
|
| 108 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 109 |
+
{'eval_loss': '2.76', 'eval_runtime': '7.927', 'eval_samples_per_second': '1202', 'eval_steps_per_second': '0.883', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '174.5', 'train/estimated_remaining_minutes': '0'}
|
| 110 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 111 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 112 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 114.15it/s]
|
| 113 |
+
100%|███���██████| 1000/1000 [02:54<00:00, 5.73it/s], ?it/s]
|
| 114 |
+
{'train_runtime': '175.4', 'train_samples_per_second': '456.1', 'train_steps_per_second': '5.701', 'train_loss': '3.692', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '174.6', 'train/estimated_remaining_minutes': '0'}
|
| 115 |
+
100%|██████████| 7/7 [00:05<00:00, 1.22it/s]
|
| 116 |
+
[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
|
| 117 |
+
- If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
|
| 118 |
+
- If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
|
| 119 |
+
- If you are not the owner of the model architecture class, please contact the model code owner to update it.
|
| 120 |
+
Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 172.45it/s]
|
| 121 |
+
>>> FINISHED mlp-linear-9L successfully
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/requirements.txt
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
asttokens==3.0.1
|
| 2 |
+
comm==0.2.3
|
| 3 |
+
debugpy==1.8.21
|
| 4 |
+
decorator==5.3.1
|
| 5 |
+
executing==2.2.1
|
| 6 |
+
nest-asyncio==1.6.0
|
| 7 |
+
parso==0.8.7
|
| 8 |
+
platformdirs==4.11.0
|
| 9 |
+
psutil==7.2.2
|
| 10 |
+
ptyprocess==0.7.0
|
| 11 |
+
pure_eval==0.2.3
|
| 12 |
+
Pygments==2.20.0
|
| 13 |
+
pyzmq==27.1.0
|
| 14 |
+
setuptools==83.0.0
|
| 15 |
+
six==1.17.0
|
| 16 |
+
tornado==6.5.7
|
| 17 |
+
traitlets==5.15.0
|
| 18 |
+
fsspec==2026.4.0
|
| 19 |
+
wcwidth==0.8.2
|
| 20 |
+
ipython_pygments_lexers==1.1.1
|
| 21 |
+
jedi==0.20.0
|
| 22 |
+
jupyter_core==5.9.1
|
| 23 |
+
matplotlib-inline==0.2.2
|
| 24 |
+
pexpect==4.9.0
|
| 25 |
+
prompt_toolkit==3.0.53
|
| 26 |
+
python-dateutil==2.9.0.post0
|
| 27 |
+
stack_data==0.6.3
|
| 28 |
+
wheel==0.47.0
|
| 29 |
+
jupyter_client==8.9.1
|
| 30 |
+
pip==26.1.2
|
| 31 |
+
ipython==9.15.0
|
| 32 |
+
ipykernel==7.2.0
|
| 33 |
+
threadpoolctl==3.6.0
|
| 34 |
+
pyparsing==3.3.2
|
| 35 |
+
typing_extensions==4.15.0
|
| 36 |
+
Jinja2==3.1.6
|
| 37 |
+
narwhals==2.24.0
|
| 38 |
+
kiwisolver==1.5.0
|
| 39 |
+
joblib==1.5.3
|
| 40 |
+
fonttools==4.63.0
|
| 41 |
+
cycler==0.12.1
|
| 42 |
+
scipy==1.17.1
|
| 43 |
+
pandas==3.0.5
|
| 44 |
+
contourpy==1.3.3
|
| 45 |
+
scikit-learn==1.9.0
|
| 46 |
+
matplotlib==3.11.1
|
| 47 |
+
urllib3==2.7.0
|
| 48 |
+
tqdm==4.70.0
|
| 49 |
+
idna==3.18
|
| 50 |
+
charset-normalizer==3.4.9
|
| 51 |
+
certifi==2026.7.22
|
| 52 |
+
requests==2.34.2
|
| 53 |
+
seaborn==0.13.2
|
| 54 |
+
uv==0.12.0
|
| 55 |
+
shellingham==1.5.4
|
| 56 |
+
mpmath==1.3.0
|
| 57 |
+
attrs==26.1.0
|
| 58 |
+
hf-xet==1.5.2
|
| 59 |
+
nvidia-nccl-cu12==2.21.5
|
| 60 |
+
MarkupSafe==3.0.3
|
| 61 |
+
regex==2026.7.19
|
| 62 |
+
importlib_metadata==9.0.0
|
| 63 |
+
httpcore==1.0.9
|
| 64 |
+
annotated-doc==0.0.5
|
| 65 |
+
multidict==6.7.1
|
| 66 |
+
aiohttp==3.14.3
|
| 67 |
+
aiosignal==1.4.0
|
| 68 |
+
xxhash==3.8.1
|
| 69 |
+
aiohappyeyeballs==2.7.1
|
| 70 |
+
mdurl==0.1.2
|
| 71 |
+
cuda-toolkit==13.0.3.0
|
| 72 |
+
networkx==3.6.1
|
| 73 |
+
PyYAML==6.0.3
|
| 74 |
+
nvidia-cufile==1.15.1.6
|
| 75 |
+
typer==0.27.0
|
| 76 |
+
torchaudio==2.6.0+cu124
|
| 77 |
+
rich==15.0.0
|
| 78 |
+
nvidia-cufft-cu12==11.2.1.3
|
| 79 |
+
h11==0.16.0
|
| 80 |
+
dill==0.4.1
|
| 81 |
+
cuda-pathfinder==1.6.0
|
| 82 |
+
filelock==3.29.0
|
| 83 |
+
nvidia-nvtx-cu12==12.4.127
|
| 84 |
+
httpx==0.28.1
|
| 85 |
+
anyio==4.14.2
|
| 86 |
+
numpy==2.4.4
|
| 87 |
+
yarl==1.24.5
|
| 88 |
+
click==8.4.2
|
| 89 |
+
triton==3.2.0
|
| 90 |
+
frozenlist==1.8.0
|
| 91 |
+
zipp==4.1.0
|
| 92 |
+
propcache==0.5.2
|
| 93 |
+
markdown-it-py==4.2.0
|
| 94 |
+
nvidia-cuda-runtime==13.0.96
|
| 95 |
+
cuda-bindings==13.3.1
|
| 96 |
+
nvidia-cuda-cupti==13.0.85
|
| 97 |
+
torch==2.6.0+cu124
|
| 98 |
+
multiprocess==0.70.19
|
| 99 |
+
pillow==12.2.0
|
| 100 |
+
transformers==5.16.0.dev0
|
| 101 |
+
wandb==0.28.1
|
| 102 |
+
nvidia-curand==10.4.0.35
|
| 103 |
+
sympy==1.13.1
|
| 104 |
+
nvidia-cusparse==12.6.3.3
|
| 105 |
+
nvidia-cuda-nvrtc==13.0.88
|
| 106 |
+
typing-inspection==0.4.2
|
| 107 |
+
nvidia-cusolver==12.0.4.66
|
| 108 |
+
nvidia-cufft==12.0.0.61
|
| 109 |
+
nvidia-cudnn-cu13==9.20.0.48
|
| 110 |
+
nvidia-cublas==13.1.1.3
|
| 111 |
+
pyarrow==25.0.0
|
| 112 |
+
evaluate==0.4.6
|
| 113 |
+
diffusers==0.39.0
|
| 114 |
+
pydantic==2.13.4
|
| 115 |
+
annotated-types==0.8.0
|
| 116 |
+
protobuf==7.35.1
|
| 117 |
+
sentry-sdk==2.66.1
|
| 118 |
+
einops==0.8.2
|
| 119 |
+
packaging==26.2
|
| 120 |
+
nvidia-nvjitlink-cu12==12.4.127
|
| 121 |
+
nvidia-curand-cu12==10.3.5.147
|
| 122 |
+
nvidia-cusparselt-cu12==0.6.2
|
| 123 |
+
nvidia-cusparse-cu12==12.3.1.170
|
| 124 |
+
nvidia-cuda-runtime-cu12==12.4.127
|
| 125 |
+
torchvision==0.21.0+cu124
|
| 126 |
+
nvidia-cuda-nvrtc-cu12==12.4.127
|
| 127 |
+
nvidia-cuda-cupti-cu12==12.4.127
|
| 128 |
+
nvidia-cusolver-cu12==11.6.1.9
|
| 129 |
+
nvidia-cublas-cu12==12.4.5.8
|
| 130 |
+
nvidia-cudnn-cu12==9.1.0.70
|
| 131 |
+
huggingface_hub==1.26.0
|
| 132 |
+
datasets==5.0.1
|
| 133 |
+
safetensors==0.8.0
|
| 134 |
+
accelerate==1.14.0
|
| 135 |
+
pydantic_core==2.46.4
|
| 136 |
+
ninja==1.13.0
|
| 137 |
+
tokenizers==0.23.1
|
| 138 |
+
autocommand==2.2.2
|
| 139 |
+
backports.tarfile==1.2.0
|
| 140 |
+
importlib_metadata==8.7.1
|
| 141 |
+
jaraco.text==4.0.0
|
| 142 |
+
jaraco.context==6.1.0
|
| 143 |
+
jaraco.functools==4.4.0
|
| 144 |
+
more-itertools==10.8.0
|
| 145 |
+
packaging==26.0
|
| 146 |
+
platformdirs==4.4.0
|
| 147 |
+
tomli==2.4.0
|
| 148 |
+
wheel==0.46.3
|
| 149 |
+
zipp==3.23.0
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35",
|
| 3 |
+
"python": "CPython 3.11.15",
|
| 4 |
+
"startedAt": "2026-08-19T17:15:41.408834Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--config",
|
| 7 |
+
"/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml",
|
| 8 |
+
"--variants",
|
| 9 |
+
"mlp-linear-9L"
|
| 10 |
+
],
|
| 11 |
+
"program": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py",
|
| 12 |
+
"codePath": "sweep.py",
|
| 13 |
+
"codePathLocal": "sweep.py",
|
| 14 |
+
"git": {
|
| 15 |
+
"remote": "https://github.com/w-ahmad1a10/Activation.git",
|
| 16 |
+
"commit": "463c9961366755fc55f02df9a0d471b3cbcb025e"
|
| 17 |
+
},
|
| 18 |
+
"email": "deepnevro@gmail.com",
|
| 19 |
+
"root": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation",
|
| 20 |
+
"host": "deeplens-k3s-node1",
|
| 21 |
+
"executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python",
|
| 22 |
+
"cpu_count": 112,
|
| 23 |
+
"cpu_count_logical": 224,
|
| 24 |
+
"gpu": "NVIDIA H100 80GB HBM3",
|
| 25 |
+
"gpu_count": 8,
|
| 26 |
+
"disk": {
|
| 27 |
+
"/": {
|
| 28 |
+
"total": "1560765693952",
|
| 29 |
+
"used": "708485218304"
|
| 30 |
+
}
|
| 31 |
+
},
|
| 32 |
+
"memory": {
|
| 33 |
+
"total": "2164089937920"
|
| 34 |
+
},
|
| 35 |
+
"gpu_nvidia": [
|
| 36 |
+
{
|
| 37 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 38 |
+
"memoryTotal": "85520809984",
|
| 39 |
+
"cudaCores": 16896,
|
| 40 |
+
"architecture": "Hopper",
|
| 41 |
+
"uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae"
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 45 |
+
"memoryTotal": "85520809984",
|
| 46 |
+
"cudaCores": 16896,
|
| 47 |
+
"architecture": "Hopper",
|
| 48 |
+
"uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3"
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 52 |
+
"memoryTotal": "85520809984",
|
| 53 |
+
"cudaCores": 16896,
|
| 54 |
+
"architecture": "Hopper",
|
| 55 |
+
"uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab"
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 59 |
+
"memoryTotal": "85520809984",
|
| 60 |
+
"cudaCores": 16896,
|
| 61 |
+
"architecture": "Hopper",
|
| 62 |
+
"uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864"
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 66 |
+
"memoryTotal": "85520809984",
|
| 67 |
+
"cudaCores": 16896,
|
| 68 |
+
"architecture": "Hopper",
|
| 69 |
+
"uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 73 |
+
"memoryTotal": "85520809984",
|
| 74 |
+
"cudaCores": 16896,
|
| 75 |
+
"architecture": "Hopper",
|
| 76 |
+
"uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54"
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 80 |
+
"memoryTotal": "85520809984",
|
| 81 |
+
"cudaCores": 16896,
|
| 82 |
+
"architecture": "Hopper",
|
| 83 |
+
"uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9"
|
| 84 |
+
},
|
| 85 |
+
{
|
| 86 |
+
"name": "NVIDIA H100 80GB HBM3",
|
| 87 |
+
"memoryTotal": "85520809984",
|
| 88 |
+
"cudaCores": 16896,
|
| 89 |
+
"architecture": "Hopper",
|
| 90 |
+
"uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea"
|
| 91 |
+
}
|
| 92 |
+
],
|
| 93 |
+
"cudaVersion": "12.4",
|
| 94 |
+
"writerId": "2l4ndnmtei7frfwveku3d762xxak1bdj"
|
| 95 |
+
}
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-summary.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"train/global_step":1000,"train_steps_per_second":5.701,"eval/samples_per_second":1206.607,"train/learning_rate":0.001,"eval/loss":2.7597591876983643,"eval/steps_per_second":0.887,"_runtime":182,"_wandb":{"runtime":182},"_timestamp":1.7871599245612671e+09,"eval/runtime":7.8957,"train/grad_norm":0.78125,"train/epoch":0.08426019548365352,"_step":61,"total_flos":3.6298555392e+14,"train_loss":3.6923015975952147,"train_runtime":175.4068,"train/loss":2.753749656677246,"train_samples_per_second":456.083}
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-core.log
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-08-19T16:38:32.839280488Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpn7u591yp/port-3590110.txt","pid":3590110,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
|
| 2 |
+
{"time":"2026-08-19T16:38:32.840574816Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3590110}
|
| 3 |
+
{"time":"2026-08-19T16:38:32.84057072Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3590110-3591582-3840970873/socket","Net":"unix"}}
|
| 4 |
+
{"time":"2026-08-19T16:38:33.017601922Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
|
| 5 |
+
{"time":"2026-08-19T16:38:45.523223978Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"}
|
| 6 |
+
{"time":"2026-08-19T16:38:45.593765665Z","level":"INFO","msg":"handleInformInit: received","streamId":"74tq2syl","id":"2(@)"}
|
| 7 |
+
{"time":"2026-08-19T16:38:45.85943264Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"74tq2syl","id":"2(@)"}
|
| 8 |
+
{"time":"2026-08-19T16:38:51.253972428Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
|
| 9 |
+
{"time":"2026-08-19T16:43:55.859246567Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
|
| 10 |
+
{"time":"2026-08-19T16:43:56.562384948Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
|
| 11 |
+
{"time":"2026-08-19T16:43:56.564075521Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"74tq2syl","id":"2(@)"}
|
| 12 |
+
{"time":"2026-08-19T16:43:56.56470085Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"74tq2syl","id":"2(@)"}
|
| 13 |
+
{"time":"2026-08-19T16:43:58.189628272Z","level":"INFO","msg":"connection: closing","id":"2(@)"}
|
| 14 |
+
{"time":"2026-08-19T16:43:58.189706084Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"}
|
| 15 |
+
{"time":"2026-08-19T16:43:58.189644595Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"}
|
| 16 |
+
{"time":"2026-08-19T16:43:58.189713806Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"}
|
| 17 |
+
{"time":"2026-08-19T16:45:34.355733713Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"}
|
| 18 |
+
{"time":"2026-08-19T16:45:34.483865974Z","level":"INFO","msg":"handleInformInit: received","streamId":"glh82iyh","id":"3(@)"}
|
| 19 |
+
{"time":"2026-08-19T16:45:34.744778959Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"glh82iyh","id":"3(@)"}
|
| 20 |
+
{"time":"2026-08-19T16:45:40.185561802Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
|
| 21 |
+
{"time":"2026-08-19T16:45:40.195891916Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
|
| 22 |
+
{"time":"2026-08-19T16:45:40.960129401Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
|
| 23 |
+
{"time":"2026-08-19T16:45:40.961603019Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"glh82iyh","id":"3(@)"}
|
| 24 |
+
{"time":"2026-08-19T16:45:40.96217202Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"glh82iyh","id":"3(@)"}
|
| 25 |
+
{"time":"2026-08-19T16:45:42.643090062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
|
| 26 |
+
{"time":"2026-08-19T16:45:42.643171917Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
|
| 27 |
+
{"time":"2026-08-19T16:45:42.643096466Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
|
| 28 |
+
{"time":"2026-08-19T16:45:42.643183903Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
|
| 29 |
+
{"time":"2026-08-19T16:45:53.775307513Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"}
|
| 30 |
+
{"time":"2026-08-19T16:45:53.86012332Z","level":"INFO","msg":"handleInformInit: received","streamId":"44x03ghu","id":"4(@)"}
|
| 31 |
+
{"time":"2026-08-19T16:45:54.119528402Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"44x03ghu","id":"4(@)"}
|
| 32 |
+
{"time":"2026-08-19T16:45:59.764320578Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
|
| 33 |
+
{"time":"2026-08-19T16:50:40.211693707Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
|
| 34 |
+
{"time":"2026-08-19T16:50:40.749569572Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
|
| 35 |
+
{"time":"2026-08-19T16:50:40.751009589Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"44x03ghu","id":"4(@)"}
|
| 36 |
+
{"time":"2026-08-19T16:50:40.751589921Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"44x03ghu","id":"4(@)"}
|
| 37 |
+
{"time":"2026-08-19T16:50:42.667568611Z","level":"INFO","msg":"connection: closing","id":"4(@)"}
|
| 38 |
+
{"time":"2026-08-19T16:50:42.66767627Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"}
|
| 39 |
+
{"time":"2026-08-19T16:50:42.667582108Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"}
|
| 40 |
+
{"time":"2026-08-19T16:50:42.667687704Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"}
|
| 41 |
+
{"time":"2026-08-19T17:08:54.476108098Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"}
|
| 42 |
+
{"time":"2026-08-19T17:08:54.716542574Z","level":"INFO","msg":"handleInformInit: received","streamId":"zhg4u13t","id":"5(@)"}
|
| 43 |
+
{"time":"2026-08-19T17:08:54.983870419Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"zhg4u13t","id":"5(@)"}
|
| 44 |
+
{"time":"2026-08-19T17:09:00.342597647Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
|
| 45 |
+
{"time":"2026-08-19T17:11:36.384264189Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
|
| 46 |
+
{"time":"2026-08-19T17:11:36.932623627Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
|
| 47 |
+
{"time":"2026-08-19T17:11:36.934203156Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"zhg4u13t","id":"5(@)"}
|
| 48 |
+
{"time":"2026-08-19T17:11:36.934659413Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"zhg4u13t","id":"5(@)"}
|
| 49 |
+
{"time":"2026-08-19T17:11:38.381994049Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
|
| 50 |
+
{"time":"2026-08-19T17:11:38.382078588Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
|
| 51 |
+
{"time":"2026-08-19T17:11:38.382003656Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
|
| 52 |
+
{"time":"2026-08-19T17:11:38.382088954Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
|
| 53 |
+
{"time":"2026-08-19T17:12:15.205004959Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"6(@)"}
|
| 54 |
+
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
+
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
+
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 57 |
+
{"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 58 |
+
{"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 59 |
+
{"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
|
| 60 |
+
{"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
|
| 61 |
+
{"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
|
| 62 |
+
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
+
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
+
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
| 65 |
+
{"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
|
| 66 |
+
{"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
|
| 67 |
+
{"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
|
| 68 |
+
{"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 69 |
+
{"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 70 |
+
{"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
|
| 71 |
+
{"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
|
| 72 |
+
{"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
|
| 73 |
+
{"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
|
| 74 |
+
{"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
|
| 75 |
+
{"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
|
| 76 |
+
{"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2026-08-19T17:15:41.412064429Z","level":"INFO","msg":"wandb-core"}
|
| 2 |
+
{"time":"2026-08-19T17:15:41.41218811Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
|
| 3 |
+
{"time":"2026-08-19T17:15:41.674976743Z","level":"INFO","msg":"stream: created new stream","id":"bhq23mh5"}
|
| 4 |
+
{"time":"2026-08-19T17:15:41.675059529Z","level":"INFO","msg":"handler: started"}
|
| 5 |
+
{"time":"2026-08-19T17:15:41.675177927Z","level":"INFO","msg":"stream: started"}
|
| 6 |
+
{"time":"2026-08-19T17:15:41.675184514Z","level":"INFO","msg":"writer: started","stream_id":"bhq23mh5"}
|
| 7 |
+
{"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
|
| 8 |
+
{"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
|
| 9 |
+
{"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 10 |
+
{"time":"2026-08-19T17:15:57.093327808Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":5,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":9,"uploaded_len":2}
|
| 11 |
+
{"time":"2026-08-19T17:15:57.208750755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 12 |
+
{"time":"2026-08-19T17:16:12.09424474Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":6,"events_offset":1,"events_lines":2,"console_offset":2,"console_lines":1}
|
| 13 |
+
{"time":"2026-08-19T17:16:12.254954463Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 14 |
+
{"time":"2026-08-19T17:16:27.093741092Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":6,"events_offset":3,"events_lines":2,"console_offset":8,"console_lines":22}
|
| 15 |
+
{"time":"2026-08-19T17:16:27.218480793Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 16 |
+
{"time":"2026-08-19T17:16:42.09555922Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":4,"events_offset":5,"events_lines":2,"console_offset":24,"console_lines":1}
|
| 17 |
+
{"time":"2026-08-19T17:16:42.264597826Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 18 |
+
{"time":"2026-08-19T17:16:57.09386987Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":5,"events_offset":7,"events_lines":2,"console_offset":30,"console_lines":19}
|
| 19 |
+
{"time":"2026-08-19T17:16:57.209800875Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 20 |
+
{"time":"2026-08-19T17:17:12.094049325Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":5,"events_offset":9,"events_lines":2,"console_offset":46,"console_lines":1}
|
| 21 |
+
{"time":"2026-08-19T17:17:12.252024387Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 22 |
+
{"time":"2026-08-19T17:17:27.093470433Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":4,"events_offset":11,"events_lines":2,"console_offset":49,"console_lines":15}
|
| 23 |
+
{"time":"2026-08-19T17:17:27.202702289Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
+
{"time":"2026-08-19T17:17:42.093533111Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":57,"console_lines":1}
|
| 25 |
+
{"time":"2026-08-19T17:17:42.240909652Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 26 |
+
{"time":"2026-08-19T17:17:57.093850562Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":63,"console_lines":23}
|
| 27 |
+
{"time":"2026-08-19T17:17:57.20571167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 28 |
+
{"time":"2026-08-19T17:18:12.094059821Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":17,"events_lines":2,"console_offset":79,"console_lines":1}
|
| 29 |
+
{"time":"2026-08-19T17:18:12.283285285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 30 |
+
{"time":"2026-08-19T17:18:27.0939628Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":5,"events_offset":19,"events_lines":2,"console_offset":85,"console_lines":21}
|
| 31 |
+
{"time":"2026-08-19T17:18:27.220681676Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 32 |
+
{"time":"2026-08-19T17:18:42.09374958Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":3,"events_offset":21,"events_lines":2,"console_offset":101,"console_lines":1}
|
| 33 |
+
{"time":"2026-08-19T17:18:42.218715049Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 34 |
+
{"time":"2026-08-19T17:18:45.014142777Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 35 |
+
{"time":"2026-08-19T17:18:45.014366349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":23,"events_lines":1,"console_offset":106,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
|
| 36 |
+
{"time":"2026-08-19T17:18:45.153108654Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 37 |
+
{"time":"2026-08-19T17:18:45.154305649Z","level":"INFO","msg":"handler: operation stats","stats":{}}
|
| 38 |
+
{"time":"2026-08-19T17:18:45.157232952Z","level":"INFO","msg":"stream: finishing up"}
|
| 39 |
+
{"time":"2026-08-19T17:18:45.157265556Z","level":"INFO","msg":"handler: closed"}
|
| 40 |
+
{"time":"2026-08-19T17:18:45.161743858Z","level":"INFO","msg":"sender: closed"}
|
| 41 |
+
{"time":"2026-08-19T17:18:45.161763043Z","level":"INFO","msg":"stream: all finished"}
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1
|
| 2 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Configure stats pid to 3953242
|
| 3 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Loading settings from environment variables
|
| 4 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log
|
| 5 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log
|
| 6 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():772] calling init triggers
|
| 7 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
|
| 8 |
+
config: {'_wandb': {}}
|
| 9 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():820] starting backend
|
| 10 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
|
| 11 |
+
2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():835] sending inform_init request
|
| 12 |
+
2026-08-19 17:15:41,675 INFO MainThread:3953242 [wandb_init.py:init():840] backend started and connected
|
| 13 |
+
2026-08-19 17:15:41,679 INFO MainThread:3953242 [wandb_init.py:init():910] updated telemetry
|
| 14 |
+
2026-08-19 17:15:41,686 INFO MainThread:3953242 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
|
| 15 |
+
2026-08-19 17:15:42,011 INFO MainThread:3953242 [wandb_init.py:init():978] starting run threads in backend
|
| 16 |
+
2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_console_start():2621] atexit reg
|
| 17 |
+
2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2471] redirect: wrap_raw
|
| 18 |
+
2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2540] Wrapping output streams.
|
| 19 |
+
2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2563] Redirects installed.
|
| 20 |
+
2026-08-19 17:15:42,089 INFO MainThread:3953242 [wandb_init.py:init():1016] run started, returning control to user process
|
| 21 |
+
2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
+
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
|
| 23 |
+
2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
|
| 24 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate-checking/bhq23mh5
|
| 25 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
|
| 26 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2570] restore
|
| 27 |
+
2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2576] restore done
|
| 28 |
+
2026-08-19 17:18:45,156 INFO MainThread:3953242 [wandb_run.py:_footer_sync_info():3993] logging synced files
|
zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:896db157ba2ca29ceb3d6b02e8ae3f8c75a1864cc416843cde980ad34c0a27a7
|
| 3 |
+
size 251327
|