Auto upload zain 2026-08-19T17:15:02.582332 (part 2)
Browse files- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt +1 -1
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json +170 -170
- zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/model.safetensors +1 -1
- zain/Activation/out/mlp-linear-3L_run/training_args.bin +1 -1
- zain/Activation/out/mlp-linear-3L_run/training_log.jsonl +9 -0
- zain/Activation/out/sweep_summary.json +2 -2
- zain/Activation/wandb/debug-internal.log +12 -0
- zain/Activation/wandb/debug.log +5 -0
- zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log +8 -0
- zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log +8 -0
- zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log +8 -0
- zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log +8 -0
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6210521539864a26a8dbb7e8e762ead750a9aebb991577b3ce8dbefe87900cef
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4089360
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3bfd43e5934f5d24798fd85b66fe3d9a69b05a1e299d72f2187c5cb93c96eca8
|
| 3 |
size 4089360
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:819e7677327ec819a829273ee6bf375a38913eb519826c73009d0dd3825f5480
|
| 3 |
size 1064
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/trainer_state.json
CHANGED
|
@@ -11,389 +11,389 @@
|
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
"epoch": 0.0016852039096730705,
|
| 14 |
-
"grad_norm": 1.
|
| 15 |
-
"learning_rate":
|
| 16 |
-
"loss": 8.
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
"epoch": 0.003370407819346141,
|
| 21 |
-
"grad_norm": 1.
|
| 22 |
-
"learning_rate": 0.
|
| 23 |
-
"loss": 7.
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
"epoch": 0.005055611729019211,
|
| 28 |
-
"grad_norm": 1.
|
| 29 |
-
"learning_rate": 0.
|
| 30 |
-
"loss": 7.
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
"epoch": 0.006740815638692282,
|
| 35 |
-
"grad_norm": 1.
|
| 36 |
-
"learning_rate": 0.
|
| 37 |
-
"loss": 6.
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
"epoch": 0.008426019548365353,
|
| 42 |
-
"grad_norm":
|
| 43 |
-
"learning_rate": 0.
|
| 44 |
-
"loss":
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"epoch": 0.008426019548365353,
|
| 49 |
-
"eval_loss":
|
| 50 |
-
"eval_runtime": 7.
|
| 51 |
-
"eval_samples_per_second":
|
| 52 |
-
"eval_steps_per_second": 0.
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
"epoch": 0.010111223458038422,
|
| 57 |
-
"grad_norm": 1.
|
| 58 |
-
"learning_rate": 0.
|
| 59 |
-
"loss": 5.
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"epoch": 0.011796427367711493,
|
| 64 |
-
"grad_norm": 1.
|
| 65 |
-
"learning_rate": 0.
|
| 66 |
-
"loss": 5.
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
"epoch": 0.013481631277384564,
|
| 71 |
-
"grad_norm":
|
| 72 |
-
"learning_rate": 0.
|
| 73 |
-
"loss":
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
"epoch": 0.015166835187057633,
|
| 78 |
-
"grad_norm":
|
| 79 |
-
"learning_rate": 0.
|
| 80 |
-
"loss": 4.
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
"epoch": 0.016852039096730706,
|
| 85 |
-
"grad_norm": 1.
|
| 86 |
-
"learning_rate": 0.
|
| 87 |
-
"loss": 4.
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
"epoch": 0.016852039096730706,
|
| 92 |
-
"eval_loss": 4.
|
| 93 |
-
"eval_runtime": 7.
|
| 94 |
-
"eval_samples_per_second":
|
| 95 |
-
"eval_steps_per_second": 0.
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
"epoch": 0.018537243006403775,
|
| 100 |
-
"grad_norm":
|
| 101 |
-
"learning_rate": 0.
|
| 102 |
-
"loss": 4.
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
"epoch": 0.020222446916076844,
|
| 107 |
-
"grad_norm": 0.
|
| 108 |
-
"learning_rate": 0.
|
| 109 |
-
"loss":
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
"epoch": 0.021907650825749917,
|
| 114 |
-
"grad_norm": 0.
|
| 115 |
-
"learning_rate": 0.
|
| 116 |
-
"loss":
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
"epoch": 0.023592854735422986,
|
| 121 |
-
"grad_norm":
|
| 122 |
-
"learning_rate": 0.
|
| 123 |
-
"loss": 3.
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
"epoch": 0.025278058645096056,
|
| 128 |
-
"grad_norm":
|
| 129 |
-
"learning_rate": 0.
|
| 130 |
-
"loss": 3.
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
"epoch": 0.025278058645096056,
|
| 135 |
-
"eval_loss": 3.
|
| 136 |
-
"eval_runtime": 7.
|
| 137 |
-
"eval_samples_per_second":
|
| 138 |
-
"eval_steps_per_second": 0.
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
"epoch": 0.026963262554769128,
|
| 143 |
-
"grad_norm": 0.
|
| 144 |
-
"learning_rate": 0.
|
| 145 |
-
"loss": 3.
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
"epoch": 0.028648466464442197,
|
| 150 |
-
"grad_norm":
|
| 151 |
-
"learning_rate": 0.
|
| 152 |
-
"loss": 3.
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
"epoch": 0.030333670374115267,
|
| 157 |
-
"grad_norm":
|
| 158 |
-
"learning_rate": 0.
|
| 159 |
-
"loss": 3.
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
"epoch": 0.032018874283788336,
|
| 164 |
-
"grad_norm":
|
| 165 |
-
"learning_rate": 0.
|
| 166 |
-
"loss": 3.
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
"epoch": 0.03370407819346141,
|
| 171 |
-
"grad_norm": 1.
|
| 172 |
-
"learning_rate": 0.
|
| 173 |
-
"loss": 3.
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
"epoch": 0.03370407819346141,
|
| 178 |
-
"eval_loss": 3.
|
| 179 |
-
"eval_runtime": 7.
|
| 180 |
-
"eval_samples_per_second":
|
| 181 |
-
"eval_steps_per_second": 0.
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
"epoch": 0.03538928210313448,
|
| 186 |
-
"grad_norm": 1.
|
| 187 |
-
"learning_rate": 0.
|
| 188 |
-
"loss": 3.
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
"epoch": 0.03707448601280755,
|
| 193 |
-
"grad_norm": 0.
|
| 194 |
-
"learning_rate": 0.
|
| 195 |
-
"loss": 3.
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
"epoch": 0.03875968992248062,
|
| 200 |
-
"grad_norm":
|
| 201 |
-
"learning_rate": 0.
|
| 202 |
-
"loss": 3.
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
"epoch": 0.04044489383215369,
|
| 207 |
-
"grad_norm": 0.
|
| 208 |
-
"learning_rate": 0.
|
| 209 |
-
"loss": 3.
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
"epoch": 0.04213009774182676,
|
| 214 |
-
"grad_norm": 0.
|
| 215 |
-
"learning_rate": 0.
|
| 216 |
-
"loss": 3.
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
"epoch": 0.04213009774182676,
|
| 221 |
-
"eval_loss": 3.
|
| 222 |
-
"eval_runtime": 7.
|
| 223 |
-
"eval_samples_per_second": 1271.
|
| 224 |
"eval_steps_per_second": 0.934,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
"epoch": 0.043815301651499834,
|
| 229 |
-
"grad_norm": 0.
|
| 230 |
-
"learning_rate": 0.
|
| 231 |
-
"loss": 3.
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
"epoch": 0.0455005055611729,
|
| 236 |
-
"grad_norm":
|
| 237 |
-
"learning_rate": 0.
|
| 238 |
-
"loss": 3.
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
"epoch": 0.04718570947084597,
|
| 243 |
-
"grad_norm":
|
| 244 |
-
"learning_rate": 0.
|
| 245 |
-
"loss": 3.
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
"epoch": 0.04887091338051904,
|
| 250 |
-
"grad_norm": 0.
|
| 251 |
-
"learning_rate": 0.
|
| 252 |
-
"loss": 3.
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
"epoch": 0.05055611729019211,
|
| 257 |
-
"grad_norm":
|
| 258 |
-
"learning_rate": 0.
|
| 259 |
-
"loss": 3.
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"epoch": 0.05055611729019211,
|
| 264 |
-
"eval_loss": 3.
|
| 265 |
-
"eval_runtime": 7.
|
| 266 |
-
"eval_samples_per_second":
|
| 267 |
-
"eval_steps_per_second": 0.
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
"epoch": 0.05224132119986518,
|
| 272 |
-
"grad_norm":
|
| 273 |
-
"learning_rate": 0.
|
| 274 |
-
"loss": 3.
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"epoch": 0.053926525109538256,
|
| 279 |
-
"grad_norm": 1.
|
| 280 |
-
"learning_rate": 0.
|
| 281 |
-
"loss": 3.
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
"epoch": 0.055611729019211326,
|
| 286 |
-
"grad_norm":
|
| 287 |
-
"learning_rate": 0.
|
| 288 |
-
"loss": 3.
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
"epoch": 0.057296932928884395,
|
| 293 |
-
"grad_norm":
|
| 294 |
-
"learning_rate": 0.
|
| 295 |
-
"loss": 3.
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
"epoch": 0.058982136838557464,
|
| 300 |
-
"grad_norm": 1.
|
| 301 |
-
"learning_rate": 0.
|
| 302 |
-
"loss": 3.
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
"epoch": 0.058982136838557464,
|
| 307 |
-
"eval_loss": 3.
|
| 308 |
-
"eval_runtime": 7.
|
| 309 |
-
"eval_samples_per_second":
|
| 310 |
-
"eval_steps_per_second": 0.
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
"epoch": 0.06066734074823053,
|
| 315 |
-
"grad_norm": 0.
|
| 316 |
-
"learning_rate": 0.
|
| 317 |
-
"loss":
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
"epoch": 0.06235254465790361,
|
| 322 |
-
"grad_norm":
|
| 323 |
-
"learning_rate": 0.
|
| 324 |
-
"loss":
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
"epoch": 0.06403774856757667,
|
| 329 |
-
"grad_norm":
|
| 330 |
-
"learning_rate": 0.
|
| 331 |
-
"loss":
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
"epoch": 0.06572295247724974,
|
| 336 |
-
"grad_norm":
|
| 337 |
-
"learning_rate": 0.
|
| 338 |
-
"loss":
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
"epoch": 0.06740815638692282,
|
| 343 |
-
"grad_norm":
|
| 344 |
-
"learning_rate": 0.
|
| 345 |
-
"loss":
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
"epoch": 0.06740815638692282,
|
| 350 |
-
"eval_loss":
|
| 351 |
-
"eval_runtime": 7.
|
| 352 |
-
"eval_samples_per_second":
|
| 353 |
-
"eval_steps_per_second": 0.
|
| 354 |
"step": 800
|
| 355 |
},
|
| 356 |
{
|
| 357 |
"epoch": 0.0690933602965959,
|
| 358 |
-
"grad_norm":
|
| 359 |
-
"learning_rate": 0.
|
| 360 |
-
"loss":
|
| 361 |
"step": 820
|
| 362 |
},
|
| 363 |
{
|
| 364 |
"epoch": 0.07077856420626896,
|
| 365 |
-
"grad_norm": 0.
|
| 366 |
-
"learning_rate": 0.
|
| 367 |
-
"loss":
|
| 368 |
"step": 840
|
| 369 |
},
|
| 370 |
{
|
| 371 |
"epoch": 0.07246376811594203,
|
| 372 |
-
"grad_norm": 0.
|
| 373 |
-
"learning_rate": 0.
|
| 374 |
-
"loss":
|
| 375 |
"step": 860
|
| 376 |
},
|
| 377 |
{
|
| 378 |
"epoch": 0.0741489720256151,
|
| 379 |
-
"grad_norm": 1.
|
| 380 |
-
"learning_rate": 0.
|
| 381 |
-
"loss":
|
| 382 |
"step": 880
|
| 383 |
},
|
| 384 |
{
|
| 385 |
"epoch": 0.07583417593528817,
|
| 386 |
-
"grad_norm":
|
| 387 |
-
"learning_rate": 0.
|
| 388 |
-
"loss":
|
| 389 |
"step": 900
|
| 390 |
},
|
| 391 |
{
|
| 392 |
"epoch": 0.07583417593528817,
|
| 393 |
-
"eval_loss":
|
| 394 |
-
"eval_runtime": 7.
|
| 395 |
-
"eval_samples_per_second":
|
| 396 |
-
"eval_steps_per_second": 0.
|
| 397 |
"step": 900
|
| 398 |
}
|
| 399 |
],
|
|
|
|
| 11 |
"log_history": [
|
| 12 |
{
|
| 13 |
"epoch": 0.0016852039096730705,
|
| 14 |
+
"grad_norm": 1.28125,
|
| 15 |
+
"learning_rate": 9.5e-05,
|
| 16 |
+
"loss": 8.230884552001953,
|
| 17 |
"step": 20
|
| 18 |
},
|
| 19 |
{
|
| 20 |
"epoch": 0.003370407819346141,
|
| 21 |
+
"grad_norm": 1.234375,
|
| 22 |
+
"learning_rate": 0.00019500000000000002,
|
| 23 |
+
"loss": 7.774678802490234,
|
| 24 |
"step": 40
|
| 25 |
},
|
| 26 |
{
|
| 27 |
"epoch": 0.005055611729019211,
|
| 28 |
+
"grad_norm": 1.1640625,
|
| 29 |
+
"learning_rate": 0.000295,
|
| 30 |
+
"loss": 7.161412048339844,
|
| 31 |
"step": 60
|
| 32 |
},
|
| 33 |
{
|
| 34 |
"epoch": 0.006740815638692282,
|
| 35 |
+
"grad_norm": 1.171875,
|
| 36 |
+
"learning_rate": 0.000395,
|
| 37 |
+
"loss": 6.473944091796875,
|
| 38 |
"step": 80
|
| 39 |
},
|
| 40 |
{
|
| 41 |
"epoch": 0.008426019548365353,
|
| 42 |
+
"grad_norm": 1.1640625,
|
| 43 |
+
"learning_rate": 0.000495,
|
| 44 |
+
"loss": 5.909336853027344,
|
| 45 |
"step": 100
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"epoch": 0.008426019548365353,
|
| 49 |
+
"eval_loss": 5.663336753845215,
|
| 50 |
+
"eval_runtime": 7.566,
|
| 51 |
+
"eval_samples_per_second": 1259.189,
|
| 52 |
+
"eval_steps_per_second": 0.925,
|
| 53 |
"step": 100
|
| 54 |
},
|
| 55 |
{
|
| 56 |
"epoch": 0.010111223458038422,
|
| 57 |
+
"grad_norm": 1.3046875,
|
| 58 |
+
"learning_rate": 0.0005949999999999999,
|
| 59 |
+
"loss": 5.44867172241211,
|
| 60 |
"step": 120
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"epoch": 0.011796427367711493,
|
| 64 |
+
"grad_norm": 1.671875,
|
| 65 |
+
"learning_rate": 0.000695,
|
| 66 |
+
"loss": 5.003098678588867,
|
| 67 |
"step": 140
|
| 68 |
},
|
| 69 |
{
|
| 70 |
"epoch": 0.013481631277384564,
|
| 71 |
+
"grad_norm": 0.98046875,
|
| 72 |
+
"learning_rate": 0.000795,
|
| 73 |
+
"loss": 4.667464065551758,
|
| 74 |
"step": 160
|
| 75 |
},
|
| 76 |
{
|
| 77 |
"epoch": 0.015166835187057633,
|
| 78 |
+
"grad_norm": 0.72265625,
|
| 79 |
+
"learning_rate": 0.0008950000000000001,
|
| 80 |
+
"loss": 4.417913818359375,
|
| 81 |
"step": 180
|
| 82 |
},
|
| 83 |
{
|
| 84 |
"epoch": 0.016852039096730706,
|
| 85 |
+
"grad_norm": 1.4140625,
|
| 86 |
+
"learning_rate": 0.000995,
|
| 87 |
+
"loss": 4.196286392211914,
|
| 88 |
"step": 200
|
| 89 |
},
|
| 90 |
{
|
| 91 |
"epoch": 0.016852039096730706,
|
| 92 |
+
"eval_loss": 4.103216648101807,
|
| 93 |
+
"eval_runtime": 7.6083,
|
| 94 |
+
"eval_samples_per_second": 1252.178,
|
| 95 |
+
"eval_steps_per_second": 0.92,
|
| 96 |
"step": 200
|
| 97 |
},
|
| 98 |
{
|
| 99 |
"epoch": 0.018537243006403775,
|
| 100 |
+
"grad_norm": 1.0078125,
|
| 101 |
+
"learning_rate": 0.001,
|
| 102 |
+
"loss": 4.029207611083985,
|
| 103 |
"step": 220
|
| 104 |
},
|
| 105 |
{
|
| 106 |
"epoch": 0.020222446916076844,
|
| 107 |
+
"grad_norm": 0.90625,
|
| 108 |
+
"learning_rate": 0.001,
|
| 109 |
+
"loss": 3.895578384399414,
|
| 110 |
"step": 240
|
| 111 |
},
|
| 112 |
{
|
| 113 |
"epoch": 0.021907650825749917,
|
| 114 |
+
"grad_norm": 0.71875,
|
| 115 |
+
"learning_rate": 0.001,
|
| 116 |
+
"loss": 3.7797645568847655,
|
| 117 |
"step": 260
|
| 118 |
},
|
| 119 |
{
|
| 120 |
"epoch": 0.023592854735422986,
|
| 121 |
+
"grad_norm": 0.83203125,
|
| 122 |
+
"learning_rate": 0.001,
|
| 123 |
+
"loss": 3.693258285522461,
|
| 124 |
"step": 280
|
| 125 |
},
|
| 126 |
{
|
| 127 |
"epoch": 0.025278058645096056,
|
| 128 |
+
"grad_norm": 1.171875,
|
| 129 |
+
"learning_rate": 0.001,
|
| 130 |
+
"loss": 3.6289249420166017,
|
| 131 |
"step": 300
|
| 132 |
},
|
| 133 |
{
|
| 134 |
"epoch": 0.025278058645096056,
|
| 135 |
+
"eval_loss": 3.5979104042053223,
|
| 136 |
+
"eval_runtime": 7.4719,
|
| 137 |
+
"eval_samples_per_second": 1275.037,
|
| 138 |
+
"eval_steps_per_second": 0.937,
|
| 139 |
"step": 300
|
| 140 |
},
|
| 141 |
{
|
| 142 |
"epoch": 0.026963262554769128,
|
| 143 |
+
"grad_norm": 0.9609375,
|
| 144 |
+
"learning_rate": 0.001,
|
| 145 |
+
"loss": 3.5430423736572267,
|
| 146 |
"step": 320
|
| 147 |
},
|
| 148 |
{
|
| 149 |
"epoch": 0.028648466464442197,
|
| 150 |
+
"grad_norm": 0.953125,
|
| 151 |
+
"learning_rate": 0.001,
|
| 152 |
+
"loss": 3.499795150756836,
|
| 153 |
"step": 340
|
| 154 |
},
|
| 155 |
{
|
| 156 |
"epoch": 0.030333670374115267,
|
| 157 |
+
"grad_norm": 1.0234375,
|
| 158 |
+
"learning_rate": 0.001,
|
| 159 |
+
"loss": 3.4527099609375,
|
| 160 |
"step": 360
|
| 161 |
},
|
| 162 |
{
|
| 163 |
"epoch": 0.032018874283788336,
|
| 164 |
+
"grad_norm": 0.9453125,
|
| 165 |
+
"learning_rate": 0.001,
|
| 166 |
+
"loss": 3.409439468383789,
|
| 167 |
"step": 380
|
| 168 |
},
|
| 169 |
{
|
| 170 |
"epoch": 0.03370407819346141,
|
| 171 |
+
"grad_norm": 1.296875,
|
| 172 |
+
"learning_rate": 0.001,
|
| 173 |
+
"loss": 3.360205078125,
|
| 174 |
"step": 400
|
| 175 |
},
|
| 176 |
{
|
| 177 |
"epoch": 0.03370407819346141,
|
| 178 |
+
"eval_loss": 3.3530941009521484,
|
| 179 |
+
"eval_runtime": 7.5215,
|
| 180 |
+
"eval_samples_per_second": 1266.63,
|
| 181 |
+
"eval_steps_per_second": 0.931,
|
| 182 |
"step": 400
|
| 183 |
},
|
| 184 |
{
|
| 185 |
"epoch": 0.03538928210313448,
|
| 186 |
+
"grad_norm": 1.3046875,
|
| 187 |
+
"learning_rate": 0.001,
|
| 188 |
+
"loss": 3.321756362915039,
|
| 189 |
"step": 420
|
| 190 |
},
|
| 191 |
{
|
| 192 |
"epoch": 0.03707448601280755,
|
| 193 |
+
"grad_norm": 0.73828125,
|
| 194 |
+
"learning_rate": 0.001,
|
| 195 |
+
"loss": 3.2804222106933594,
|
| 196 |
"step": 440
|
| 197 |
},
|
| 198 |
{
|
| 199 |
"epoch": 0.03875968992248062,
|
| 200 |
+
"grad_norm": 0.796875,
|
| 201 |
+
"learning_rate": 0.001,
|
| 202 |
+
"loss": 3.2532787322998047,
|
| 203 |
"step": 460
|
| 204 |
},
|
| 205 |
{
|
| 206 |
"epoch": 0.04044489383215369,
|
| 207 |
+
"grad_norm": 0.953125,
|
| 208 |
+
"learning_rate": 0.001,
|
| 209 |
+
"loss": 3.22874755859375,
|
| 210 |
"step": 480
|
| 211 |
},
|
| 212 |
{
|
| 213 |
"epoch": 0.04213009774182676,
|
| 214 |
+
"grad_norm": 0.890625,
|
| 215 |
+
"learning_rate": 0.001,
|
| 216 |
+
"loss": 3.197536659240723,
|
| 217 |
"step": 500
|
| 218 |
},
|
| 219 |
{
|
| 220 |
"epoch": 0.04213009774182676,
|
| 221 |
+
"eval_loss": 3.1904470920562744,
|
| 222 |
+
"eval_runtime": 7.4922,
|
| 223 |
+
"eval_samples_per_second": 1271.584,
|
| 224 |
"eval_steps_per_second": 0.934,
|
| 225 |
"step": 500
|
| 226 |
},
|
| 227 |
{
|
| 228 |
"epoch": 0.043815301651499834,
|
| 229 |
+
"grad_norm": 0.96875,
|
| 230 |
+
"learning_rate": 0.001,
|
| 231 |
+
"loss": 3.184717559814453,
|
| 232 |
"step": 520
|
| 233 |
},
|
| 234 |
{
|
| 235 |
"epoch": 0.0455005055611729,
|
| 236 |
+
"grad_norm": 0.828125,
|
| 237 |
+
"learning_rate": 0.001,
|
| 238 |
+
"loss": 3.144283103942871,
|
| 239 |
"step": 540
|
| 240 |
},
|
| 241 |
{
|
| 242 |
"epoch": 0.04718570947084597,
|
| 243 |
+
"grad_norm": 1.09375,
|
| 244 |
+
"learning_rate": 0.001,
|
| 245 |
+
"loss": 3.1321605682373046,
|
| 246 |
"step": 560
|
| 247 |
},
|
| 248 |
{
|
| 249 |
"epoch": 0.04887091338051904,
|
| 250 |
+
"grad_norm": 0.94140625,
|
| 251 |
+
"learning_rate": 0.001,
|
| 252 |
+
"loss": 3.0999223709106447,
|
| 253 |
"step": 580
|
| 254 |
},
|
| 255 |
{
|
| 256 |
"epoch": 0.05055611729019211,
|
| 257 |
+
"grad_norm": 0.93359375,
|
| 258 |
+
"learning_rate": 0.001,
|
| 259 |
+
"loss": 3.0892059326171877,
|
| 260 |
"step": 600
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"epoch": 0.05055611729019211,
|
| 264 |
+
"eval_loss": 3.084484100341797,
|
| 265 |
+
"eval_runtime": 7.5354,
|
| 266 |
+
"eval_samples_per_second": 1264.304,
|
| 267 |
+
"eval_steps_per_second": 0.929,
|
| 268 |
"step": 600
|
| 269 |
},
|
| 270 |
{
|
| 271 |
"epoch": 0.05224132119986518,
|
| 272 |
+
"grad_norm": 0.90625,
|
| 273 |
+
"learning_rate": 0.001,
|
| 274 |
+
"loss": 3.0873672485351564,
|
| 275 |
"step": 620
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"epoch": 0.053926525109538256,
|
| 279 |
+
"grad_norm": 1.0234375,
|
| 280 |
+
"learning_rate": 0.001,
|
| 281 |
+
"loss": 3.0600521087646486,
|
| 282 |
"step": 640
|
| 283 |
},
|
| 284 |
{
|
| 285 |
"epoch": 0.055611729019211326,
|
| 286 |
+
"grad_norm": 0.81640625,
|
| 287 |
+
"learning_rate": 0.001,
|
| 288 |
+
"loss": 3.0599395751953127,
|
| 289 |
"step": 660
|
| 290 |
},
|
| 291 |
{
|
| 292 |
"epoch": 0.057296932928884395,
|
| 293 |
+
"grad_norm": 0.98828125,
|
| 294 |
+
"learning_rate": 0.001,
|
| 295 |
+
"loss": 3.0222793579101563,
|
| 296 |
"step": 680
|
| 297 |
},
|
| 298 |
{
|
| 299 |
"epoch": 0.058982136838557464,
|
| 300 |
+
"grad_norm": 1.03125,
|
| 301 |
+
"learning_rate": 0.001,
|
| 302 |
+
"loss": 3.005843734741211,
|
| 303 |
"step": 700
|
| 304 |
},
|
| 305 |
{
|
| 306 |
"epoch": 0.058982136838557464,
|
| 307 |
+
"eval_loss": 3.0020792484283447,
|
| 308 |
+
"eval_runtime": 7.4884,
|
| 309 |
+
"eval_samples_per_second": 1272.237,
|
| 310 |
+
"eval_steps_per_second": 0.935,
|
| 311 |
"step": 700
|
| 312 |
},
|
| 313 |
{
|
| 314 |
"epoch": 0.06066734074823053,
|
| 315 |
+
"grad_norm": 0.90625,
|
| 316 |
+
"learning_rate": 0.001,
|
| 317 |
+
"loss": 2.989686965942383,
|
| 318 |
"step": 720
|
| 319 |
},
|
| 320 |
{
|
| 321 |
"epoch": 0.06235254465790361,
|
| 322 |
+
"grad_norm": 1.1953125,
|
| 323 |
+
"learning_rate": 0.001,
|
| 324 |
+
"loss": 2.98439884185791,
|
| 325 |
"step": 740
|
| 326 |
},
|
| 327 |
{
|
| 328 |
"epoch": 0.06403774856757667,
|
| 329 |
+
"grad_norm": 1.1328125,
|
| 330 |
+
"learning_rate": 0.001,
|
| 331 |
+
"loss": 2.956541061401367,
|
| 332 |
"step": 760
|
| 333 |
},
|
| 334 |
{
|
| 335 |
"epoch": 0.06572295247724974,
|
| 336 |
+
"grad_norm": 0.75390625,
|
| 337 |
+
"learning_rate": 0.001,
|
| 338 |
+
"loss": 2.9446905136108397,
|
| 339 |
"step": 780
|
| 340 |
},
|
| 341 |
{
|
| 342 |
"epoch": 0.06740815638692282,
|
| 343 |
+
"grad_norm": 0.88671875,
|
| 344 |
+
"learning_rate": 0.001,
|
| 345 |
+
"loss": 2.953716850280762,
|
| 346 |
"step": 800
|
| 347 |
},
|
| 348 |
{
|
| 349 |
"epoch": 0.06740815638692282,
|
| 350 |
+
"eval_loss": 2.9370172023773193,
|
| 351 |
+
"eval_runtime": 7.5152,
|
| 352 |
+
"eval_samples_per_second": 1267.706,
|
| 353 |
+
"eval_steps_per_second": 0.931,
|
| 354 |
"step": 800
|
| 355 |
},
|
| 356 |
{
|
| 357 |
"epoch": 0.0690933602965959,
|
| 358 |
+
"grad_norm": 0.7421875,
|
| 359 |
+
"learning_rate": 0.001,
|
| 360 |
+
"loss": 2.9271785736083986,
|
| 361 |
"step": 820
|
| 362 |
},
|
| 363 |
{
|
| 364 |
"epoch": 0.07077856420626896,
|
| 365 |
+
"grad_norm": 0.8828125,
|
| 366 |
+
"learning_rate": 0.001,
|
| 367 |
+
"loss": 2.9229846954345704,
|
| 368 |
"step": 840
|
| 369 |
},
|
| 370 |
{
|
| 371 |
"epoch": 0.07246376811594203,
|
| 372 |
+
"grad_norm": 0.80078125,
|
| 373 |
+
"learning_rate": 0.001,
|
| 374 |
+
"loss": 2.909313774108887,
|
| 375 |
"step": 860
|
| 376 |
},
|
| 377 |
{
|
| 378 |
"epoch": 0.0741489720256151,
|
| 379 |
+
"grad_norm": 1.1328125,
|
| 380 |
+
"learning_rate": 0.001,
|
| 381 |
+
"loss": 2.9065679550170898,
|
| 382 |
"step": 880
|
| 383 |
},
|
| 384 |
{
|
| 385 |
"epoch": 0.07583417593528817,
|
| 386 |
+
"grad_norm": 0.9921875,
|
| 387 |
+
"learning_rate": 0.001,
|
| 388 |
+
"loss": 2.868302917480469,
|
| 389 |
"step": 900
|
| 390 |
},
|
| 391 |
{
|
| 392 |
"epoch": 0.07583417593528817,
|
| 393 |
+
"eval_loss": 2.8896827697753906,
|
| 394 |
+
"eval_runtime": 7.735,
|
| 395 |
+
"eval_samples_per_second": 1231.67,
|
| 396 |
+
"eval_steps_per_second": 0.905,
|
| 397 |
"step": 900
|
| 398 |
}
|
| 399 |
],
|
zain/Activation/out/mlp-linear-3L_run/checkpoint-900/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:69f6cee920e32c4b94b50a7a99ad53c35ddac080df64aa3ba5950fa167685ce5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2036216
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:86881b88db1d7407158e4b85e34b7ba9e34a2e5052213243fc0962ba82adbe96
|
| 3 |
size 2036216
|
zain/Activation/out/mlp-linear-3L_run/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4920
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:69f6cee920e32c4b94b50a7a99ad53c35ddac080df64aa3ba5950fa167685ce5
|
| 3 |
size 4920
|
zain/Activation/out/mlp-linear-3L_run/training_log.jsonl
CHANGED
|
@@ -548,3 +548,12 @@
|
|
| 548 |
{"step": 860, "epoch": 0.07246376811594203, "timestamp": 1787159663.8000612, "loss": 2.909313774108887, "grad_norm": 0.80078125, "learning_rate": 0.001, "train/total_time_seconds": 15.076439164578915, "train/time_per_step_avg": 0.016551108807325365, "train/epoch_time_elapsed": 127.67662629112601, "train/estimated_remaining_minutes": 0.0409050675007955}
|
| 549 |
{"step": 880, "epoch": 0.0741489720256151, "timestamp": 1787159665.3079937, "loss": 2.9065679550170898, "grad_norm": 1.1328125, "learning_rate": 0.001, "train/total_time_seconds": 15.406978622078896, "train/time_per_step_avg": 0.01655862484127283, "train/epoch_time_elapsed": 129.18455854058266, "train/estimated_remaining_minutes": 0.035015860504724765}
|
| 550 |
{"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159666.8079824, "loss": 2.868302917480469, "grad_norm": 0.9921875, "learning_rate": 0.001, "train/total_time_seconds": 15.737355303019285, "train/time_per_step_avg": 0.01655557971447706, "train/epoch_time_elapsed": 130.68454774469137, "train/estimated_remaining_minutes": 0.029143250561146822}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 548 |
{"step": 860, "epoch": 0.07246376811594203, "timestamp": 1787159663.8000612, "loss": 2.909313774108887, "grad_norm": 0.80078125, "learning_rate": 0.001, "train/total_time_seconds": 15.076439164578915, "train/time_per_step_avg": 0.016551108807325365, "train/epoch_time_elapsed": 127.67662629112601, "train/estimated_remaining_minutes": 0.0409050675007955}
|
| 549 |
{"step": 880, "epoch": 0.0741489720256151, "timestamp": 1787159665.3079937, "loss": 2.9065679550170898, "grad_norm": 1.1328125, "learning_rate": 0.001, "train/total_time_seconds": 15.406978622078896, "train/time_per_step_avg": 0.01655862484127283, "train/epoch_time_elapsed": 129.18455854058266, "train/estimated_remaining_minutes": 0.035015860504724765}
|
| 550 |
{"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159666.8079824, "loss": 2.868302917480469, "grad_norm": 0.9921875, "learning_rate": 0.001, "train/total_time_seconds": 15.737355303019285, "train/time_per_step_avg": 0.01655557971447706, "train/epoch_time_elapsed": 130.68454774469137, "train/estimated_remaining_minutes": 0.029143250561146822}
|
| 551 |
+
{"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159674.5448534, "eval_loss": 2.8896827697753906, "eval_runtime": 7.735, "eval_samples_per_second": 1231.67, "eval_steps_per_second": 0.905, "train/total_time_seconds": 15.737355303019285, "train/time_per_step_avg": 0.01655557971447706, "train/epoch_time_elapsed": 138.42141803354025, "train/estimated_remaining_minutes": 0.029143250561146822}
|
| 552 |
+
{"step": 920, "epoch": 0.07751937984496124, "timestamp": 1787159676.0897264, "loss": 2.8754199981689452, "grad_norm": 0.8125, "learning_rate": 0.001, "train/total_time_seconds": 16.06913860887289, "train/time_per_step_avg": 0.01656209945678711, "train/epoch_time_elapsed": 139.96629093587399, "train/estimated_remaining_minutes": 0.02328860667952593}
|
| 553 |
+
{"step": 940, "epoch": 0.07920458375463431, "timestamp": 1787159677.5917127, "loss": 2.8692115783691405, "grad_norm": 0.8046875, "learning_rate": 0.001, "train/total_time_seconds": 16.399292301386595, "train/time_per_step_avg": 0.01655644714832306, "train/epoch_time_elapsed": 141.46827787905931, "train/estimated_remaining_minutes": 0.017446055639772973}
|
| 554 |
+
{"step": 960, "epoch": 0.08088978766430738, "timestamp": 1787159679.0892675, "loss": 2.8660629272460936, "grad_norm": 0.88671875, "learning_rate": 0.001, "train/total_time_seconds": 16.729277562350035, "train/time_per_step_avg": 0.0165283839777112, "train/epoch_time_elapsed": 142.96583257243037, "train/estimated_remaining_minutes": 0.01161755386274308}
|
| 555 |
+
{"step": 980, "epoch": 0.08257499157398045, "timestamp": 1787159680.5901031, "loss": 2.8608421325683593, "grad_norm": 0.73828125, "learning_rate": 0.001, "train/total_time_seconds": 17.05958204343915, "train/time_per_step_avg": 0.016526034213602544, "train/epoch_time_elapsed": 144.46666844561696, "train/estimated_remaining_minutes": 0.005802578926339846}
|
| 556 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159682.1642146, "loss": 2.8389766693115233, "grad_norm": 0.88671875, "learning_rate": 0.001, "train/total_time_seconds": 17.390934593975544, "train/time_per_step_avg": 0.01653579290956259, "train/epoch_time_elapsed": 146.0407790467143, "train/estimated_remaining_minutes": 0.0}
|
| 557 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159689.7607174, "eval_loss": 2.841240167617798, "eval_runtime": 7.5948, "eval_samples_per_second": 1254.405, "eval_steps_per_second": 0.922, "train/total_time_seconds": 17.390934593975544, "train/time_per_step_avg": 0.01653579290956259, "train/epoch_time_elapsed": 153.63728144019842, "train/estimated_remaining_minutes": 0.0}
|
| 558 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159689.8023045, "train_runtime": 154.6026, "train_samples_per_second": 517.456, "train_steps_per_second": 6.468, "total_flos": 121016156160000.0, "train_loss": 3.7283405075073244, "train/total_time_seconds": 17.390934593975544, "train/time_per_step_avg": 0.01653579290956259, "train/epoch_time_elapsed": 153.67886726930737, "train/estimated_remaining_minutes": 0.0}
|
| 559 |
+
{"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159697.355951, "eval_loss": 2.841240167617798, "eval_runtime": 7.5505, "eval_samples_per_second": 1261.772, "eval_steps_per_second": 0.927, "train/total_time_seconds": 17.390934593975544, "train/time_per_step_avg": 0.01653579290956259, "train/epoch_time_elapsed": 161.23251363635063, "train/estimated_remaining_minutes": 0.0}
|
zain/Activation/out/sweep_summary.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
"variant": "mlp-linear-3L",
|
| 4 |
-
"eval_loss":
|
| 5 |
"out": "out/mlp-linear-3L_run",
|
| 6 |
-
"run_name": "LM-mlp-linear-3L-1.0M-20260819-
|
| 7 |
"status": "success"
|
| 8 |
}
|
| 9 |
]
|
|
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
"variant": "mlp-linear-3L",
|
| 4 |
+
"eval_loss": 2.841240167617798,
|
| 5 |
"out": "out/mlp-linear-3L_run",
|
| 6 |
+
"run_name": "LM-mlp-linear-3L-1.0M-20260819-171214",
|
| 7 |
"status": "success"
|
| 8 |
}
|
| 9 |
]
|
zain/Activation/wandb/debug-internal.log
CHANGED
|
@@ -23,3 +23,15 @@
|
|
| 23 |
{"time":"2026-08-19T17:14:01.314421382Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
{"time":"2026-08-19T17:14:16.125013671Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":68,"console_lines":1}
|
| 25 |
{"time":"2026-08-19T17:14:16.73098493Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
{"time":"2026-08-19T17:14:01.314421382Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 24 |
{"time":"2026-08-19T17:14:16.125013671Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":68,"console_lines":1}
|
| 25 |
{"time":"2026-08-19T17:14:16.73098493Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 26 |
+
{"time":"2026-08-19T17:14:31.124676234Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":74,"console_lines":23}
|
| 27 |
+
{"time":"2026-08-19T17:14:31.25448083Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 28 |
+
{"time":"2026-08-19T17:14:46.124973259Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":6,"events_offset":17,"events_lines":2,"console_offset":90,"console_lines":1}
|
| 29 |
+
{"time":"2026-08-19T17:14:46.269794228Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 30 |
+
{"time":"2026-08-19T17:14:58.136247949Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 31 |
+
{"time":"2026-08-19T17:14:58.136517121Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":59,"history_lines":3,"events_offset":19,"events_lines":1,"console_offset":96,"console_lines":25,"uploaded_len":3,"complete":true,"exit_code":0}
|
| 32 |
+
{"time":"2026-08-19T17:14:58.277374581Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
|
| 33 |
+
{"time":"2026-08-19T17:14:58.278584446Z","level":"INFO","msg":"handler: operation stats","stats":{}}
|
| 34 |
+
{"time":"2026-08-19T17:14:58.281745009Z","level":"INFO","msg":"stream: finishing up"}
|
| 35 |
+
{"time":"2026-08-19T17:14:58.281775882Z","level":"INFO","msg":"handler: closed"}
|
| 36 |
+
{"time":"2026-08-19T17:14:58.281829879Z","level":"INFO","msg":"sender: closed"}
|
| 37 |
+
{"time":"2026-08-19T17:14:58.281833448Z","level":"INFO","msg":"stream: all finished"}
|
zain/Activation/wandb/debug.log
CHANGED
|
@@ -21,3 +21,8 @@ config: {'_wandb': {}}
|
|
| 21 |
2026-08-19 17:12:16,120 INFO MainThread:3920475 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 3, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-3L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-3L-1.0M-20260819-171214', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-3L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
2026-08-19 17:12:16,122 INFO MainThread:3920475 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 1016704 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14f219615650>>
|
| 23 |
2026-08-19 17:12:16,122 INFO MainThread:3920475 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 1016704 None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 21 |
2026-08-19 17:12:16,120 INFO MainThread:3920475 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 3, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-3L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-3L-1.0M-20260819-171214', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-3L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
|
| 22 |
2026-08-19 17:12:16,122 INFO MainThread:3920475 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 1016704 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x14f219615650>>
|
| 23 |
2026-08-19 17:12:16,122 INFO MainThread:3920475 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 1016704 None
|
| 24 |
+
2026-08-19 17:14:57,366 INFO MainThread:3920475 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate-checking/2nil4rqn
|
| 25 |
+
2026-08-19 17:14:57,367 INFO MainThread:3920475 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
|
| 26 |
+
2026-08-19 17:14:57,367 INFO MainThread:3920475 [wandb_run.py:_restore():2570] restore
|
| 27 |
+
2026-08-19 17:14:57,367 INFO MainThread:3920475 [wandb_run.py:_restore():2576] restore done
|
| 28 |
+
2026-08-19 17:14:58,281 INFO MainThread:3920475 [wandb_run.py:_footer_sync_info():3993] logging synced files
|
zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log
CHANGED
|
@@ -54,3 +54,11 @@
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 57 |
+
{"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 58 |
+
{"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 59 |
+
{"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
|
| 60 |
+
{"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
|
| 61 |
+
{"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
|
| 62 |
+
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
+
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
+
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log
CHANGED
|
@@ -54,3 +54,11 @@
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 57 |
+
{"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 58 |
+
{"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 59 |
+
{"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
|
| 60 |
+
{"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
|
| 61 |
+
{"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
|
| 62 |
+
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
+
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
+
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log
CHANGED
|
@@ -54,3 +54,11 @@
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 57 |
+
{"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 58 |
+
{"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 59 |
+
{"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
|
| 60 |
+
{"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
|
| 61 |
+
{"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
|
| 62 |
+
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
+
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
+
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|
zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log
CHANGED
|
@@ -54,3 +54,11 @@
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
{"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
|
| 55 |
{"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
|
| 56 |
{"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 57 |
+
{"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 58 |
+
{"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
|
| 59 |
+
{"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
|
| 60 |
+
{"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
|
| 61 |
+
{"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
|
| 62 |
+
{"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
|
| 63 |
+
{"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
|
| 64 |
+
{"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
|