w-ahmad commited on
Commit
08afdea
·
verified ·
1 Parent(s): 92a26d2

Auto upload zain 2026-08-19T17:31:52.723720 (part 2)

Browse files
Files changed (49) hide show
  1. .gitattributes +1 -0
  2. zain/Activation/out/mlp-linear-9L_run/checkpoint-400/scheduler.pt +1 -1
  3. zain/Activation/out/mlp-linear-9L_run/checkpoint-400/trainer_state.json +104 -104
  4. zain/Activation/out/mlp-linear-9L_run/checkpoint-400/training_args.bin +1 -1
  5. zain/Activation/out/mlp-linear-9L_run/checkpoint-500/model.safetensors +1 -1
  6. zain/Activation/out/mlp-linear-9L_run/checkpoint-500/optimizer.pt +1 -1
  7. zain/Activation/out/mlp-linear-9L_run/checkpoint-500/scheduler.pt +1 -1
  8. zain/Activation/out/mlp-linear-9L_run/checkpoint-500/trainer_state.json +129 -129
  9. zain/Activation/out/mlp-linear-9L_run/checkpoint-500/training_args.bin +1 -1
  10. zain/Activation/out/mlp-linear-9L_run/checkpoint-600/model.safetensors +1 -1
  11. zain/Activation/out/mlp-linear-9L_run/checkpoint-600/optimizer.pt +1 -1
  12. zain/Activation/out/mlp-linear-9L_run/checkpoint-600/scheduler.pt +1 -1
  13. zain/Activation/out/mlp-linear-9L_run/checkpoint-600/trainer_state.json +154 -154
  14. zain/Activation/out/mlp-linear-9L_run/checkpoint-600/training_args.bin +1 -1
  15. zain/Activation/out/mlp-linear-9L_run/checkpoint-700/model.safetensors +1 -1
  16. zain/Activation/out/mlp-linear-9L_run/checkpoint-700/optimizer.pt +1 -1
  17. zain/Activation/out/mlp-linear-9L_run/checkpoint-700/scheduler.pt +1 -1
  18. zain/Activation/out/mlp-linear-9L_run/checkpoint-700/trainer_state.json +179 -179
  19. zain/Activation/out/mlp-linear-9L_run/checkpoint-700/training_args.bin +1 -1
  20. zain/Activation/out/mlp-linear-9L_run/checkpoint-800/model.safetensors +1 -1
  21. zain/Activation/out/mlp-linear-9L_run/checkpoint-800/optimizer.pt +1 -1
  22. zain/Activation/out/mlp-linear-9L_run/checkpoint-800/scheduler.pt +1 -1
  23. zain/Activation/out/mlp-linear-9L_run/checkpoint-800/trainer_state.json +204 -204
  24. zain/Activation/out/mlp-linear-9L_run/checkpoint-800/training_args.bin +1 -1
  25. zain/Activation/out/mlp-linear-9L_run/checkpoint-900/model.safetensors +1 -1
  26. zain/Activation/out/mlp-linear-9L_run/checkpoint-900/optimizer.pt +1 -1
  27. zain/Activation/out/mlp-linear-9L_run/checkpoint-900/scheduler.pt +1 -1
  28. zain/Activation/out/mlp-linear-9L_run/checkpoint-900/trainer_state.json +229 -229
  29. zain/Activation/out/mlp-linear-9L_run/checkpoint-900/training_args.bin +1 -1
  30. zain/Activation/out/mlp-linear-9L_run/model.safetensors +1 -1
  31. zain/Activation/out/mlp-linear-9L_run/training_args.bin +1 -1
  32. zain/Activation/out/mlp-linear-9L_run/training_log.jsonl +62 -152
  33. zain/Activation/out/sweep_summary.json +4 -4
  34. zain/Activation/wandb/debug-internal.log +32 -0
  35. zain/Activation/wandb/debug.log +5 -0
  36. zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log +9 -0
  37. zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log +9 -0
  38. zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log +9 -0
  39. zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log +9 -0
  40. zain/Activation/wandb/run-20260819_171215-2nil4rqn/logs/debug-core.log +12 -0
  41. zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/config.yaml +435 -0
  42. zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/output.log +121 -0
  43. zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/requirements.txt +149 -0
  44. zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-metadata.json +95 -0
  45. zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-summary.json +1 -0
  46. zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-core.log +76 -0
  47. zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log +41 -0
  48. zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log +28 -0
  49. zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb +3 -0
.gitattributes CHANGED
@@ -143,3 +143,4 @@ zain/Activation/wandb/run-20260819_163845-74tq2syl/run-74tq2syl.wandb filter=lfs
143
  zain/Activation/wandb/run-20260819_164553-44x03ghu/run-44x03ghu.wandb filter=lfs diff=lfs merge=lfs -text
144
  zain/Activation/wandb/run-20260819_170854-zhg4u13t/run-zhg4u13t.wandb filter=lfs diff=lfs merge=lfs -text
145
  zain/Activation/wandb/run-20260819_171215-2nil4rqn/run-2nil4rqn.wandb filter=lfs diff=lfs merge=lfs -text
 
 
143
  zain/Activation/wandb/run-20260819_164553-44x03ghu/run-44x03ghu.wandb filter=lfs diff=lfs merge=lfs -text
144
  zain/Activation/wandb/run-20260819_170854-zhg4u13t/run-zhg4u13t.wandb filter=lfs diff=lfs merge=lfs -text
145
  zain/Activation/wandb/run-20260819_171215-2nil4rqn/run-2nil4rqn.wandb filter=lfs diff=lfs merge=lfs -text
146
+ zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb filter=lfs diff=lfs merge=lfs -text
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f768f3dc231ac5aded23c6f019b283714352126180577c2c811e2d205bc7704f
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af1ea62c89448929159c1b05ebc75e926e6be85e7ce3551310793c351ba3ade1
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.013481631277384564,
6
  "eval_steps": 100,
7
  "global_step": 400,
8
  "is_hyper_param_search": false,
@@ -10,180 +10,180 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  }
184
  ],
185
  "logging_steps": 20,
186
- "max_steps": 2500,
187
  "num_input_tokens_seen": 0,
188
  "num_train_epochs": 1,
189
  "save_steps": 100,
@@ -199,8 +199,8 @@
199
  "attributes": {}
200
  }
201
  },
202
- "total_flos": 58077688627200.0,
203
- "train_batch_size": 32,
204
  "trial_name": null,
205
  "trial_params": null
206
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.03370407819346141,
6
  "eval_steps": 100,
7
  "global_step": 400,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  }
184
  ],
185
  "logging_steps": 20,
186
+ "max_steps": 1000,
187
  "num_input_tokens_seen": 0,
188
  "num_train_epochs": 1,
189
  "save_steps": 100,
 
199
  "attributes": {}
200
  }
201
  },
202
+ "total_flos": 145194221568000.0,
203
+ "train_batch_size": 80,
204
  "trial_name": null,
205
  "trial_params": null
206
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-400/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5872fbf4c61dbb81c15291d837bafe638d1bfbbcf6e5867744b7390e99c1ad18
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5712d6f569182e98a2720f28f3ef489d602a3c6473ce0b6e3c4f1f7907e3c8cf
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:45bc91dd0e3cd9af5c4d382e7c9e83a79b2fee89488ede8a44ab5697a568f0d1
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63b2d05dc2871dc45cdfc3ecc136218a493535e4bde6da2a670418746fe34818
3
  size 8068282
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7aade0a5cf98168c48557061a263a0ce94127833f42effbb1795ad03b5bd88a5
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b11eb900cb7af9ecb8797771dfa83159f999b3d7b4f3c5b2ff6e19c51da8de4a
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.016852039096730706,
6
  "eval_steps": 100,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
@@ -10,223 +10,223 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 2.140625,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.935818481445312,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.1875,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.853317642211914,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.5390625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.7492321014404295,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.921875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.629489898681641,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.7890625,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.565845108032226,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.530772686004639,
222
- "eval_runtime": 8.4284,
223
- "eval_samples_per_second": 1130.345,
224
- "eval_steps_per_second": 0.831,
225
  "step": 500
226
  }
227
  ],
228
  "logging_steps": 20,
229
- "max_steps": 2500,
230
  "num_input_tokens_seen": 0,
231
  "num_train_epochs": 1,
232
  "save_steps": 100,
@@ -242,8 +242,8 @@
242
  "attributes": {}
243
  }
244
  },
245
- "total_flos": 72597110784000.0,
246
- "train_batch_size": 32,
247
  "trial_name": null,
248
  "trial_params": null
249
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.04213009774182676,
6
  "eval_steps": 100,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  },
184
  {
185
+ "epoch": 0.03538928210313448,
186
+ "grad_norm": 0.7109375,
187
+ "learning_rate": 0.001,
188
+ "loss": 3.3047794342041015,
189
  "step": 420
190
  },
191
  {
192
+ "epoch": 0.03707448601280755,
193
+ "grad_norm": 0.71875,
194
+ "learning_rate": 0.001,
195
+ "loss": 3.258700942993164,
196
  "step": 440
197
  },
198
  {
199
+ "epoch": 0.03875968992248062,
200
+ "grad_norm": 0.73046875,
201
+ "learning_rate": 0.001,
202
+ "loss": 3.229313278198242,
203
  "step": 460
204
  },
205
  {
206
+ "epoch": 0.04044489383215369,
207
+ "grad_norm": 0.703125,
208
+ "learning_rate": 0.001,
209
+ "loss": 3.202067565917969,
210
  "step": 480
211
  },
212
  {
213
+ "epoch": 0.04213009774182676,
214
+ "grad_norm": 0.7265625,
215
+ "learning_rate": 0.001,
216
+ "loss": 3.163772201538086,
217
  "step": 500
218
  },
219
  {
220
+ "epoch": 0.04213009774182676,
221
+ "eval_loss": 3.157355308532715,
222
+ "eval_runtime": 7.9503,
223
+ "eval_samples_per_second": 1198.315,
224
+ "eval_steps_per_second": 0.88,
225
  "step": 500
226
  }
227
  ],
228
  "logging_steps": 20,
229
+ "max_steps": 1000,
230
  "num_input_tokens_seen": 0,
231
  "num_train_epochs": 1,
232
  "save_steps": 100,
 
242
  "attributes": {}
243
  }
244
  },
245
+ "total_flos": 181492776960000.0,
246
+ "train_batch_size": 80,
247
  "trial_name": null,
248
  "trial_params": null
249
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a4c2f0e7e1bd95005c2e46113faf5cd14e9b94fc90f937ea9d51aeb93a33a04e
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fdd583528bce16f5f52bc31609f36b44fc4dc0de87540acf616f1442cb07fde9
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:946a1da1fa4a395f840276f2b24058ef226b048d8eab3397ee86488c7d82670a
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8fe1053b1cf05e779ceab88f2312865667a8cb4999b4439b1a72908989024233
3
  size 8068282
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:295d5fdc0e0bbae04d3db91b9c6051a7cccac49c004495947e95dcd6c59adda3
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:31ff2c480a5c30c49d6c8b2f5ae72d7dc39d2d3f7060f88a23705954d049a506
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.020222446916076844,
6
  "eval_steps": 100,
7
  "global_step": 600,
8
  "is_hyper_param_search": false,
@@ -10,266 +10,266 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 2.140625,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.935818481445312,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.1875,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.853317642211914,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.5390625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.7492321014404295,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.921875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.629489898681641,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.7890625,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.565845108032226,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.530772686004639,
222
- "eval_runtime": 8.4284,
223
- "eval_samples_per_second": 1130.345,
224
- "eval_steps_per_second": 0.831,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.140625,
230
- "learning_rate": 0.0003,
231
- "loss": 4.50256233215332,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.7265625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.419484710693359,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.3125,
244
- "learning_rate": 0.0003,
245
- "loss": 4.363323593139649,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.5,
251
- "learning_rate": 0.0003,
252
- "loss": 4.3072765350341795,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.9765625,
258
- "learning_rate": 0.0003,
259
- "loss": 4.263522338867188,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.239859580993652,
265
- "eval_runtime": 8.5836,
266
- "eval_samples_per_second": 1109.908,
267
- "eval_steps_per_second": 0.816,
268
  "step": 600
269
  }
270
  ],
271
  "logging_steps": 20,
272
- "max_steps": 2500,
273
  "num_input_tokens_seen": 0,
274
  "num_train_epochs": 1,
275
  "save_steps": 100,
@@ -285,8 +285,8 @@
285
  "attributes": {}
286
  }
287
  },
288
- "total_flos": 87116532940800.0,
289
- "train_batch_size": 32,
290
  "trial_name": null,
291
  "trial_params": null
292
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.05055611729019211,
6
  "eval_steps": 100,
7
  "global_step": 600,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  },
184
  {
185
+ "epoch": 0.03538928210313448,
186
+ "grad_norm": 0.7109375,
187
+ "learning_rate": 0.001,
188
+ "loss": 3.3047794342041015,
189
  "step": 420
190
  },
191
  {
192
+ "epoch": 0.03707448601280755,
193
+ "grad_norm": 0.71875,
194
+ "learning_rate": 0.001,
195
+ "loss": 3.258700942993164,
196
  "step": 440
197
  },
198
  {
199
+ "epoch": 0.03875968992248062,
200
+ "grad_norm": 0.73046875,
201
+ "learning_rate": 0.001,
202
+ "loss": 3.229313278198242,
203
  "step": 460
204
  },
205
  {
206
+ "epoch": 0.04044489383215369,
207
+ "grad_norm": 0.703125,
208
+ "learning_rate": 0.001,
209
+ "loss": 3.202067565917969,
210
  "step": 480
211
  },
212
  {
213
+ "epoch": 0.04213009774182676,
214
+ "grad_norm": 0.7265625,
215
+ "learning_rate": 0.001,
216
+ "loss": 3.163772201538086,
217
  "step": 500
218
  },
219
  {
220
+ "epoch": 0.04213009774182676,
221
+ "eval_loss": 3.157355308532715,
222
+ "eval_runtime": 7.9503,
223
+ "eval_samples_per_second": 1198.315,
224
+ "eval_steps_per_second": 0.88,
225
  "step": 500
226
  },
227
  {
228
+ "epoch": 0.043815301651499834,
229
+ "grad_norm": 0.6953125,
230
+ "learning_rate": 0.001,
231
+ "loss": 3.1448848724365233,
232
  "step": 520
233
  },
234
  {
235
+ "epoch": 0.0455005055611729,
236
+ "grad_norm": 0.828125,
237
+ "learning_rate": 0.001,
238
+ "loss": 3.103527069091797,
239
  "step": 540
240
  },
241
  {
242
+ "epoch": 0.04718570947084597,
243
+ "grad_norm": 0.7578125,
244
+ "learning_rate": 0.001,
245
+ "loss": 3.08404541015625,
246
  "step": 560
247
  },
248
  {
249
+ "epoch": 0.04887091338051904,
250
+ "grad_norm": 0.76953125,
251
+ "learning_rate": 0.001,
252
+ "loss": 3.0501741409301757,
253
  "step": 580
254
  },
255
  {
256
+ "epoch": 0.05055611729019211,
257
+ "grad_norm": 0.96875,
258
+ "learning_rate": 0.001,
259
+ "loss": 3.0376760482788088,
260
  "step": 600
261
  },
262
  {
263
+ "epoch": 0.05055611729019211,
264
+ "eval_loss": 3.0310890674591064,
265
+ "eval_runtime": 8.1195,
266
+ "eval_samples_per_second": 1173.346,
267
+ "eval_steps_per_second": 0.862,
268
  "step": 600
269
  }
270
  ],
271
  "logging_steps": 20,
272
+ "max_steps": 1000,
273
  "num_input_tokens_seen": 0,
274
  "num_train_epochs": 1,
275
  "save_steps": 100,
 
285
  "attributes": {}
286
  }
287
  },
288
+ "total_flos": 217791332352000.0,
289
+ "train_batch_size": 80,
290
  "trial_name": null,
291
  "trial_params": null
292
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-600/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1dce8189770ce838ce3e65dd567433fe02803f116d52a94c356237378a46e5c1
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:97deb4d2fe5491f671435ead6997073978dc5974641a5f915ea901926274c1a1
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:58e5fb749b87b488b4c7fc432602eba6ba868802571d6b14fdf6a60ec4240857
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1f5e62abb7b4309d85d8e7e3f37e3449916b0bac155e62f602e3bbd0aa0a7951
3
  size 8068282
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d6c38b4e6128ebabd81a22bafe942090015968e75f90e5e9ef25cd5ebdef8357
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dd7c397523408b6e567c3990f49ad726aa8ff2c476b37da2c996606b1b8ddb75
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.023592854735422986,
6
  "eval_steps": 100,
7
  "global_step": 700,
8
  "is_hyper_param_search": false,
@@ -10,309 +10,309 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 2.140625,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.935818481445312,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.1875,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.853317642211914,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.5390625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.7492321014404295,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.921875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.629489898681641,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.7890625,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.565845108032226,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.530772686004639,
222
- "eval_runtime": 8.4284,
223
- "eval_samples_per_second": 1130.345,
224
- "eval_steps_per_second": 0.831,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.140625,
230
- "learning_rate": 0.0003,
231
- "loss": 4.50256233215332,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.7265625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.419484710693359,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.3125,
244
- "learning_rate": 0.0003,
245
- "loss": 4.363323593139649,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.5,
251
- "learning_rate": 0.0003,
252
- "loss": 4.3072765350341795,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.9765625,
258
- "learning_rate": 0.0003,
259
- "loss": 4.263522338867188,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.239859580993652,
265
- "eval_runtime": 8.5836,
266
- "eval_samples_per_second": 1109.908,
267
- "eval_steps_per_second": 0.816,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.2265625,
273
- "learning_rate": 0.0003,
274
- "loss": 4.207575988769531,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.3359375,
280
- "learning_rate": 0.0003,
281
- "loss": 4.179453659057617,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.09375,
287
- "learning_rate": 0.0003,
288
- "loss": 4.13963508605957,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 1.328125,
294
- "learning_rate": 0.0003,
295
- "loss": 4.0913646697998045,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.609375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.082810974121093,
303
  "step": 700
304
  },
305
  {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.061285495758057,
308
- "eval_runtime": 8.3862,
309
- "eval_samples_per_second": 1136.028,
310
- "eval_steps_per_second": 0.835,
311
  "step": 700
312
  }
313
  ],
314
  "logging_steps": 20,
315
- "max_steps": 2500,
316
  "num_input_tokens_seen": 0,
317
  "num_train_epochs": 1,
318
  "save_steps": 100,
@@ -328,8 +328,8 @@
328
  "attributes": {}
329
  }
330
  },
331
- "total_flos": 101635955097600.0,
332
- "train_batch_size": 32,
333
  "trial_name": null,
334
  "trial_params": null
335
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.058982136838557464,
6
  "eval_steps": 100,
7
  "global_step": 700,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  },
184
  {
185
+ "epoch": 0.03538928210313448,
186
+ "grad_norm": 0.7109375,
187
+ "learning_rate": 0.001,
188
+ "loss": 3.3047794342041015,
189
  "step": 420
190
  },
191
  {
192
+ "epoch": 0.03707448601280755,
193
+ "grad_norm": 0.71875,
194
+ "learning_rate": 0.001,
195
+ "loss": 3.258700942993164,
196
  "step": 440
197
  },
198
  {
199
+ "epoch": 0.03875968992248062,
200
+ "grad_norm": 0.73046875,
201
+ "learning_rate": 0.001,
202
+ "loss": 3.229313278198242,
203
  "step": 460
204
  },
205
  {
206
+ "epoch": 0.04044489383215369,
207
+ "grad_norm": 0.703125,
208
+ "learning_rate": 0.001,
209
+ "loss": 3.202067565917969,
210
  "step": 480
211
  },
212
  {
213
+ "epoch": 0.04213009774182676,
214
+ "grad_norm": 0.7265625,
215
+ "learning_rate": 0.001,
216
+ "loss": 3.163772201538086,
217
  "step": 500
218
  },
219
  {
220
+ "epoch": 0.04213009774182676,
221
+ "eval_loss": 3.157355308532715,
222
+ "eval_runtime": 7.9503,
223
+ "eval_samples_per_second": 1198.315,
224
+ "eval_steps_per_second": 0.88,
225
  "step": 500
226
  },
227
  {
228
+ "epoch": 0.043815301651499834,
229
+ "grad_norm": 0.6953125,
230
+ "learning_rate": 0.001,
231
+ "loss": 3.1448848724365233,
232
  "step": 520
233
  },
234
  {
235
+ "epoch": 0.0455005055611729,
236
+ "grad_norm": 0.828125,
237
+ "learning_rate": 0.001,
238
+ "loss": 3.103527069091797,
239
  "step": 540
240
  },
241
  {
242
+ "epoch": 0.04718570947084597,
243
+ "grad_norm": 0.7578125,
244
+ "learning_rate": 0.001,
245
+ "loss": 3.08404541015625,
246
  "step": 560
247
  },
248
  {
249
+ "epoch": 0.04887091338051904,
250
+ "grad_norm": 0.76953125,
251
+ "learning_rate": 0.001,
252
+ "loss": 3.0501741409301757,
253
  "step": 580
254
  },
255
  {
256
+ "epoch": 0.05055611729019211,
257
+ "grad_norm": 0.96875,
258
+ "learning_rate": 0.001,
259
+ "loss": 3.0376760482788088,
260
  "step": 600
261
  },
262
  {
263
+ "epoch": 0.05055611729019211,
264
+ "eval_loss": 3.0310890674591064,
265
+ "eval_runtime": 8.1195,
266
+ "eval_samples_per_second": 1173.346,
267
+ "eval_steps_per_second": 0.862,
268
  "step": 600
269
  },
270
  {
271
+ "epoch": 0.05224132119986518,
272
+ "grad_norm": 0.82421875,
273
+ "learning_rate": 0.001,
274
+ "loss": 3.0314205169677733,
275
  "step": 620
276
  },
277
  {
278
+ "epoch": 0.053926525109538256,
279
+ "grad_norm": 0.640625,
280
+ "learning_rate": 0.001,
281
+ "loss": 3.0014928817749023,
282
  "step": 640
283
  },
284
  {
285
+ "epoch": 0.055611729019211326,
286
+ "grad_norm": 0.70703125,
287
+ "learning_rate": 0.001,
288
+ "loss": 2.9963293075561523,
289
  "step": 660
290
  },
291
  {
292
+ "epoch": 0.057296932928884395,
293
+ "grad_norm": 0.7109375,
294
+ "learning_rate": 0.001,
295
+ "loss": 2.9568761825561523,
296
  "step": 680
297
  },
298
  {
299
+ "epoch": 0.058982136838557464,
300
+ "grad_norm": 0.69921875,
301
+ "learning_rate": 0.001,
302
+ "loss": 2.9395275115966797,
303
  "step": 700
304
  },
305
  {
306
+ "epoch": 0.058982136838557464,
307
+ "eval_loss": 2.939429759979248,
308
+ "eval_runtime": 7.975,
309
+ "eval_samples_per_second": 1194.602,
310
+ "eval_steps_per_second": 0.878,
311
  "step": 700
312
  }
313
  ],
314
  "logging_steps": 20,
315
+ "max_steps": 1000,
316
  "num_input_tokens_seen": 0,
317
  "num_train_epochs": 1,
318
  "save_steps": 100,
 
328
  "attributes": {}
329
  }
330
  },
331
+ "total_flos": 254089887744000.0,
332
+ "train_batch_size": 80,
333
  "trial_name": null,
334
  "trial_params": null
335
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-700/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:75a3bd4fb21fd4ee5917ef4a3effa2e4004eaa865437229e112bf6ed25224797
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c337edbdecc4bf136ef452013ba3dadf0cb21c44e46ffe85d9ad094448bd625
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32854c5978c8b2f1d33b5d75fb09ef38c174912a2701efac761ca76686bc3446
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:99e89d2c66df90ceb6b05584fce25f5cd08caf172652dfb64c617b95863d99bf
3
  size 8068282
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2224470b783e3f05273b6a4ac609fa6ba76b1099ae64a843a85fea2d2bf1e0dd
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:71e4310b54de76d098945960239794c83dba2710505618327c0dcd1468c3497f
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.026963262554769128,
6
  "eval_steps": 100,
7
  "global_step": 800,
8
  "is_hyper_param_search": false,
@@ -10,352 +10,352 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 2.140625,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.935818481445312,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.1875,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.853317642211914,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.5390625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.7492321014404295,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.921875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.629489898681641,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.7890625,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.565845108032226,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.530772686004639,
222
- "eval_runtime": 8.4284,
223
- "eval_samples_per_second": 1130.345,
224
- "eval_steps_per_second": 0.831,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.140625,
230
- "learning_rate": 0.0003,
231
- "loss": 4.50256233215332,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.7265625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.419484710693359,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.3125,
244
- "learning_rate": 0.0003,
245
- "loss": 4.363323593139649,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.5,
251
- "learning_rate": 0.0003,
252
- "loss": 4.3072765350341795,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.9765625,
258
- "learning_rate": 0.0003,
259
- "loss": 4.263522338867188,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.239859580993652,
265
- "eval_runtime": 8.5836,
266
- "eval_samples_per_second": 1109.908,
267
- "eval_steps_per_second": 0.816,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.2265625,
273
- "learning_rate": 0.0003,
274
- "loss": 4.207575988769531,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.3359375,
280
- "learning_rate": 0.0003,
281
- "loss": 4.179453659057617,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.09375,
287
- "learning_rate": 0.0003,
288
- "loss": 4.13963508605957,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 1.328125,
294
- "learning_rate": 0.0003,
295
- "loss": 4.0913646697998045,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.609375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.082810974121093,
303
  "step": 700
304
  },
305
  {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.061285495758057,
308
- "eval_runtime": 8.3862,
309
- "eval_samples_per_second": 1136.028,
310
- "eval_steps_per_second": 0.835,
311
  "step": 700
312
  },
313
  {
314
- "epoch": 0.024266936299292215,
315
- "grad_norm": 1.453125,
316
- "learning_rate": 0.0003,
317
- "loss": 4.052593612670899,
318
  "step": 720
319
  },
320
  {
321
- "epoch": 0.02494101786316144,
322
- "grad_norm": 1.46875,
323
- "learning_rate": 0.0003,
324
- "loss": 4.038698196411133,
325
  "step": 740
326
  },
327
  {
328
- "epoch": 0.02561509942703067,
329
- "grad_norm": 1.671875,
330
- "learning_rate": 0.0003,
331
- "loss": 3.99466552734375,
332
  "step": 760
333
  },
334
  {
335
- "epoch": 0.0262891809908999,
336
- "grad_norm": 1.3671875,
337
- "learning_rate": 0.0003,
338
- "loss": 3.9530941009521485,
339
  "step": 780
340
  },
341
  {
342
- "epoch": 0.026963262554769128,
343
- "grad_norm": 1.4375,
344
- "learning_rate": 0.0003,
345
- "loss": 3.95872802734375,
346
  "step": 800
347
  },
348
  {
349
- "epoch": 0.026963262554769128,
350
- "eval_loss": 3.9487836360931396,
351
- "eval_runtime": 8.7643,
352
- "eval_samples_per_second": 1087.026,
353
- "eval_steps_per_second": 0.799,
354
  "step": 800
355
  }
356
  ],
357
  "logging_steps": 20,
358
- "max_steps": 2500,
359
  "num_input_tokens_seen": 0,
360
  "num_train_epochs": 1,
361
  "save_steps": 100,
@@ -371,8 +371,8 @@
371
  "attributes": {}
372
  }
373
  },
374
- "total_flos": 116155377254400.0,
375
- "train_batch_size": 32,
376
  "trial_name": null,
377
  "trial_params": null
378
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.06740815638692282,
6
  "eval_steps": 100,
7
  "global_step": 800,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  },
184
  {
185
+ "epoch": 0.03538928210313448,
186
+ "grad_norm": 0.7109375,
187
+ "learning_rate": 0.001,
188
+ "loss": 3.3047794342041015,
189
  "step": 420
190
  },
191
  {
192
+ "epoch": 0.03707448601280755,
193
+ "grad_norm": 0.71875,
194
+ "learning_rate": 0.001,
195
+ "loss": 3.258700942993164,
196
  "step": 440
197
  },
198
  {
199
+ "epoch": 0.03875968992248062,
200
+ "grad_norm": 0.73046875,
201
+ "learning_rate": 0.001,
202
+ "loss": 3.229313278198242,
203
  "step": 460
204
  },
205
  {
206
+ "epoch": 0.04044489383215369,
207
+ "grad_norm": 0.703125,
208
+ "learning_rate": 0.001,
209
+ "loss": 3.202067565917969,
210
  "step": 480
211
  },
212
  {
213
+ "epoch": 0.04213009774182676,
214
+ "grad_norm": 0.7265625,
215
+ "learning_rate": 0.001,
216
+ "loss": 3.163772201538086,
217
  "step": 500
218
  },
219
  {
220
+ "epoch": 0.04213009774182676,
221
+ "eval_loss": 3.157355308532715,
222
+ "eval_runtime": 7.9503,
223
+ "eval_samples_per_second": 1198.315,
224
+ "eval_steps_per_second": 0.88,
225
  "step": 500
226
  },
227
  {
228
+ "epoch": 0.043815301651499834,
229
+ "grad_norm": 0.6953125,
230
+ "learning_rate": 0.001,
231
+ "loss": 3.1448848724365233,
232
  "step": 520
233
  },
234
  {
235
+ "epoch": 0.0455005055611729,
236
+ "grad_norm": 0.828125,
237
+ "learning_rate": 0.001,
238
+ "loss": 3.103527069091797,
239
  "step": 540
240
  },
241
  {
242
+ "epoch": 0.04718570947084597,
243
+ "grad_norm": 0.7578125,
244
+ "learning_rate": 0.001,
245
+ "loss": 3.08404541015625,
246
  "step": 560
247
  },
248
  {
249
+ "epoch": 0.04887091338051904,
250
+ "grad_norm": 0.76953125,
251
+ "learning_rate": 0.001,
252
+ "loss": 3.0501741409301757,
253
  "step": 580
254
  },
255
  {
256
+ "epoch": 0.05055611729019211,
257
+ "grad_norm": 0.96875,
258
+ "learning_rate": 0.001,
259
+ "loss": 3.0376760482788088,
260
  "step": 600
261
  },
262
  {
263
+ "epoch": 0.05055611729019211,
264
+ "eval_loss": 3.0310890674591064,
265
+ "eval_runtime": 8.1195,
266
+ "eval_samples_per_second": 1173.346,
267
+ "eval_steps_per_second": 0.862,
268
  "step": 600
269
  },
270
  {
271
+ "epoch": 0.05224132119986518,
272
+ "grad_norm": 0.82421875,
273
+ "learning_rate": 0.001,
274
+ "loss": 3.0314205169677733,
275
  "step": 620
276
  },
277
  {
278
+ "epoch": 0.053926525109538256,
279
+ "grad_norm": 0.640625,
280
+ "learning_rate": 0.001,
281
+ "loss": 3.0014928817749023,
282
  "step": 640
283
  },
284
  {
285
+ "epoch": 0.055611729019211326,
286
+ "grad_norm": 0.70703125,
287
+ "learning_rate": 0.001,
288
+ "loss": 2.9963293075561523,
289
  "step": 660
290
  },
291
  {
292
+ "epoch": 0.057296932928884395,
293
+ "grad_norm": 0.7109375,
294
+ "learning_rate": 0.001,
295
+ "loss": 2.9568761825561523,
296
  "step": 680
297
  },
298
  {
299
+ "epoch": 0.058982136838557464,
300
+ "grad_norm": 0.69921875,
301
+ "learning_rate": 0.001,
302
+ "loss": 2.9395275115966797,
303
  "step": 700
304
  },
305
  {
306
+ "epoch": 0.058982136838557464,
307
+ "eval_loss": 2.939429759979248,
308
+ "eval_runtime": 7.975,
309
+ "eval_samples_per_second": 1194.602,
310
+ "eval_steps_per_second": 0.878,
311
  "step": 700
312
  },
313
  {
314
+ "epoch": 0.06066734074823053,
315
+ "grad_norm": 0.6484375,
316
+ "learning_rate": 0.001,
317
+ "loss": 2.922671890258789,
318
  "step": 720
319
  },
320
  {
321
+ "epoch": 0.06235254465790361,
322
+ "grad_norm": 0.95703125,
323
+ "learning_rate": 0.001,
324
+ "loss": 2.9145633697509767,
325
  "step": 740
326
  },
327
  {
328
+ "epoch": 0.06403774856757667,
329
+ "grad_norm": 0.7109375,
330
+ "learning_rate": 0.001,
331
+ "loss": 2.8876668930053713,
332
  "step": 760
333
  },
334
  {
335
+ "epoch": 0.06572295247724974,
336
+ "grad_norm": 0.69140625,
337
+ "learning_rate": 0.001,
338
+ "loss": 2.8692466735839846,
339
  "step": 780
340
  },
341
  {
342
+ "epoch": 0.06740815638692282,
343
+ "grad_norm": 0.68359375,
344
+ "learning_rate": 0.001,
345
+ "loss": 2.8788633346557617,
346
  "step": 800
347
  },
348
  {
349
+ "epoch": 0.06740815638692282,
350
+ "eval_loss": 2.8632190227508545,
351
+ "eval_runtime": 8.2773,
352
+ "eval_samples_per_second": 1150.973,
353
+ "eval_steps_per_second": 0.846,
354
  "step": 800
355
  }
356
  ],
357
  "logging_steps": 20,
358
+ "max_steps": 1000,
359
  "num_input_tokens_seen": 0,
360
  "num_train_epochs": 1,
361
  "save_steps": 100,
 
371
  "attributes": {}
372
  }
373
  },
374
+ "total_flos": 290388443136000.0,
375
+ "train_batch_size": 80,
376
  "trial_name": null,
377
  "trial_params": null
378
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-800/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4137a389ac7c7428d4244efe52cec11b1ef7635b5e3ab39a01d3009f59c58e47
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82e06f7126a42b84896bee7f96709259393d400ad27e92239a7c07ad6c0e0fd4
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7c462ec1376bb41ba70524306b41158fb5e54895755f46744c0cd2a70fd91f9b
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:412b1a58a5c56f4c0d4c86023f862cac2da0249009c2d10b79e1506090879e57
3
  size 8068282
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:389b3459cdd290ee62a7be41d029a930e2587531b43c7e5ade84614c6ebc3507
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:819e7677327ec819a829273ee6bf375a38913eb519826c73009d0dd3825f5480
3
  size 1064
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.030333670374115267,
6
  "eval_steps": 100,
7
  "global_step": 900,
8
  "is_hyper_param_search": false,
@@ -10,395 +10,395 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.796875,
15
- "learning_rate": 1.14e-05,
16
- "loss": 8.322640228271485,
17
  "step": 20
18
  },
19
  {
20
- "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.75,
22
- "learning_rate": 2.34e-05,
23
- "loss": 8.267462158203125,
24
  "step": 40
25
  },
26
  {
27
- "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.3046875,
29
- "learning_rate": 3.539999999999999e-05,
30
- "loss": 8.106219482421874,
31
  "step": 60
32
  },
33
  {
34
- "epoch": 0.002696326255476913,
35
- "grad_norm": 1.28125,
36
- "learning_rate": 4.7399999999999993e-05,
37
- "loss": 7.948130798339844,
38
  "step": 80
39
  },
40
  {
41
- "epoch": 0.003370407819346141,
42
- "grad_norm": 1.25,
43
- "learning_rate": 5.94e-05,
44
- "loss": 7.777301788330078,
45
  "step": 100
46
  },
47
  {
48
- "epoch": 0.003370407819346141,
49
- "eval_loss": 7.679966926574707,
50
- "eval_runtime": 8.4901,
51
- "eval_samples_per_second": 1122.134,
52
- "eval_steps_per_second": 0.824,
53
  "step": 100
54
  },
55
  {
56
- "epoch": 0.004044489383215369,
57
- "grad_norm": 1.2578125,
58
- "learning_rate": 7.139999999999999e-05,
59
- "loss": 7.585383605957031,
60
  "step": 120
61
  },
62
  {
63
- "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.25,
65
- "learning_rate": 8.34e-05,
66
- "loss": 7.3637535095214846,
67
  "step": 140
68
  },
69
  {
70
- "epoch": 0.005392652510953826,
71
- "grad_norm": 1.25,
72
- "learning_rate": 9.539999999999999e-05,
73
- "loss": 7.1375579833984375,
74
  "step": 160
75
  },
76
  {
77
- "epoch": 0.006066734074823054,
78
- "grad_norm": 1.34375,
79
- "learning_rate": 0.00010739999999999998,
80
- "loss": 6.910031127929687,
81
  "step": 180
82
  },
83
  {
84
- "epoch": 0.006740815638692282,
85
- "grad_norm": 1.3359375,
86
- "learning_rate": 0.0001194,
87
- "loss": 6.671029663085937,
88
  "step": 200
89
  },
90
  {
91
- "epoch": 0.006740815638692282,
92
- "eval_loss": 6.548758506774902,
93
- "eval_runtime": 8.6601,
94
- "eval_samples_per_second": 1100.105,
95
- "eval_steps_per_second": 0.808,
96
  "step": 200
97
  },
98
  {
99
- "epoch": 0.00741489720256151,
100
- "grad_norm": 1.578125,
101
- "learning_rate": 0.0001314,
102
- "loss": 6.457804107666016,
103
  "step": 220
104
  },
105
  {
106
- "epoch": 0.008088978766430738,
107
- "grad_norm": 1.4921875,
108
- "learning_rate": 0.0001434,
109
- "loss": 6.254596328735351,
110
  "step": 240
111
  },
112
  {
113
- "epoch": 0.008763060330299966,
114
- "grad_norm": 1.171875,
115
- "learning_rate": 0.00015539999999999998,
116
- "loss": 6.047513961791992,
117
  "step": 260
118
  },
119
  {
120
- "epoch": 0.009437141894169195,
121
- "grad_norm": 1.8828125,
122
- "learning_rate": 0.0001674,
123
- "loss": 5.870236587524414,
124
  "step": 280
125
  },
126
  {
127
- "epoch": 0.010111223458038422,
128
- "grad_norm": 1.359375,
129
- "learning_rate": 0.00017939999999999997,
130
- "loss": 5.72708740234375,
131
  "step": 300
132
  },
133
  {
134
- "epoch": 0.010111223458038422,
135
- "eval_loss": 5.652859687805176,
136
- "eval_runtime": 8.4927,
137
- "eval_samples_per_second": 1121.794,
138
- "eval_steps_per_second": 0.824,
139
  "step": 300
140
  },
141
  {
142
- "epoch": 0.010785305021907651,
143
- "grad_norm": 1.609375,
144
- "learning_rate": 0.0001914,
145
- "loss": 5.585579681396484,
146
  "step": 320
147
  },
148
  {
149
- "epoch": 0.011459386585776879,
150
- "grad_norm": 1.671875,
151
- "learning_rate": 0.00020339999999999998,
152
- "loss": 5.470914077758789,
153
  "step": 340
154
  },
155
  {
156
- "epoch": 0.012133468149646108,
157
- "grad_norm": 1.4140625,
158
- "learning_rate": 0.00021539999999999998,
159
- "loss": 5.318555068969727,
160
  "step": 360
161
  },
162
  {
163
- "epoch": 0.012807549713515335,
164
- "grad_norm": 1.6015625,
165
- "learning_rate": 0.00022739999999999997,
166
- "loss": 5.185982894897461,
167
  "step": 380
168
  },
169
  {
170
- "epoch": 0.013481631277384564,
171
- "grad_norm": 1.2578125,
172
- "learning_rate": 0.0002394,
173
- "loss": 5.065654754638672,
174
  "step": 400
175
  },
176
  {
177
- "epoch": 0.013481631277384564,
178
- "eval_loss": 5.008047103881836,
179
- "eval_runtime": 8.5503,
180
- "eval_samples_per_second": 1114.232,
181
- "eval_steps_per_second": 0.819,
182
  "step": 400
183
  },
184
  {
185
- "epoch": 0.014155712841253791,
186
- "grad_norm": 2.140625,
187
- "learning_rate": 0.0002514,
188
- "loss": 4.935818481445312,
189
  "step": 420
190
  },
191
  {
192
- "epoch": 0.01482979440512302,
193
- "grad_norm": 1.1875,
194
- "learning_rate": 0.00026339999999999995,
195
- "loss": 4.853317642211914,
196
  "step": 440
197
  },
198
  {
199
- "epoch": 0.015503875968992248,
200
- "grad_norm": 1.5390625,
201
- "learning_rate": 0.00027539999999999997,
202
- "loss": 4.7492321014404295,
203
  "step": 460
204
  },
205
  {
206
- "epoch": 0.016177957532861477,
207
- "grad_norm": 1.921875,
208
- "learning_rate": 0.00028739999999999994,
209
- "loss": 4.629489898681641,
210
  "step": 480
211
  },
212
  {
213
- "epoch": 0.016852039096730706,
214
- "grad_norm": 1.7890625,
215
- "learning_rate": 0.00029939999999999996,
216
- "loss": 4.565845108032226,
217
  "step": 500
218
  },
219
  {
220
- "epoch": 0.016852039096730706,
221
- "eval_loss": 4.530772686004639,
222
- "eval_runtime": 8.4284,
223
- "eval_samples_per_second": 1130.345,
224
- "eval_steps_per_second": 0.831,
225
  "step": 500
226
  },
227
  {
228
- "epoch": 0.01752612066059993,
229
- "grad_norm": 1.140625,
230
- "learning_rate": 0.0003,
231
- "loss": 4.50256233215332,
232
  "step": 520
233
  },
234
  {
235
- "epoch": 0.01820020222446916,
236
- "grad_norm": 1.7265625,
237
- "learning_rate": 0.0003,
238
- "loss": 4.419484710693359,
239
  "step": 540
240
  },
241
  {
242
- "epoch": 0.01887428378833839,
243
- "grad_norm": 1.3125,
244
- "learning_rate": 0.0003,
245
- "loss": 4.363323593139649,
246
  "step": 560
247
  },
248
  {
249
- "epoch": 0.01954836535220762,
250
- "grad_norm": 1.5,
251
- "learning_rate": 0.0003,
252
- "loss": 4.3072765350341795,
253
  "step": 580
254
  },
255
  {
256
- "epoch": 0.020222446916076844,
257
- "grad_norm": 1.9765625,
258
- "learning_rate": 0.0003,
259
- "loss": 4.263522338867188,
260
  "step": 600
261
  },
262
  {
263
- "epoch": 0.020222446916076844,
264
- "eval_loss": 4.239859580993652,
265
- "eval_runtime": 8.5836,
266
- "eval_samples_per_second": 1109.908,
267
- "eval_steps_per_second": 0.816,
268
  "step": 600
269
  },
270
  {
271
- "epoch": 0.020896528479946073,
272
- "grad_norm": 1.2265625,
273
- "learning_rate": 0.0003,
274
- "loss": 4.207575988769531,
275
  "step": 620
276
  },
277
  {
278
- "epoch": 0.021570610043815303,
279
- "grad_norm": 1.3359375,
280
- "learning_rate": 0.0003,
281
- "loss": 4.179453659057617,
282
  "step": 640
283
  },
284
  {
285
- "epoch": 0.022244691607684528,
286
- "grad_norm": 1.09375,
287
- "learning_rate": 0.0003,
288
- "loss": 4.13963508605957,
289
  "step": 660
290
  },
291
  {
292
- "epoch": 0.022918773171553757,
293
- "grad_norm": 1.328125,
294
- "learning_rate": 0.0003,
295
- "loss": 4.0913646697998045,
296
  "step": 680
297
  },
298
  {
299
- "epoch": 0.023592854735422986,
300
- "grad_norm": 1.609375,
301
- "learning_rate": 0.0003,
302
- "loss": 4.082810974121093,
303
  "step": 700
304
  },
305
  {
306
- "epoch": 0.023592854735422986,
307
- "eval_loss": 4.061285495758057,
308
- "eval_runtime": 8.3862,
309
- "eval_samples_per_second": 1136.028,
310
- "eval_steps_per_second": 0.835,
311
  "step": 700
312
  },
313
  {
314
- "epoch": 0.024266936299292215,
315
- "grad_norm": 1.453125,
316
- "learning_rate": 0.0003,
317
- "loss": 4.052593612670899,
318
  "step": 720
319
  },
320
  {
321
- "epoch": 0.02494101786316144,
322
- "grad_norm": 1.46875,
323
- "learning_rate": 0.0003,
324
- "loss": 4.038698196411133,
325
  "step": 740
326
  },
327
  {
328
- "epoch": 0.02561509942703067,
329
- "grad_norm": 1.671875,
330
- "learning_rate": 0.0003,
331
- "loss": 3.99466552734375,
332
  "step": 760
333
  },
334
  {
335
- "epoch": 0.0262891809908999,
336
- "grad_norm": 1.3671875,
337
- "learning_rate": 0.0003,
338
- "loss": 3.9530941009521485,
339
  "step": 780
340
  },
341
  {
342
- "epoch": 0.026963262554769128,
343
- "grad_norm": 1.4375,
344
- "learning_rate": 0.0003,
345
- "loss": 3.95872802734375,
346
  "step": 800
347
  },
348
  {
349
- "epoch": 0.026963262554769128,
350
- "eval_loss": 3.9487836360931396,
351
- "eval_runtime": 8.7643,
352
- "eval_samples_per_second": 1087.026,
353
- "eval_steps_per_second": 0.799,
354
  "step": 800
355
  },
356
  {
357
- "epoch": 0.027637344118638354,
358
- "grad_norm": 1.2734375,
359
- "learning_rate": 0.0003,
360
- "loss": 3.917556381225586,
361
  "step": 820
362
  },
363
  {
364
- "epoch": 0.028311425682507583,
365
- "grad_norm": 1.390625,
366
- "learning_rate": 0.0003,
367
- "loss": 3.913285827636719,
368
  "step": 840
369
  },
370
  {
371
- "epoch": 0.028985507246376812,
372
- "grad_norm": 1.453125,
373
- "learning_rate": 0.0003,
374
- "loss": 3.9367324829101564,
375
  "step": 860
376
  },
377
  {
378
- "epoch": 0.02965958881024604,
379
- "grad_norm": 1.328125,
380
- "learning_rate": 0.0003,
381
- "loss": 3.8899993896484375,
382
  "step": 880
383
  },
384
  {
385
- "epoch": 0.030333670374115267,
386
- "grad_norm": 1.6875,
387
- "learning_rate": 0.0003,
388
- "loss": 3.8575370788574217,
389
  "step": 900
390
  },
391
  {
392
- "epoch": 0.030333670374115267,
393
- "eval_loss": 3.871619701385498,
394
- "eval_runtime": 8.7624,
395
- "eval_samples_per_second": 1087.263,
396
- "eval_steps_per_second": 0.799,
397
  "step": 900
398
  }
399
  ],
400
  "logging_steps": 20,
401
- "max_steps": 2500,
402
  "num_input_tokens_seen": 0,
403
  "num_train_epochs": 1,
404
  "save_steps": 100,
@@ -414,8 +414,8 @@
414
  "attributes": {}
415
  }
416
  },
417
- "total_flos": 130674799411200.0,
418
- "train_batch_size": 32,
419
  "trial_name": null,
420
  "trial_params": null
421
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.07583417593528817,
6
  "eval_steps": 100,
7
  "global_step": 900,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.0016852039096730705,
14
+ "grad_norm": 1.2421875,
15
+ "learning_rate": 9.5e-05,
16
+ "loss": 8.200721740722656,
17
  "step": 20
18
  },
19
  {
20
+ "epoch": 0.003370407819346141,
21
+ "grad_norm": 1.2421875,
22
+ "learning_rate": 0.00019500000000000002,
23
+ "loss": 7.764720153808594,
24
  "step": 40
25
  },
26
  {
27
+ "epoch": 0.005055611729019211,
28
+ "grad_norm": 1.2265625,
29
+ "learning_rate": 0.000295,
30
+ "loss": 7.160160064697266,
31
  "step": 60
32
  },
33
  {
34
+ "epoch": 0.006740815638692282,
35
+ "grad_norm": 1.2109375,
36
+ "learning_rate": 0.000395,
37
+ "loss": 6.480591583251953,
38
  "step": 80
39
  },
40
  {
41
+ "epoch": 0.008426019548365353,
42
+ "grad_norm": 1.0078125,
43
+ "learning_rate": 0.000495,
44
+ "loss": 5.919136428833008,
45
  "step": 100
46
  },
47
  {
48
+ "epoch": 0.008426019548365353,
49
+ "eval_loss": 5.634258270263672,
50
+ "eval_runtime": 7.97,
51
+ "eval_samples_per_second": 1195.356,
52
+ "eval_steps_per_second": 0.878,
53
  "step": 100
54
  },
55
  {
56
+ "epoch": 0.010111223458038422,
57
+ "grad_norm": 1.171875,
58
+ "learning_rate": 0.0005949999999999999,
59
+ "loss": 5.412916946411133,
60
  "step": 120
61
  },
62
  {
63
+ "epoch": 0.011796427367711493,
64
+ "grad_norm": 1.515625,
65
+ "learning_rate": 0.000695,
66
+ "loss": 5.0327880859375,
67
  "step": 140
68
  },
69
  {
70
+ "epoch": 0.013481631277384564,
71
+ "grad_norm": 0.9765625,
72
+ "learning_rate": 0.000795,
73
+ "loss": 4.69476089477539,
74
  "step": 160
75
  },
76
  {
77
+ "epoch": 0.015166835187057633,
78
+ "grad_norm": 0.8828125,
79
+ "learning_rate": 0.0008950000000000001,
80
+ "loss": 4.4246673583984375,
81
  "step": 180
82
  },
83
  {
84
+ "epoch": 0.016852039096730706,
85
+ "grad_norm": 0.87890625,
86
+ "learning_rate": 0.000995,
87
+ "loss": 4.204695129394532,
88
  "step": 200
89
  },
90
  {
91
+ "epoch": 0.016852039096730706,
92
+ "eval_loss": 4.133424282073975,
93
+ "eval_runtime": 7.9329,
94
+ "eval_samples_per_second": 1200.942,
95
+ "eval_steps_per_second": 0.882,
96
  "step": 200
97
  },
98
  {
99
+ "epoch": 0.018537243006403775,
100
+ "grad_norm": 0.6875,
101
+ "learning_rate": 0.001,
102
+ "loss": 4.048733520507812,
103
  "step": 220
104
  },
105
  {
106
+ "epoch": 0.020222446916076844,
107
+ "grad_norm": 0.73828125,
108
+ "learning_rate": 0.001,
109
+ "loss": 3.9190834045410154,
110
  "step": 240
111
  },
112
  {
113
+ "epoch": 0.021907650825749917,
114
+ "grad_norm": 0.68359375,
115
+ "learning_rate": 0.001,
116
+ "loss": 3.7833847045898437,
117
  "step": 260
118
  },
119
  {
120
+ "epoch": 0.023592854735422986,
121
+ "grad_norm": 0.82421875,
122
+ "learning_rate": 0.001,
123
+ "loss": 3.700960159301758,
124
  "step": 280
125
  },
126
  {
127
+ "epoch": 0.025278058645096056,
128
+ "grad_norm": 0.984375,
129
+ "learning_rate": 0.001,
130
+ "loss": 3.6359439849853517,
131
  "step": 300
132
  },
133
  {
134
+ "epoch": 0.025278058645096056,
135
+ "eval_loss": 3.5975606441497803,
136
+ "eval_runtime": 7.973,
137
+ "eval_samples_per_second": 1194.91,
138
+ "eval_steps_per_second": 0.878,
139
  "step": 300
140
  },
141
  {
142
+ "epoch": 0.026963262554769128,
143
+ "grad_norm": 0.80859375,
144
+ "learning_rate": 0.001,
145
+ "loss": 3.5393722534179686,
146
  "step": 320
147
  },
148
  {
149
+ "epoch": 0.028648466464442197,
150
+ "grad_norm": 0.72265625,
151
+ "learning_rate": 0.001,
152
+ "loss": 3.492219924926758,
153
  "step": 340
154
  },
155
  {
156
+ "epoch": 0.030333670374115267,
157
+ "grad_norm": 0.8203125,
158
+ "learning_rate": 0.001,
159
+ "loss": 3.4430694580078125,
160
  "step": 360
161
  },
162
  {
163
+ "epoch": 0.032018874283788336,
164
+ "grad_norm": 0.8203125,
165
+ "learning_rate": 0.001,
166
+ "loss": 3.3989883422851563,
167
  "step": 380
168
  },
169
  {
170
+ "epoch": 0.03370407819346141,
171
+ "grad_norm": 0.6484375,
172
+ "learning_rate": 0.001,
173
+ "loss": 3.341664123535156,
174
  "step": 400
175
  },
176
  {
177
+ "epoch": 0.03370407819346141,
178
+ "eval_loss": 3.3295202255249023,
179
+ "eval_runtime": 8.1687,
180
+ "eval_samples_per_second": 1166.287,
181
+ "eval_steps_per_second": 0.857,
182
  "step": 400
183
  },
184
  {
185
+ "epoch": 0.03538928210313448,
186
+ "grad_norm": 0.7109375,
187
+ "learning_rate": 0.001,
188
+ "loss": 3.3047794342041015,
189
  "step": 420
190
  },
191
  {
192
+ "epoch": 0.03707448601280755,
193
+ "grad_norm": 0.71875,
194
+ "learning_rate": 0.001,
195
+ "loss": 3.258700942993164,
196
  "step": 440
197
  },
198
  {
199
+ "epoch": 0.03875968992248062,
200
+ "grad_norm": 0.73046875,
201
+ "learning_rate": 0.001,
202
+ "loss": 3.229313278198242,
203
  "step": 460
204
  },
205
  {
206
+ "epoch": 0.04044489383215369,
207
+ "grad_norm": 0.703125,
208
+ "learning_rate": 0.001,
209
+ "loss": 3.202067565917969,
210
  "step": 480
211
  },
212
  {
213
+ "epoch": 0.04213009774182676,
214
+ "grad_norm": 0.7265625,
215
+ "learning_rate": 0.001,
216
+ "loss": 3.163772201538086,
217
  "step": 500
218
  },
219
  {
220
+ "epoch": 0.04213009774182676,
221
+ "eval_loss": 3.157355308532715,
222
+ "eval_runtime": 7.9503,
223
+ "eval_samples_per_second": 1198.315,
224
+ "eval_steps_per_second": 0.88,
225
  "step": 500
226
  },
227
  {
228
+ "epoch": 0.043815301651499834,
229
+ "grad_norm": 0.6953125,
230
+ "learning_rate": 0.001,
231
+ "loss": 3.1448848724365233,
232
  "step": 520
233
  },
234
  {
235
+ "epoch": 0.0455005055611729,
236
+ "grad_norm": 0.828125,
237
+ "learning_rate": 0.001,
238
+ "loss": 3.103527069091797,
239
  "step": 540
240
  },
241
  {
242
+ "epoch": 0.04718570947084597,
243
+ "grad_norm": 0.7578125,
244
+ "learning_rate": 0.001,
245
+ "loss": 3.08404541015625,
246
  "step": 560
247
  },
248
  {
249
+ "epoch": 0.04887091338051904,
250
+ "grad_norm": 0.76953125,
251
+ "learning_rate": 0.001,
252
+ "loss": 3.0501741409301757,
253
  "step": 580
254
  },
255
  {
256
+ "epoch": 0.05055611729019211,
257
+ "grad_norm": 0.96875,
258
+ "learning_rate": 0.001,
259
+ "loss": 3.0376760482788088,
260
  "step": 600
261
  },
262
  {
263
+ "epoch": 0.05055611729019211,
264
+ "eval_loss": 3.0310890674591064,
265
+ "eval_runtime": 8.1195,
266
+ "eval_samples_per_second": 1173.346,
267
+ "eval_steps_per_second": 0.862,
268
  "step": 600
269
  },
270
  {
271
+ "epoch": 0.05224132119986518,
272
+ "grad_norm": 0.82421875,
273
+ "learning_rate": 0.001,
274
+ "loss": 3.0314205169677733,
275
  "step": 620
276
  },
277
  {
278
+ "epoch": 0.053926525109538256,
279
+ "grad_norm": 0.640625,
280
+ "learning_rate": 0.001,
281
+ "loss": 3.0014928817749023,
282
  "step": 640
283
  },
284
  {
285
+ "epoch": 0.055611729019211326,
286
+ "grad_norm": 0.70703125,
287
+ "learning_rate": 0.001,
288
+ "loss": 2.9963293075561523,
289
  "step": 660
290
  },
291
  {
292
+ "epoch": 0.057296932928884395,
293
+ "grad_norm": 0.7109375,
294
+ "learning_rate": 0.001,
295
+ "loss": 2.9568761825561523,
296
  "step": 680
297
  },
298
  {
299
+ "epoch": 0.058982136838557464,
300
+ "grad_norm": 0.69921875,
301
+ "learning_rate": 0.001,
302
+ "loss": 2.9395275115966797,
303
  "step": 700
304
  },
305
  {
306
+ "epoch": 0.058982136838557464,
307
+ "eval_loss": 2.939429759979248,
308
+ "eval_runtime": 7.975,
309
+ "eval_samples_per_second": 1194.602,
310
+ "eval_steps_per_second": 0.878,
311
  "step": 700
312
  },
313
  {
314
+ "epoch": 0.06066734074823053,
315
+ "grad_norm": 0.6484375,
316
+ "learning_rate": 0.001,
317
+ "loss": 2.922671890258789,
318
  "step": 720
319
  },
320
  {
321
+ "epoch": 0.06235254465790361,
322
+ "grad_norm": 0.95703125,
323
+ "learning_rate": 0.001,
324
+ "loss": 2.9145633697509767,
325
  "step": 740
326
  },
327
  {
328
+ "epoch": 0.06403774856757667,
329
+ "grad_norm": 0.7109375,
330
+ "learning_rate": 0.001,
331
+ "loss": 2.8876668930053713,
332
  "step": 760
333
  },
334
  {
335
+ "epoch": 0.06572295247724974,
336
+ "grad_norm": 0.69140625,
337
+ "learning_rate": 0.001,
338
+ "loss": 2.8692466735839846,
339
  "step": 780
340
  },
341
  {
342
+ "epoch": 0.06740815638692282,
343
+ "grad_norm": 0.68359375,
344
+ "learning_rate": 0.001,
345
+ "loss": 2.8788633346557617,
346
  "step": 800
347
  },
348
  {
349
+ "epoch": 0.06740815638692282,
350
+ "eval_loss": 2.8632190227508545,
351
+ "eval_runtime": 8.2773,
352
+ "eval_samples_per_second": 1150.973,
353
+ "eval_steps_per_second": 0.846,
354
  "step": 800
355
  },
356
  {
357
+ "epoch": 0.0690933602965959,
358
+ "grad_norm": 0.6796875,
359
+ "learning_rate": 0.001,
360
+ "loss": 2.8496471405029298,
361
  "step": 820
362
  },
363
  {
364
+ "epoch": 0.07077856420626896,
365
+ "grad_norm": 0.62109375,
366
+ "learning_rate": 0.001,
367
+ "loss": 2.847636604309082,
368
  "step": 840
369
  },
370
  {
371
+ "epoch": 0.07246376811594203,
372
+ "grad_norm": 0.73046875,
373
+ "learning_rate": 0.001,
374
+ "loss": 2.8339916229248048,
375
  "step": 860
376
  },
377
  {
378
+ "epoch": 0.0741489720256151,
379
+ "grad_norm": 0.69921875,
380
+ "learning_rate": 0.001,
381
+ "loss": 2.8253028869628904,
382
  "step": 880
383
  },
384
  {
385
+ "epoch": 0.07583417593528817,
386
+ "grad_norm": 0.6875,
387
+ "learning_rate": 0.001,
388
+ "loss": 2.7853120803833007,
389
  "step": 900
390
  },
391
  {
392
+ "epoch": 0.07583417593528817,
393
+ "eval_loss": 2.8034849166870117,
394
+ "eval_runtime": 7.9593,
395
+ "eval_samples_per_second": 1196.967,
396
+ "eval_steps_per_second": 0.879,
397
  "step": 900
398
  }
399
  ],
400
  "logging_steps": 20,
401
+ "max_steps": 1000,
402
  "num_input_tokens_seen": 0,
403
  "num_train_epochs": 1,
404
  "save_steps": 100,
 
414
  "attributes": {}
415
  }
416
  },
417
+ "total_flos": 326686998528000.0,
418
+ "train_batch_size": 80,
419
  "trial_name": null,
420
  "trial_params": null
421
  }
zain/Activation/out/mlp-linear-9L_run/checkpoint-900/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9b2c45512c657af2390a8ad017958eac3003533c6b983fd882bc7a35e1a187ed
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e9c917aa3b25bdbc40f473e1cf017022bafceb91187441f0ef9ae86b2f5e2e6
3
  size 4010544
zain/Activation/out/mlp-linear-9L_run/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2150980b46702de80cf5b9c95700c86dedab0889248b4672dd0d83fc71b3aa0
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1168ea4bbfb8182db6bef374718cd5e9bd631ffa3eb5aaea5cc2742de1e3c4e5
3
  size 4920
zain/Activation/out/mlp-linear-9L_run/training_log.jsonl CHANGED
@@ -1,152 +1,62 @@
1
- {"step": 20, "epoch": 0.0006740815638692282, "timestamp": 1787090024.3510127, "loss": 8.322640228271485, "grad_norm": 1.796875, "learning_rate": 1.14e-05, "train/total_time_seconds": 0.4871659614145756, "train/time_per_step_avg": 0.024358298070728777, "train/epoch_time_elapsed": 1.0523655116558075, "train/estimated_remaining_minutes": 1.0068096535901228}
2
- {"step": 40, "epoch": 0.0013481631277384564, "timestamp": 1787090025.2673562, "loss": 8.267462158203125, "grad_norm": 1.75, "learning_rate": 2.34e-05, "train/total_time_seconds": 0.9049943648278713, "train/time_per_step_avg": 0.022624859120696783, "train/epoch_time_elapsed": 1.9687108546495438, "train/estimated_remaining_minutes": 0.9276192239485681}
3
- {"step": 60, "epoch": 0.0020222446916076846, "timestamp": 1787090026.191169, "loss": 8.106219482421874, "grad_norm": 1.3046875, "learning_rate": 3.539999999999999e-05, "train/total_time_seconds": 1.3297606445848942, "train/time_per_step_avg": 0.022162677409748237, "train/epoch_time_elapsed": 2.8925236389040947, "train/estimated_remaining_minutes": 0.901282214663095}
4
- {"step": 80, "epoch": 0.002696326255476913, "timestamp": 1787090027.185015, "loss": 7.948130798339844, "grad_norm": 1.28125, "learning_rate": 4.7399999999999993e-05, "train/total_time_seconds": 1.776447206735611, "train/time_per_step_avg": 0.022205590084195138, "train/epoch_time_elapsed": 3.8863690681755543, "train/estimated_remaining_minutes": 0.8956254667292038}
5
- {"step": 100, "epoch": 0.003370407819346141, "timestamp": 1787090028.1093025, "loss": 7.777301788330078, "grad_norm": 1.25, "learning_rate": 5.94e-05, "train/total_time_seconds": 2.2005495578050613, "train/time_per_step_avg": 0.022005495578050614, "train/epoch_time_elapsed": 4.810656946152449, "train/estimated_remaining_minutes": 0.8802198231220245}
6
- {"step": 100, "epoch": 0.003370407819346141, "timestamp": 1787090036.6009336, "eval_loss": 7.679966926574707, "eval_runtime": 8.4901, "eval_samples_per_second": 1122.134, "eval_steps_per_second": 0.824, "train/total_time_seconds": 2.2005495578050613, "train/time_per_step_avg": 0.022005495578050614, "train/epoch_time_elapsed": 13.302287008613348, "train/estimated_remaining_minutes": 0.8802198231220245}
7
- {"step": 120, "epoch": 0.004044489383215369, "timestamp": 1787090037.5545268, "loss": 7.585383605957031, "grad_norm": 1.2578125, "learning_rate": 7.139999999999999e-05, "train/total_time_seconds": 2.619401503354311, "train/time_per_step_avg": 0.021322355419397355, "train/epoch_time_elapsed": 14.255881257355213, "train/estimated_remaining_minutes": 0.8658577191643417}
8
- {"step": 140, "epoch": 0.0047185709470845974, "timestamp": 1787090038.4570768, "loss": 7.3637535095214846, "grad_norm": 1.25, "learning_rate": 8.34e-05, "train/total_time_seconds": 3.0328244753181934, "train/time_per_step_avg": 0.02127830110490322, "train/epoch_time_elapsed": 15.15843179449439, "train/estimated_remaining_minutes": 0.852079257351302}
9
- {"step": 160, "epoch": 0.005392652510953826, "timestamp": 1787090039.3583179, "loss": 7.1375579833984375, "grad_norm": 1.25, "learning_rate": 9.539999999999999e-05, "train/total_time_seconds": 3.4454963505268097, "train/time_per_step_avg": 0.021157357059419155, "train/epoch_time_elapsed": 16.059672601521015, "train/estimated_remaining_minutes": 0.8398397354409098}
10
- {"step": 180, "epoch": 0.006066734074823054, "timestamp": 1787090040.2629287, "loss": 6.910031127929687, "grad_norm": 1.34375, "learning_rate": 0.00010739999999999998, "train/total_time_seconds": 3.861355599015951, "train/time_per_step_avg": 0.0208490839228034, "train/epoch_time_elapsed": 16.964283030480146, "train/estimated_remaining_minutes": 0.8294763879367599}
11
- {"step": 200, "epoch": 0.006740815638692282, "timestamp": 1787090041.1701689, "loss": 6.671029663085937, "grad_norm": 1.3359375, "learning_rate": 0.0001194, "train/total_time_seconds": 4.279542360454798, "train/time_per_step_avg": 0.020789928026497363, "train/epoch_time_elapsed": 17.871523462235928, "train/estimated_remaining_minutes": 0.8202456190871695}
12
- {"step": 200, "epoch": 0.006740815638692282, "timestamp": 1787090049.8318973, "eval_loss": 6.548758506774902, "eval_runtime": 8.6601, "eval_samples_per_second": 1100.105, "eval_steps_per_second": 0.808, "train/total_time_seconds": 4.279542360454798, "train/time_per_step_avg": 0.020789928026497363, "train/epoch_time_elapsed": 26.53324942290783, "train/estimated_remaining_minutes": 0.8202456190871695}
13
- {"step": 220, "epoch": 0.00741489720256151, "timestamp": 1787090050.7864394, "loss": 6.457804107666016, "grad_norm": 1.578125, "learning_rate": 0.0001314, "train/total_time_seconds": 4.701243340969086, "train/time_per_step_avg": 0.020818418376147747, "train/epoch_time_elapsed": 27.48779420554638, "train/estimated_remaining_minutes": 0.812032940712842}
14
- {"step": 240, "epoch": 0.008088978766430738, "timestamp": 1787090051.7082744, "loss": 6.254596328735351, "grad_norm": 1.4921875, "learning_rate": 0.0001434, "train/total_time_seconds": 5.123800829052925, "train/time_per_step_avg": 0.020909763537347317, "train/epoch_time_elapsed": 28.40962889045477, "train/estimated_remaining_minutes": 0.8041520745596952}
15
- {"step": 260, "epoch": 0.008763060330299966, "timestamp": 1787090052.6360495, "loss": 6.047513961791992, "grad_norm": 1.171875, "learning_rate": 0.00015539999999999998, "train/total_time_seconds": 5.546675357967615, "train/time_per_step_avg": 0.021011790074408055, "train/epoch_time_elapsed": 29.337404031306505, "train/estimated_remaining_minutes": 0.7964456924261191}
16
- {"step": 280, "epoch": 0.009437141894169195, "timestamp": 1787090053.5612154, "loss": 5.870236587524414, "grad_norm": 1.8828125, "learning_rate": 0.0001674, "train/total_time_seconds": 5.969429299235344, "train/time_per_step_avg": 0.021080737002193928, "train/epoch_time_elapsed": 30.262570057064295, "train/estimated_remaining_minutes": 0.7888174431132419}
17
- {"step": 300, "epoch": 0.010111223458038422, "timestamp": 1787090054.5063047, "loss": 5.72708740234375, "grad_norm": 1.359375, "learning_rate": 0.00017939999999999997, "train/total_time_seconds": 6.396400835365057, "train/time_per_step_avg": 0.021168584749102593, "train/epoch_time_elapsed": 31.207659542560577, "train/estimated_remaining_minutes": 0.7817823243223958}
18
- {"step": 300, "epoch": 0.010111223458038422, "timestamp": 1787090063.000556, "eval_loss": 5.652859687805176, "eval_runtime": 8.4927, "eval_samples_per_second": 1121.794, "eval_steps_per_second": 0.824, "train/total_time_seconds": 6.396400835365057, "train/time_per_step_avg": 0.021168584749102593, "train/epoch_time_elapsed": 39.70190812647343, "train/estimated_remaining_minutes": 0.7817823243223958}
19
- {"step": 320, "epoch": 0.010785305021907651, "timestamp": 1787090064.0466053, "loss": 5.585579681396484, "grad_norm": 1.609375, "learning_rate": 0.0001914, "train/total_time_seconds": 6.901587720960379, "train/time_per_step_avg": 0.02200344379991293, "train/epoch_time_elapsed": 40.74796038493514, "train/estimated_remaining_minutes": 0.783617772484043}
20
- {"step": 340, "epoch": 0.011459386585776879, "timestamp": 1787090065.0750475, "loss": 5.470914077758789, "grad_norm": 1.671875, "learning_rate": 0.00020339999999999998, "train/total_time_seconds": 7.3942382000386715, "train/time_per_step_avg": 0.022704373709857464, "train/epoch_time_elapsed": 41.776401951909065, "train/estimated_remaining_minutes": 0.782919338827624}
21
- {"step": 360, "epoch": 0.012133468149646108, "timestamp": 1787090066.010938, "loss": 5.318555068969727, "grad_norm": 1.4140625, "learning_rate": 0.00021539999999999998, "train/total_time_seconds": 7.831252176314592, "train/time_per_step_avg": 0.022845768183469773, "train/epoch_time_elapsed": 42.71229239925742, "train/estimated_remaining_minutes": 0.7758740582089457}
22
- {"step": 380, "epoch": 0.012807549713515335, "timestamp": 1787090066.942198, "loss": 5.185982894897461, "grad_norm": 1.6015625, "learning_rate": 0.00022739999999999997, "train/total_time_seconds": 8.253654949367046, "train/time_per_step_avg": 0.022842256501317024, "train/epoch_time_elapsed": 43.643552385270596, "train/estimated_remaining_minutes": 0.7674451093271114}
23
- {"step": 400, "epoch": 0.013481631277384564, "timestamp": 1787090067.867534, "loss": 5.065654754638672, "grad_norm": 1.2578125, "learning_rate": 0.0002394, "train/total_time_seconds": 8.672141395509243, "train/time_per_step_avg": 0.02275740560144186, "train/epoch_time_elapsed": 44.56888834387064, "train/estimated_remaining_minutes": 0.7588123721070588}
24
- {"step": 400, "epoch": 0.013481631277384564, "timestamp": 1787090076.4197314, "eval_loss": 5.008047103881836, "eval_runtime": 8.5503, "eval_samples_per_second": 1114.232, "eval_steps_per_second": 0.819, "train/total_time_seconds": 8.672141395509243, "train/time_per_step_avg": 0.02275740560144186, "train/epoch_time_elapsed": 53.12108425050974, "train/estimated_remaining_minutes": 0.7588123721070588}
25
- {"step": 420, "epoch": 0.014155712841253791, "timestamp": 1787090077.4167783, "loss": 4.935818481445312, "grad_norm": 2.140625, "learning_rate": 0.0002514, "train/total_time_seconds": 9.099947553128004, "train/time_per_step_avg": 0.021983598321676255, "train/epoch_time_elapsed": 54.11813282221556, "train/estimated_remaining_minutes": 0.7511067821629464}
26
- {"step": 440, "epoch": 0.01482979440512302, "timestamp": 1787090078.3350494, "loss": 4.853317642211914, "grad_norm": 1.1875, "learning_rate": 0.00026339999999999995, "train/total_time_seconds": 9.511507794260979, "train/time_per_step_avg": 0.021172695942223072, "train/epoch_time_elapsed": 55.03640407696366, "train/estimated_remaining_minutes": 0.7421858354612733}
27
- {"step": 460, "epoch": 0.015503875968992248, "timestamp": 1787090079.2550533, "loss": 4.7492321014404295, "grad_norm": 1.5390625, "learning_rate": 0.00027539999999999997, "train/total_time_seconds": 9.924589660018682, "train/time_per_step_avg": 0.0209333748370409, "train/epoch_time_elapsed": 55.95640867948532, "train/estimated_remaining_minutes": 0.7335566270448591}
28
- {"step": 480, "epoch": 0.016177957532861477, "timestamp": 1787090080.1738203, "loss": 4.629489898681641, "grad_norm": 1.921875, "learning_rate": 0.00028739999999999994, "train/total_time_seconds": 10.338026851415634, "train/time_per_step_avg": 0.020843719020485877, "train/epoch_time_elapsed": 56.8751747533679, "train/estimated_remaining_minutes": 0.725097716661791}
29
- {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1787090081.1055214, "loss": 4.565845108032226, "grad_norm": 1.7890625, "learning_rate": 0.00029939999999999996, "train/total_time_seconds": 10.760399051010609, "train/time_per_step_avg": 0.020882576555013656, "train/epoch_time_elapsed": 57.80687604099512, "train/estimated_remaining_minutes": 0.7173599367340405}
30
- {"step": 500, "epoch": 0.016852039096730706, "timestamp": 1787090089.5357459, "eval_loss": 4.530772686004639, "eval_runtime": 8.4284, "eval_samples_per_second": 1130.345, "eval_steps_per_second": 0.831, "train/total_time_seconds": 10.760399051010609, "train/time_per_step_avg": 0.020882576555013656, "train/epoch_time_elapsed": 66.23709952831268, "train/estimated_remaining_minutes": 0.7173599367340405}
31
- {"step": 520, "epoch": 0.01752612066059993, "timestamp": 1787090090.4781623, "loss": 4.50256233215332, "grad_norm": 1.140625, "learning_rate": 0.0003, "train/total_time_seconds": 11.175844598561525, "train/time_per_step_avg": 0.020758970454335213, "train/epoch_time_elapsed": 67.17951717972755, "train/estimated_remaining_minutes": 0.709236291831789}
32
- {"step": 540, "epoch": 0.01820020222446916, "timestamp": 1787090091.3960457, "loss": 4.419484710693359, "grad_norm": 1.7265625, "learning_rate": 0.0003, "train/total_time_seconds": 11.595506254583597, "train/time_per_step_avg": 0.020839984603226183, "train/epoch_time_elapsed": 68.09740042313933, "train/estimated_remaining_minutes": 0.7014565512032054}
33
- {"step": 560, "epoch": 0.01887428378833839, "timestamp": 1787090092.3256834, "loss": 4.363323593139649, "grad_norm": 1.3125, "learning_rate": 0.0003, "train/total_time_seconds": 12.02891132235527, "train/time_per_step_avg": 0.02104321662336588, "train/epoch_time_elapsed": 69.02703780308366, "train/estimated_remaining_minutes": 0.6945264275407507}
34
- {"step": 580, "epoch": 0.01954836535220762, "timestamp": 1787090093.2448857, "loss": 4.3072765350341795, "grad_norm": 1.5, "learning_rate": 0.0003, "train/total_time_seconds": 12.451076030731201, "train/time_per_step_avg": 0.02113049179315567, "train/epoch_time_elapsed": 69.94624043628573, "train/estimated_remaining_minutes": 0.6869559189368939}
35
- {"step": 600, "epoch": 0.020222446916076844, "timestamp": 1787090094.1727655, "loss": 4.263522338867188, "grad_norm": 1.9765625, "learning_rate": 0.0003, "train/total_time_seconds": 12.882123444229364, "train/time_per_step_avg": 0.021217243932187557, "train/epoch_time_elapsed": 70.87412008270621, "train/estimated_remaining_minutes": 0.6798898484454388}
36
- {"step": 600, "epoch": 0.020222446916076844, "timestamp": 1787090102.7580671, "eval_loss": 4.239859580993652, "eval_runtime": 8.5836, "eval_samples_per_second": 1109.908, "eval_steps_per_second": 0.816, "train/total_time_seconds": 12.882123444229364, "train/time_per_step_avg": 0.021217243932187557, "train/epoch_time_elapsed": 79.45942096784711, "train/estimated_remaining_minutes": 0.6798898484454388}
37
- {"step": 620, "epoch": 0.020896528479946073, "timestamp": 1787090103.7138457, "loss": 4.207575988769531, "grad_norm": 1.2265625, "learning_rate": 0.0003, "train/total_time_seconds": 13.297651916742325, "train/time_per_step_avg": 0.021218073181807996, "train/epoch_time_elapsed": 80.41520067676902, "train/estimated_remaining_minutes": 0.6720318710611711}
38
- {"step": 640, "epoch": 0.021570610043815303, "timestamp": 1787090104.6780584, "loss": 4.179453659057617, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 13.743419598788023, "train/time_per_step_avg": 0.02147913344204426, "train/epoch_time_elapsed": 81.37941356003284, "train/estimated_remaining_minutes": 0.6656968868162949}
39
- {"step": 660, "epoch": 0.022244691607684528, "timestamp": 1787090105.6800876, "loss": 4.13963508605957, "grad_norm": 1.09375, "learning_rate": 0.0003, "train/total_time_seconds": 14.178391981869936, "train/time_per_step_avg": 0.021494806595146656, "train/epoch_time_elapsed": 82.38144203275442, "train/estimated_remaining_minutes": 0.6587939708747647}
40
- {"step": 680, "epoch": 0.022918773171553757, "timestamp": 1787090106.6163764, "loss": 4.0913646697998045, "grad_norm": 1.328125, "learning_rate": 0.0003, "train/total_time_seconds": 14.603615425527096, "train/time_per_step_avg": 0.021525393947958948, "train/epoch_time_elapsed": 83.31773166358471, "train/estimated_remaining_minutes": 0.6514357861387087}
41
- {"step": 700, "epoch": 0.023592854735422986, "timestamp": 1787090107.6156392, "loss": 4.082810974121093, "grad_norm": 1.609375, "learning_rate": 0.0003, "train/total_time_seconds": 15.048711270093918, "train/time_per_step_avg": 0.021665878258645535, "train/epoch_time_elapsed": 84.31699389219284, "train/estimated_remaining_minutes": 0.6449447687183107}
42
- {"step": 700, "epoch": 0.023592854735422986, "timestamp": 1787090116.0037506, "eval_loss": 4.061285495758057, "eval_runtime": 8.3862, "eval_samples_per_second": 1136.028, "eval_steps_per_second": 0.835, "train/total_time_seconds": 15.048711270093918, "train/time_per_step_avg": 0.021665878258645535, "train/epoch_time_elapsed": 92.7051047757268, "train/estimated_remaining_minutes": 0.6449447687183107}
43
- {"step": 720, "epoch": 0.024266936299292215, "timestamp": 1787090116.9622593, "loss": 4.052593612670899, "grad_norm": 1.453125, "learning_rate": 0.0003, "train/total_time_seconds": 15.459270983934402, "train/time_per_step_avg": 0.021616190671920776, "train/epoch_time_elapsed": 93.66361425817013, "train/estimated_remaining_minutes": 0.6369792210972971}
44
- {"step": 740, "epoch": 0.02494101786316144, "timestamp": 1787090117.886146, "loss": 4.038698196411133, "grad_norm": 1.46875, "learning_rate": 0.0003, "train/total_time_seconds": 15.874074883759022, "train/time_per_step_avg": 0.02130655284970999, "train/epoch_time_elapsed": 94.5875009149313, "train/estimated_remaining_minutes": 0.6292426080048621}
45
- {"step": 760, "epoch": 0.02561509942703067, "timestamp": 1787090118.8126843, "loss": 3.99466552734375, "grad_norm": 1.671875, "learning_rate": 0.0003, "train/total_time_seconds": 16.288574557751417, "train/time_per_step_avg": 0.02110182575881481, "train/epoch_time_elapsed": 95.51403892412782, "train/estimated_remaining_minutes": 0.6215377133878831}
46
- {"step": 780, "epoch": 0.0262891809908999, "timestamp": 1787090119.7369952, "loss": 3.9530941009521485, "grad_norm": 1.3671875, "learning_rate": 0.0003, "train/total_time_seconds": 16.70217404142022, "train/time_per_step_avg": 0.020985586158931254, "train/epoch_time_elapsed": 96.43835016340017, "train/estimated_remaining_minutes": 0.6138405844282645}
47
- {"step": 800, "epoch": 0.026963262554769128, "timestamp": 1787090120.6631215, "loss": 3.95872802734375, "grad_norm": 1.4375, "learning_rate": 0.0003, "train/total_time_seconds": 17.1165285743773, "train/time_per_step_avg": 0.020678173042833804, "train/epoch_time_elapsed": 97.36447604000568, "train/estimated_remaining_minutes": 0.6062103870091959}
48
- {"step": 800, "epoch": 0.026963262554769128, "timestamp": 1787090129.4290054, "eval_loss": 3.9487836360931396, "eval_runtime": 8.7643, "eval_samples_per_second": 1087.026, "eval_steps_per_second": 0.799, "train/total_time_seconds": 17.1165285743773, "train/time_per_step_avg": 0.020678173042833804, "train/epoch_time_elapsed": 106.13035868108273, "train/estimated_remaining_minutes": 0.6062103870091959}
49
- {"step": 820, "epoch": 0.027637344118638354, "timestamp": 1787090130.3672554, "loss": 3.917556381225586, "grad_norm": 1.2734375, "learning_rate": 0.0003, "train/total_time_seconds": 17.527744065970182, "train/time_per_step_avg": 0.0206847308203578, "train/epoch_time_elapsed": 107.06861016899347, "train/estimated_remaining_minutes": 0.5985083339599574}
50
- {"step": 840, "epoch": 0.028311425682507583, "timestamp": 1787090131.2712123, "loss": 3.913285827636719, "grad_norm": 1.390625, "learning_rate": 0.0003, "train/total_time_seconds": 17.939896412193775, "train/time_per_step_avg": 0.020658215284347536, "train/epoch_time_elapsed": 107.97256690636277, "train/estimated_remaining_minutes": 0.5908775405603506}
51
- {"step": 860, "epoch": 0.028985507246376812, "timestamp": 1787090132.1687262, "loss": 3.9367324829101564, "grad_norm": 1.453125, "learning_rate": 0.0003, "train/total_time_seconds": 18.34703105315566, "train/time_per_step_avg": 0.020584564954042435, "train/epoch_time_elapsed": 108.87008076161146, "train/estimated_remaining_minutes": 0.5831226923871179}
52
- {"step": 880, "epoch": 0.02965958881024604, "timestamp": 1787090133.0716784, "loss": 3.8899993896484375, "grad_norm": 1.328125, "learning_rate": 0.0003, "train/total_time_seconds": 18.755650278180838, "train/time_per_step_avg": 0.020534762367606162, "train/epoch_time_elapsed": 109.77303267642856, "train/estimated_remaining_minutes": 0.5754574517169121}
53
- {"step": 900, "epoch": 0.030333670374115267, "timestamp": 1787090133.972401, "loss": 3.8575370788574217, "grad_norm": 1.6875, "learning_rate": 0.0003, "train/total_time_seconds": 19.16444643959403, "train/time_per_step_avg": 0.02047917865216732, "train/epoch_time_elapsed": 110.67375592142344, "train/estimated_remaining_minutes": 0.5678354500620454}
54
- {"step": 900, "epoch": 0.030333670374115267, "timestamp": 1787090142.736271, "eval_loss": 3.871619701385498, "eval_runtime": 8.7624, "eval_samples_per_second": 1087.263, "eval_steps_per_second": 0.799, "train/total_time_seconds": 19.16444643959403, "train/time_per_step_avg": 0.02047917865216732, "train/epoch_time_elapsed": 119.43762295693159, "train/estimated_remaining_minutes": 0.5678354500620454}
55
- {"step": 920, "epoch": 0.031007751937984496, "timestamp": 1787090143.6765184, "loss": 3.870378112792969, "grad_norm": 1.1875, "learning_rate": 0.0003, "train/total_time_seconds": 19.576474107801914, "train/time_per_step_avg": 0.020487300418317318, "train/epoch_time_elapsed": 120.37787260860205, "train/estimated_remaining_minutes": 0.5603411067088229}
56
- {"step": 940, "epoch": 0.031681833501853725, "timestamp": 1787090144.648713, "loss": 3.852782440185547, "grad_norm": 1.234375, "learning_rate": 0.0003, "train/total_time_seconds": 20.05193056166172, "train/time_per_step_avg": 0.02112034149467945, "train/epoch_time_elapsed": 121.3500671684742, "train/estimated_remaining_minutes": 0.554627866599154}
57
- {"step": 960, "epoch": 0.032355915065722954, "timestamp": 1787090145.5583317, "loss": 3.821004867553711, "grad_norm": 1.515625, "learning_rate": 0.0003, "train/total_time_seconds": 20.465665958821774, "train/time_per_step_avg": 0.021186349056661127, "train/epoch_time_elapsed": 122.25968647748232, "train/estimated_remaining_minutes": 0.5471723190379432}
58
- {"step": 980, "epoch": 0.03302999662959218, "timestamp": 1787090146.4644017, "loss": 3.8098331451416017, "grad_norm": 1.4921875, "learning_rate": 0.0003, "train/total_time_seconds": 20.878857355564833, "train/time_per_step_avg": 0.02123207077383995, "train/epoch_time_elapsed": 123.16575583443046, "train/estimated_remaining_minutes": 0.5397255642935126}
59
- {"step": 1000, "epoch": 0.03370407819346141, "timestamp": 1787090147.4082043, "loss": 3.806192398071289, "grad_norm": 1.4453125, "learning_rate": 0.0003, "train/total_time_seconds": 21.309228148311377, "train/time_per_step_avg": 0.02144781708717346, "train/epoch_time_elapsed": 124.10955806076527, "train/estimated_remaining_minutes": 0.5327307037077844}
60
- {"step": 1000, "epoch": 0.03370407819346141, "timestamp": 1787090155.9603746, "eval_loss": 3.805492401123047, "eval_runtime": 8.55, "eval_samples_per_second": 1114.265, "eval_steps_per_second": 0.819, "train/total_time_seconds": 21.309228148311377, "train/time_per_step_avg": 0.02144781708717346, "train/epoch_time_elapsed": 132.6617270372808, "train/estimated_remaining_minutes": 0.5327307037077844}
61
- {"step": 1020, "epoch": 0.034378159757330634, "timestamp": 1787090156.9198642, "loss": 3.7934722900390625, "grad_norm": 1.375, "learning_rate": 0.0003, "train/total_time_seconds": 21.73783740401268, "train/time_per_step_avg": 0.021613632962107658, "train/epoch_time_elapsed": 133.6212188899517, "train/estimated_remaining_minutes": 0.5256862640186073}
62
- {"step": 1040, "epoch": 0.03505224132119986, "timestamp": 1787090157.882276, "loss": 3.773797607421875, "grad_norm": 1.3515625, "learning_rate": 0.0003, "train/total_time_seconds": 22.17101999744773, "train/time_per_step_avg": 0.021190894357860087, "train/epoch_time_elapsed": 134.58363071084023, "train/estimated_remaining_minutes": 0.5187450191710526}
63
- {"step": 1060, "epoch": 0.03572632288506909, "timestamp": 1787090158.801586, "loss": 3.7539413452148436, "grad_norm": 1.734375, "learning_rate": 0.0003, "train/total_time_seconds": 22.594436164945364, "train/time_per_step_avg": 0.021287702061235904, "train/epoch_time_elapsed": 135.50294019281864, "train/estimated_remaining_minutes": 0.5115721395836687}
64
- {"step": 1080, "epoch": 0.03640040444893832, "timestamp": 1787090159.7369173, "loss": 3.742329406738281, "grad_norm": 2.109375, "learning_rate": 0.0003, "train/total_time_seconds": 23.02944030240178, "train/time_per_step_avg": 0.021505829468369483, "train/epoch_time_elapsed": 136.4382721632719, "train/estimated_remaining_minutes": 0.5046574881081871}
65
- {"step": 1100, "epoch": 0.03707448601280755, "timestamp": 1787090160.6459656, "loss": 3.737290954589844, "grad_norm": 1.84375, "learning_rate": 0.0003, "train/total_time_seconds": 23.445339139550924, "train/time_per_step_avg": 0.02136110991239548, "train/epoch_time_elapsed": 137.34732024744153, "train/estimated_remaining_minutes": 0.49732537568744384}
66
- {"step": 1100, "epoch": 0.03707448601280755, "timestamp": 1787090169.0898743, "eval_loss": 3.744091272354126, "eval_runtime": 8.4423, "eval_samples_per_second": 1128.49, "eval_steps_per_second": 0.829, "train/total_time_seconds": 23.445339139550924, "train/time_per_step_avg": 0.02136110991239548, "train/epoch_time_elapsed": 145.79122596606612, "train/estimated_remaining_minutes": 0.49732537568744384}
67
- {"step": 1120, "epoch": 0.03774856757667678, "timestamp": 1787090170.1453664, "loss": 3.7377304077148437, "grad_norm": 1.2890625, "learning_rate": 0.0003, "train/total_time_seconds": 23.914364136755466, "train/time_per_step_avg": 0.021765267327427864, "train/epoch_time_elapsed": 146.84672116488218, "train/estimated_remaining_minutes": 0.49109854923694257}
68
- {"step": 1140, "epoch": 0.03842264914054601, "timestamp": 1787090171.1464329, "loss": 3.713370513916016, "grad_norm": 1.9140625, "learning_rate": 0.0003, "train/total_time_seconds": 24.421343587338924, "train/time_per_step_avg": 0.02250323589891195, "train/epoch_time_elapsed": 147.8477869592607, "train/estimated_remaining_minutes": 0.4855705742511833}
69
- {"step": 1160, "epoch": 0.03909673070441524, "timestamp": 1787090172.061697, "loss": 3.7126068115234374, "grad_norm": 1.703125, "learning_rate": 0.0003, "train/total_time_seconds": 24.84479048475623, "train/time_per_step_avg": 0.022503543198108673, "train/epoch_time_elapsed": 148.76305164396763, "train/estimated_remaining_minutes": 0.4783336099076631}
70
- {"step": 1180, "epoch": 0.03977081226828446, "timestamp": 1787090172.96644, "loss": 3.7255340576171876, "grad_norm": 1.3515625, "learning_rate": 0.0003, "train/total_time_seconds": 25.258656304329634, "train/time_per_step_avg": 0.022292160019278525, "train/epoch_time_elapsed": 149.66779493540525, "train/estimated_remaining_minutes": 0.47092410058919654}
71
- {"step": 1200, "epoch": 0.04044489383215369, "timestamp": 1787090173.8752575, "loss": 3.6966705322265625, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 25.67636000365019, "train/time_per_step_avg": 0.022310208640992642, "train/epoch_time_elapsed": 150.57661206647754, "train/estimated_remaining_minutes": 0.4636009445103506}
72
- {"step": 1200, "epoch": 0.04044489383215369, "timestamp": 1787090182.3466508, "eval_loss": 3.6958041191101074, "eval_runtime": 8.4698, "eval_samples_per_second": 1124.816, "eval_steps_per_second": 0.826, "train/total_time_seconds": 25.67636000365019, "train/time_per_step_avg": 0.022310208640992642, "train/epoch_time_elapsed": 159.04800394922495, "train/estimated_remaining_minutes": 0.4636009445103506}
73
- {"step": 1220, "epoch": 0.04111897539602292, "timestamp": 1787090183.3028944, "loss": 3.6724082946777346, "grad_norm": 1.6171875, "learning_rate": 0.0003, "train/total_time_seconds": 26.09580424427986, "train/time_per_step_avg": 0.02181440107524395, "train/epoch_time_elapsed": 160.0042488835752, "train/estimated_remaining_minutes": 0.45632007421691567}
74
- {"step": 1240, "epoch": 0.04179305695989215, "timestamp": 1787090184.2335675, "loss": 3.6844207763671877, "grad_norm": 1.5234375, "learning_rate": 0.0003, "train/total_time_seconds": 26.528886377811432, "train/time_per_step_avg": 0.021075427904725073, "train/epoch_time_elapsed": 160.9349226243794, "train/estimated_remaining_minutes": 0.44927952736616134}
75
- {"step": 1260, "epoch": 0.042467138523761376, "timestamp": 1787090185.199458, "loss": 3.6926769256591796, "grad_norm": 1.375, "learning_rate": 0.0003, "train/total_time_seconds": 26.96605222299695, "train/time_per_step_avg": 0.02121261738240719, "train/epoch_time_elapsed": 161.90081299096346, "train/estimated_remaining_minutes": 0.4423003274671457}
76
- {"step": 1280, "epoch": 0.043141220087630605, "timestamp": 1787090186.1345897, "loss": 3.6725536346435548, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 27.402600783854723, "train/time_per_step_avg": 0.02143944479525089, "train/epoch_time_elapsed": 162.83594417572021, "train/estimated_remaining_minutes": 0.4353017312018589}
77
- {"step": 1300, "epoch": 0.043815301651499834, "timestamp": 1787090187.0789473, "loss": 3.6652542114257813, "grad_norm": 1.4921875, "learning_rate": 0.0003, "train/total_time_seconds": 27.839303817600012, "train/time_per_step_avg": 0.021629438139498233, "train/epoch_time_elapsed": 163.78030264377594, "train/estimated_remaining_minutes": 0.4282969818092309}
78
- {"step": 1300, "epoch": 0.043815301651499834, "timestamp": 1787090195.647478, "eval_loss": 3.6601669788360596, "eval_runtime": 8.567, "eval_samples_per_second": 1112.052, "eval_steps_per_second": 0.817, "train/total_time_seconds": 27.839303817600012, "train/time_per_step_avg": 0.021629438139498233, "train/epoch_time_elapsed": 172.34883080422878, "train/estimated_remaining_minutes": 0.4282969818092309}
79
- {"step": 1320, "epoch": 0.044489383215369056, "timestamp": 1787090196.6083207, "loss": 3.6304325103759765, "grad_norm": 1.203125, "learning_rate": 0.0003, "train/total_time_seconds": 28.265503082424402, "train/time_per_step_avg": 0.021696988381445407, "train/epoch_time_elapsed": 173.3096746765077, "train/estimated_remaining_minutes": 0.421127444914909}
80
- {"step": 1340, "epoch": 0.045163464779238285, "timestamp": 1787090197.5520518, "loss": 3.654827880859375, "grad_norm": 1.328125, "learning_rate": 0.0003, "train/total_time_seconds": 28.698285780847073, "train/time_per_step_avg": 0.021693994030356406, "train/epoch_time_elapsed": 174.25340588018298, "train/estimated_remaining_minutes": 0.414054869474908}
81
- {"step": 1360, "epoch": 0.045837546343107514, "timestamp": 1787090198.5060043, "loss": 3.634907531738281, "grad_norm": 1.3671875, "learning_rate": 0.0003, "train/total_time_seconds": 29.14553214609623, "train/time_per_step_avg": 0.021794799230992794, "train/epoch_time_elapsed": 175.20735903084278, "train/estimated_remaining_minutes": 0.4071802285116385}
82
- {"step": 1380, "epoch": 0.046511627906976744, "timestamp": 1787090199.4406104, "loss": 3.631271743774414, "grad_norm": 1.3203125, "learning_rate": 0.0003, "train/total_time_seconds": 29.57575212791562, "train/time_per_step_avg": 0.021731513440608977, "train/epoch_time_elapsed": 176.14196480065584, "train/estimated_remaining_minutes": 0.4000584828896799}
83
- {"step": 1400, "epoch": 0.04718570947084597, "timestamp": 1787090200.3831522, "loss": 3.6088893890380858, "grad_norm": 1.4765625, "learning_rate": 0.0003, "train/total_time_seconds": 30.00523616373539, "train/time_per_step_avg": 0.02165932346135378, "train/epoch_time_elapsed": 177.08450701460242, "train/estimated_remaining_minutes": 0.3929257116679634}
84
- {"step": 1400, "epoch": 0.04718570947084597, "timestamp": 1787090208.7806206, "eval_loss": 3.620887517929077, "eval_runtime": 8.3958, "eval_samples_per_second": 1134.733, "eval_steps_per_second": 0.834, "train/total_time_seconds": 30.00523616373539, "train/time_per_step_avg": 0.02165932346135378, "train/epoch_time_elapsed": 185.48197343945503, "train/estimated_remaining_minutes": 0.3929257116679634}
85
- {"step": 1420, "epoch": 0.0478597910347152, "timestamp": 1787090209.758754, "loss": 3.6118431091308594, "grad_norm": 1.3125, "learning_rate": 0.0003, "train/total_time_seconds": 30.44477339833975, "train/time_per_step_avg": 0.02179270315915346, "train/epoch_time_elapsed": 186.46010866761208, "train/estimated_remaining_minutes": 0.385919662795856}
86
- {"step": 1440, "epoch": 0.04853387259858443, "timestamp": 1787090210.6889102, "loss": 3.6007781982421876, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 30.87500672414899, "train/time_per_step_avg": 0.02176720943301916, "train/epoch_time_elapsed": 187.390264775604, "train/estimated_remaining_minutes": 0.37879059175460567}
87
- {"step": 1460, "epoch": 0.04920795416245366, "timestamp": 1787090211.629568, "loss": 3.577118682861328, "grad_norm": 1.171875, "learning_rate": 0.0003, "train/total_time_seconds": 31.31340730935335, "train/time_per_step_avg": 0.02167875163257122, "train/epoch_time_elapsed": 188.33092243596911, "train/estimated_remaining_minutes": 0.3717573470516837}
88
- {"step": 1480, "epoch": 0.04988203572632288, "timestamp": 1787090212.5728993, "loss": 3.600400924682617, "grad_norm": 1.453125, "learning_rate": 0.0003, "train/total_time_seconds": 31.75088044628501, "train/time_per_step_avg": 0.021751283183693886, "train/epoch_time_elapsed": 189.27425381541252, "train/estimated_remaining_minutes": 0.3647060591803008}
89
- {"step": 1500, "epoch": 0.05055611729019211, "timestamp": 1787090213.5248115, "loss": 3.5930843353271484, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 32.199186488986015, "train/time_per_step_avg": 0.021939503252506255, "train/epoch_time_elapsed": 190.22616611793637, "train/estimated_remaining_minutes": 0.3577687387665113}
90
- {"step": 1500, "epoch": 0.05055611729019211, "timestamp": 1787090222.1475577, "eval_loss": 3.5886731147766113, "eval_runtime": 8.6212, "eval_samples_per_second": 1105.07, "eval_steps_per_second": 0.812, "train/total_time_seconds": 32.199186488986015, "train/time_per_step_avg": 0.021939503252506255, "train/epoch_time_elapsed": 198.84891088306904, "train/estimated_remaining_minutes": 0.3577687387665113}
91
- {"step": 1520, "epoch": 0.05123019885406134, "timestamp": 1787090223.1094587, "loss": 3.5856658935546877, "grad_norm": 1.703125, "learning_rate": 0.0003, "train/total_time_seconds": 32.62315646559, "train/time_per_step_avg": 0.021783830672502516, "train/epoch_time_elapsed": 199.81081285327673, "train/estimated_remaining_minutes": 0.35055584798550654}
92
- {"step": 1540, "epoch": 0.05190428041793057, "timestamp": 1787090224.036868, "loss": 3.592951202392578, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 33.04965380206704, "train/time_per_step_avg": 0.021746470779180526, "train/epoch_time_elapsed": 200.7382232248783, "train/estimated_remaining_minutes": 0.34337302651498225}
93
- {"step": 1560, "epoch": 0.0525783619817998, "timestamp": 1787090224.9690807, "loss": 3.5940937042236327, "grad_norm": 1.8046875, "learning_rate": 0.0003, "train/total_time_seconds": 33.478857297450304, "train/time_per_step_avg": 0.021654499880969524, "train/epoch_time_elapsed": 201.67043430358171, "train/estimated_remaining_minutes": 0.33621929337183}
94
- {"step": 1580, "epoch": 0.05325244354566903, "timestamp": 1787090225.8977115, "loss": 3.5703506469726562, "grad_norm": 1.375, "learning_rate": 0.0003, "train/total_time_seconds": 33.90842756256461, "train/time_per_step_avg": 0.021575471162796022, "train/epoch_time_elapsed": 202.59906630590558, "train/estimated_remaining_minutes": 0.32906912824429796}
95
- {"step": 1600, "epoch": 0.053926525109538256, "timestamp": 1787090226.8264098, "loss": 3.5667308807373046, "grad_norm": 1.2578125, "learning_rate": 0.0003, "train/total_time_seconds": 34.33643752709031, "train/time_per_step_avg": 0.02137251038104296, "train/epoch_time_elapsed": 203.5277648679912, "train/estimated_remaining_minutes": 0.32190410181647167}
96
- {"step": 1600, "epoch": 0.053926525109538256, "timestamp": 1787090235.2771208, "eval_loss": 3.55812931060791, "eval_runtime": 8.4492, "eval_samples_per_second": 1127.557, "eval_steps_per_second": 0.828, "train/total_time_seconds": 34.33643752709031, "train/time_per_step_avg": 0.02137251038104296, "train/epoch_time_elapsed": 211.97847399488091, "train/estimated_remaining_minutes": 0.32190410181647167}
97
- {"step": 1620, "epoch": 0.054600606673407485, "timestamp": 1787090236.2274826, "loss": 3.578482437133789, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 34.754911847412586, "train/time_per_step_avg": 0.02131755381822586, "train/epoch_time_elapsed": 212.92883710190654, "train/estimated_remaining_minutes": 0.31465352289838555}
98
- {"step": 1640, "epoch": 0.05527468823727671, "timestamp": 1787090237.15594, "loss": 3.5558090209960938, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 35.1784917190671, "train/time_per_step_avg": 0.02128837917000055, "train/epoch_time_elapsed": 213.8572948165238, "train/estimated_remaining_minutes": 0.30745429754469206}
99
- {"step": 1660, "epoch": 0.05594876980114594, "timestamp": 1787090238.085788, "loss": 3.566594696044922, "grad_norm": 1.359375, "learning_rate": 0.0003, "train/total_time_seconds": 35.59984588995576, "train/time_per_step_avg": 0.02120988592505455, "train/epoch_time_elapsed": 214.78714298456907, "train/estimated_remaining_minutes": 0.30023966413215697}
100
- {"step": 1680, "epoch": 0.056622851365015166, "timestamp": 1787090239.0095909, "loss": 3.5346706390380858, "grad_norm": 1.6015625, "learning_rate": 0.0003, "train/total_time_seconds": 36.021644342690706, "train/time_per_step_avg": 0.021132167801260947, "train/epoch_time_elapsed": 215.71094546467066, "train/estimated_remaining_minutes": 0.2930332178671267}
101
- {"step": 1700, "epoch": 0.057296932928884395, "timestamp": 1787090239.9389343, "loss": 3.530739974975586, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 36.447514064610004, "train/time_per_step_avg": 0.021110765375196933, "train/epoch_time_elapsed": 216.64028869196773, "train/estimated_remaining_minutes": 0.2858628554087059}
102
- {"step": 1700, "epoch": 0.057296932928884395, "timestamp": 1787090248.842449, "eval_loss": 3.530102014541626, "eval_runtime": 8.9019, "eval_samples_per_second": 1070.223, "eval_steps_per_second": 0.786, "train/total_time_seconds": 36.447514064610004, "train/time_per_step_avg": 0.021110765375196933, "train/epoch_time_elapsed": 225.54380098730326, "train/estimated_remaining_minutes": 0.2858628554087059}
103
- {"step": 1720, "epoch": 0.057971014492753624, "timestamp": 1787090249.8342133, "loss": 3.519282913208008, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 36.8907287530601, "train/time_per_step_avg": 0.02135816905647516, "train/epoch_time_elapsed": 226.53556845709682, "train/estimated_remaining_minutes": 0.2788252754591752}
104
- {"step": 1740, "epoch": 0.05864509605662285, "timestamp": 1787090250.8161082, "loss": 3.5207279205322264, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 37.33156434074044, "train/time_per_step_avg": 0.021530726216733454, "train/epoch_time_elapsed": 227.51746324449778, "train/estimated_remaining_minutes": 0.2717623457755051}
105
- {"step": 1760, "epoch": 0.05931917762049208, "timestamp": 1787090251.812636, "loss": 3.5107025146484374, "grad_norm": 1.4140625, "learning_rate": 0.0003, "train/total_time_seconds": 37.78914465382695, "train/time_per_step_avg": 0.02189298763871193, "train/epoch_time_elapsed": 228.51399037614465, "train/estimated_remaining_minutes": 0.2648102939756813}
106
- {"step": 1780, "epoch": 0.05999325918436131, "timestamp": 1787090252.737064, "loss": 3.5151645660400392, "grad_norm": 1.4921875, "learning_rate": 0.0003, "train/total_time_seconds": 38.21554037556052, "train/time_per_step_avg": 0.02193896032869816, "train/epoch_time_elapsed": 229.43841843679547, "train/estimated_remaining_minutes": 0.25763285646445294}
107
- {"step": 1800, "epoch": 0.06066734074823053, "timestamp": 1787090253.6519675, "loss": 3.506744384765625, "grad_norm": 1.34375, "learning_rate": 0.0003, "train/total_time_seconds": 38.63394458219409, "train/time_per_step_avg": 0.021864305175840856, "train/epoch_time_elapsed": 230.35332256928086, "train/estimated_remaining_minutes": 0.2504051963660728}
108
- {"step": 1800, "epoch": 0.06066734074823053, "timestamp": 1787090262.13157, "eval_loss": 3.510505437850952, "eval_runtime": 8.4781, "eval_samples_per_second": 1123.716, "eval_steps_per_second": 0.826, "train/total_time_seconds": 38.63394458219409, "train/time_per_step_avg": 0.021864305175840856, "train/epoch_time_elapsed": 238.83292232453823, "train/estimated_remaining_minutes": 0.2504051963660728}
109
- {"step": 1820, "epoch": 0.06134142231209976, "timestamp": 1787090263.1293309, "loss": 3.4982723236083983, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 39.06876355037093, "train/time_per_step_avg": 0.02178034797310829, "train/epoch_time_elapsed": 239.83068535104394, "train/estimated_remaining_minutes": 0.24328534078985564}
110
- {"step": 1840, "epoch": 0.06201550387596899, "timestamp": 1787090264.0554924, "loss": 3.4931972503662108, "grad_norm": 1.6015625, "learning_rate": 0.0003, "train/total_time_seconds": 39.49600052088499, "train/time_per_step_avg": 0.021644361801445484, "train/epoch_time_elapsed": 240.75684678554535, "train/estimated_remaining_minutes": 0.23611739441833418}
111
- {"step": 1860, "epoch": 0.06268958543983821, "timestamp": 1787090264.9857557, "loss": 3.5061672210693358, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 39.91950794309378, "train/time_per_step_avg": 0.021303632892668248, "train/epoch_time_elapsed": 241.68710988014936, "train/estimated_remaining_minutes": 0.2289290778098568}
112
- {"step": 1880, "epoch": 0.06336366700370745, "timestamp": 1787090265.9066734, "loss": 3.4790119171142577, "grad_norm": 1.7421875, "learning_rate": 0.0003, "train/total_time_seconds": 40.34255151450634, "train/time_per_step_avg": 0.02127011138945818, "train/epoch_time_elapsed": 242.60802837461233, "train/estimated_remaining_minutes": 0.22174097463647102}
113
- {"step": 1900, "epoch": 0.06403774856757667, "timestamp": 1787090266.8239949, "loss": 3.4856170654296874, "grad_norm": 1.4296875, "learning_rate": 0.0003, "train/total_time_seconds": 40.76226783171296, "train/time_per_step_avg": 0.021283232495188712, "train/epoch_time_elapsed": 243.5253492332995, "train/estimated_remaining_minutes": 0.2145382517458577}
114
- {"step": 1900, "epoch": 0.06403774856757667, "timestamp": 1787090275.2962322, "eval_loss": 3.486959457397461, "eval_runtime": 8.4707, "eval_samples_per_second": 1124.702, "eval_steps_per_second": 0.826, "train/total_time_seconds": 40.76226783171296, "train/time_per_step_avg": 0.021283232495188712, "train/epoch_time_elapsed": 251.99758548289537, "train/estimated_remaining_minutes": 0.2145382517458577}
115
- {"step": 1920, "epoch": 0.06471183013144591, "timestamp": 1787090276.2503629, "loss": 3.4741825103759765, "grad_norm": 1.625, "learning_rate": 0.0003, "train/total_time_seconds": 41.185888312757015, "train/time_per_step_avg": 0.021171247623860835, "train/epoch_time_elapsed": 252.95171785727143, "train/estimated_remaining_minutes": 0.20735950713020024}
116
- {"step": 1940, "epoch": 0.06538591169531513, "timestamp": 1787090277.180064, "loss": 3.459843063354492, "grad_norm": 1.484375, "learning_rate": 0.0003, "train/total_time_seconds": 41.61065972223878, "train/time_per_step_avg": 0.021146592013537885, "train/epoch_time_elapsed": 253.8814189210534, "train/estimated_remaining_minutes": 0.20018874093173297}
117
- {"step": 1960, "epoch": 0.06605999325918437, "timestamp": 1787090278.1718454, "loss": 3.482832336425781, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 42.0616734996438, "train/time_per_step_avg": 0.02142165556550026, "train/epoch_time_elapsed": 254.873200006783, "train/estimated_remaining_minutes": 0.1931403374983644}
118
- {"step": 1980, "epoch": 0.06673407482305359, "timestamp": 1787090279.1152725, "loss": 3.490843963623047, "grad_norm": 1.296875, "learning_rate": 0.0003, "train/total_time_seconds": 42.502584993839264, "train/time_per_step_avg": 0.021600334793329238, "train/epoch_time_elapsed": 255.81662721559405, "train/estimated_remaining_minutes": 0.18603825081478467}
119
- {"step": 2000, "epoch": 0.06740815638692282, "timestamp": 1787090280.0470285, "loss": 3.471944046020508, "grad_norm": 1.609375, "learning_rate": 0.0003, "train/total_time_seconds": 42.93809688463807, "train/time_per_step_avg": 0.021758290529251097, "train/epoch_time_elapsed": 256.7483834400773, "train/estimated_remaining_minutes": 0.1789087370193253}
120
- {"step": 2000, "epoch": 0.06740815638692282, "timestamp": 1787090288.5251565, "eval_loss": 3.4656903743743896, "eval_runtime": 8.4764, "eval_samples_per_second": 1123.949, "eval_steps_per_second": 0.826, "train/total_time_seconds": 42.93809688463807, "train/time_per_step_avg": 0.021758290529251097, "train/epoch_time_elapsed": 265.22650999203324, "train/estimated_remaining_minutes": 0.1789087370193253}
121
- {"step": 2020, "epoch": 0.06808223795079205, "timestamp": 1787090289.4874835, "loss": 3.4606929779052735, "grad_norm": 1.8046875, "learning_rate": 0.0003, "train/total_time_seconds": 43.36348307877779, "train/time_per_step_avg": 0.02177594766020775, "train/epoch_time_elapsed": 266.188837967813, "train/estimated_remaining_minutes": 0.1717365666486249}
122
- {"step": 2040, "epoch": 0.06875631951466127, "timestamp": 1787090290.4118824, "loss": 3.4594757080078127, "grad_norm": 1.515625, "learning_rate": 0.0003, "train/total_time_seconds": 43.79138084128499, "train/time_per_step_avg": 0.02180721119046211, "train/epoch_time_elapsed": 267.1132374033332, "train/estimated_remaining_minutes": 0.16457545087411027}
123
- {"step": 2060, "epoch": 0.0694304010785305, "timestamp": 1787090291.3374932, "loss": 3.446879577636719, "grad_norm": 1.2265625, "learning_rate": 0.0003, "train/total_time_seconds": 44.2190666384995, "train/time_per_step_avg": 0.02157393138855696, "train/epoch_time_elapsed": 268.0388476587832, "train/estimated_remaining_minutes": 0.15741415308203704}
124
- {"step": 2080, "epoch": 0.07010448264239973, "timestamp": 1787090292.3561628, "loss": 3.4532211303710936, "grad_norm": 1.34375, "learning_rate": 0.0003, "train/total_time_seconds": 44.67429492995143, "train/time_per_step_avg": 0.021717099361121654, "train/epoch_time_elapsed": 269.0575177781284, "train/estimated_remaining_minutes": 0.15034618486041346}
125
- {"step": 2100, "epoch": 0.07077856420626896, "timestamp": 1787090293.307748, "loss": 3.445978546142578, "grad_norm": 1.21875, "learning_rate": 0.0003, "train/total_time_seconds": 45.11734665185213, "train/time_per_step_avg": 0.021792497672140598, "train/epoch_time_elapsed": 270.0091031193733, "train/estimated_remaining_minutes": 0.14322967191064168}
126
- {"step": 2100, "epoch": 0.07077856420626896, "timestamp": 1787090301.7926538, "eval_loss": 3.4454257488250732, "eval_runtime": 8.4831, "eval_samples_per_second": 1123.053, "eval_steps_per_second": 0.825, "train/total_time_seconds": 45.11734665185213, "train/time_per_step_avg": 0.021792497672140598, "train/epoch_time_elapsed": 278.49400681629777, "train/estimated_remaining_minutes": 0.14322967191064168}
127
- {"step": 2120, "epoch": 0.07145264577013818, "timestamp": 1787090302.752118, "loss": 3.446368408203125, "grad_norm": 1.3984375, "learning_rate": 0.0003, "train/total_time_seconds": 45.54672980308533, "train/time_per_step_avg": 0.02183246724307537, "train/epoch_time_elapsed": 279.4534733183682, "train/estimated_remaining_minutes": 0.13606727456896558}
128
- {"step": 2140, "epoch": 0.07212672733400742, "timestamp": 1787090303.6751964, "loss": 3.4467777252197265, "grad_norm": 1.9453125, "learning_rate": 0.0003, "train/total_time_seconds": 45.97253290563822, "train/time_per_step_avg": 0.021811520643532277, "train/epoch_time_elapsed": 280.37655137851834, "train/estimated_remaining_minutes": 0.12889495207188284}
129
- {"step": 2160, "epoch": 0.07280080889787664, "timestamp": 1787090304.601299, "loss": 3.427969741821289, "grad_norm": 1.2421875, "learning_rate": 0.0003, "train/total_time_seconds": 46.40290119126439, "train/time_per_step_avg": 0.021838345527648927, "train/epoch_time_elapsed": 281.30265340954065, "train/estimated_remaining_minutes": 0.1217360062116504}
130
- {"step": 2180, "epoch": 0.07347489046174586, "timestamp": 1787090305.5422475, "loss": 3.437174606323242, "grad_norm": 1.546875, "learning_rate": 0.0003, "train/total_time_seconds": 46.82929648458958, "train/time_per_step_avg": 0.021550015546381474, "train/epoch_time_elapsed": 282.2436023950577, "train/estimated_remaining_minutes": 0.11456708620083077}
131
- {"step": 2200, "epoch": 0.0741489720256151, "timestamp": 1787090306.4651196, "loss": 3.437343215942383, "grad_norm": 1.5, "learning_rate": 0.0003, "train/total_time_seconds": 47.253708347678185, "train/time_per_step_avg": 0.021363616958260535, "train/epoch_time_elapsed": 283.16647424921393, "train/estimated_remaining_minutes": 0.1073947916992686}
132
- {"step": 2200, "epoch": 0.0741489720256151, "timestamp": 1787090314.8796115, "eval_loss": 3.431370735168457, "eval_runtime": 8.4127, "eval_samples_per_second": 1132.457, "eval_steps_per_second": 0.832, "train/total_time_seconds": 47.253708347678185, "train/time_per_step_avg": 0.021363616958260535, "train/epoch_time_elapsed": 291.5809638015926, "train/estimated_remaining_minutes": 0.1073947916992686}
133
- {"step": 2220, "epoch": 0.07482305358948432, "timestamp": 1787090315.8496833, "loss": 3.3987106323242187, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 47.68398727849126, "train/time_per_step_avg": 0.021372574754059313, "train/epoch_time_elapsed": 292.55103803798556, "train/estimated_remaining_minutes": 0.10023660989472637}
134
- {"step": 2240, "epoch": 0.07549713515335356, "timestamp": 1787090316.7768753, "loss": 3.4200679779052736, "grad_norm": 1.25, "learning_rate": 0.0003, "train/total_time_seconds": 48.1130493208766, "train/time_per_step_avg": 0.021405164152383804, "train/epoch_time_elapsed": 293.47823068127036, "train/estimated_remaining_minutes": 0.09307583946002912}
135
- {"step": 2260, "epoch": 0.07617121671722278, "timestamp": 1787090317.712523, "loss": 3.391765594482422, "grad_norm": 1.46875, "learning_rate": 0.0003, "train/total_time_seconds": 48.5431542173028, "train/time_per_step_avg": 0.021402530260384082, "train/epoch_time_elapsed": 294.4138778038323, "train/estimated_remaining_minutes": 0.08591708711027043}
136
- {"step": 2280, "epoch": 0.07684529828109202, "timestamp": 1787090318.6368556, "loss": 3.4147933959960937, "grad_norm": 1.546875, "learning_rate": 0.0003, "train/total_time_seconds": 48.96943225711584, "train/time_per_step_avg": 0.02140135772526264, "train/epoch_time_elapsed": 295.3382097110152, "train/estimated_remaining_minutes": 0.07875201093980616}
137
- {"step": 2300, "epoch": 0.07751937984496124, "timestamp": 1787090319.5716066, "loss": 3.411570358276367, "grad_norm": 1.4609375, "learning_rate": 0.0003, "train/total_time_seconds": 49.40087690204382, "train/time_per_step_avg": 0.02147168554365635, "train/epoch_time_elapsed": 296.272961769253, "train/estimated_remaining_minutes": 0.071595473771078}
138
- {"step": 2300, "epoch": 0.07751937984496124, "timestamp": 1787090327.969581, "eval_loss": 3.4143009185791016, "eval_runtime": 8.3964, "eval_samples_per_second": 1134.65, "eval_steps_per_second": 0.834, "train/total_time_seconds": 49.40087690204382, "train/time_per_step_avg": 0.02147168554365635, "train/epoch_time_elapsed": 304.6709332689643, "train/estimated_remaining_minutes": 0.071595473771078}
139
- {"step": 2320, "epoch": 0.07819346140883048, "timestamp": 1787090328.9620771, "loss": 3.408540725708008, "grad_norm": 1.4765625, "learning_rate": 0.0003, "train/total_time_seconds": 49.833336658775806, "train/time_per_step_avg": 0.021493493802845477, "train/epoch_time_elapsed": 305.66343190521, "train/estimated_remaining_minutes": 0.06443965947255492}
140
- {"step": 2340, "epoch": 0.0788675429726997, "timestamp": 1787090329.8789449, "loss": 3.393010711669922, "grad_norm": 1.46875, "learning_rate": 0.0003, "train/total_time_seconds": 50.2537716627121, "train/time_per_step_avg": 0.02140722341835499, "train/epoch_time_elapsed": 306.58029908686876, "train/estimated_remaining_minutes": 0.05726925545608216}
141
- {"step": 2360, "epoch": 0.07954162453656892, "timestamp": 1787090330.7985592, "loss": 3.4179660797119142, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 50.677648298442364, "train/time_per_step_avg": 0.021344940811395645, "train/epoch_time_elapsed": 307.4999114535749, "train/estimated_remaining_minutes": 0.050105019504109685}
142
- {"step": 2380, "epoch": 0.08021570610043816, "timestamp": 1787090331.7210913, "loss": 3.4083057403564454, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 51.10111549496651, "train/time_per_step_avg": 0.02131683237850666, "train/epoch_time_elapsed": 308.42244643718004, "train/estimated_remaining_minutes": 0.04294211386131639}
143
- {"step": 2400, "epoch": 0.08088978766430738, "timestamp": 1787090332.637041, "loss": 3.405569839477539, "grad_norm": 1.6640625, "learning_rate": 0.0003, "train/total_time_seconds": 51.520632427185774, "train/time_per_step_avg": 0.021197555251419545, "train/epoch_time_elapsed": 309.3383958786726, "train/estimated_remaining_minutes": 0.03577821696332346}
144
- {"step": 2400, "epoch": 0.08088978766430738, "timestamp": 1787090341.087288, "eval_loss": 3.405479907989502, "eval_runtime": 8.4486, "eval_samples_per_second": 1127.639, "eval_steps_per_second": 0.829, "train/total_time_seconds": 51.520632427185774, "train/time_per_step_avg": 0.021197555251419545, "train/epoch_time_elapsed": 317.7886408865452, "train/estimated_remaining_minutes": 0.03577821696332346}
145
- {"step": 2420, "epoch": 0.08156386922817661, "timestamp": 1787090342.070789, "loss": 3.422397232055664, "grad_norm": 1.359375, "learning_rate": 0.0003, "train/total_time_seconds": 51.95656028389931, "train/time_per_step_avg": 0.021232236251235007, "train/epoch_time_elapsed": 318.77214378118515, "train/estimated_remaining_minutes": 0.028626204013167664}
146
- {"step": 2440, "epoch": 0.08223795079204584, "timestamp": 1787090343.04529, "loss": 3.372893524169922, "grad_norm": 1.5234375, "learning_rate": 0.0003, "train/total_time_seconds": 52.38708958029747, "train/time_per_step_avg": 0.021333179175853728, "train/epoch_time_elapsed": 319.74664357304573, "train/estimated_remaining_minutes": 0.021470118680449783}
147
- {"step": 2460, "epoch": 0.08291203235591507, "timestamp": 1787090344.0064027, "loss": 3.4056819915771483, "grad_norm": 1.5078125, "learning_rate": 0.0003, "train/total_time_seconds": 52.83087908104062, "train/time_per_step_avg": 0.02153230782598257, "train/epoch_time_elapsed": 320.7077578008175, "train/estimated_remaining_minutes": 0.01431731140407605}
148
- {"step": 2480, "epoch": 0.0835861139197843, "timestamp": 1787090344.98779, "loss": 3.3840850830078124, "grad_norm": 1.6328125, "learning_rate": 0.0003, "train/total_time_seconds": 53.28470703586936, "train/time_per_step_avg": 0.02183591540902853, "train/epoch_time_elapsed": 321.68914468213916, "train/estimated_remaining_minutes": 0.007161922988692117}
149
- {"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090345.914262, "loss": 3.3768421173095704, "grad_norm": 1.3359375, "learning_rate": 0.0003, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 322.6156167127192, "train/estimated_remaining_minutes": 0.0}
150
- {"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090354.314014, "eval_loss": 3.386671304702759, "eval_runtime": 8.3982, "eval_samples_per_second": 1134.413, "eval_steps_per_second": 0.834, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 331.015366576612, "train/estimated_remaining_minutes": 0.0}
151
- {"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090354.3694549, "train_runtime": 332.0558, "train_samples_per_second": 240.923, "train_steps_per_second": 7.529, "total_flos": 362985553920000.0, "train_loss": 4.1883163818359375, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 331.0708075389266, "train/estimated_remaining_minutes": 0.0}
152
- {"step": 2500, "epoch": 0.08426019548365352, "timestamp": 1787090363.1779864, "eval_loss": 3.386671304702759, "eval_runtime": 8.8055, "eval_samples_per_second": 1081.938, "eval_steps_per_second": 0.795, "train/total_time_seconds": 53.71119434013963, "train/time_per_step_avg": 0.021905619129538537, "train/epoch_time_elapsed": 339.8793397396803, "train/estimated_remaining_minutes": 0.0}
 
1
+ {"step": 20, "epoch": 0.0016852039096730705, "timestamp": 1787159744.744951, "loss": 8.200721740722656, "grad_norm": 1.2421875, "learning_rate": 9.5e-05, "train/total_time_seconds": 1.1820026859641075, "train/time_per_step_avg": 0.059100134298205376, "train/epoch_time_elapsed": 2.652187306433916, "train/estimated_remaining_minutes": 0.9653021935373545}
2
+ {"step": 40, "epoch": 0.003370407819346141, "timestamp": 1787159746.6767337, "loss": 7.764720153808594, "grad_norm": 1.2421875, "learning_rate": 0.00019500000000000002, "train/total_time_seconds": 1.8186652921140194, "train/time_per_step_avg": 0.04546663230285049, "train/epoch_time_elapsed": 4.583965107798576, "train/estimated_remaining_minutes": 0.7274661168456078}
3
+ {"step": 60, "epoch": 0.005055611729019211, "timestamp": 1787159748.5245914, "loss": 7.160160064697266, "grad_norm": 1.2265625, "learning_rate": 0.000295, "train/total_time_seconds": 2.4542956836521626, "train/time_per_step_avg": 0.04090492806086938, "train/epoch_time_elapsed": 6.4318275190889835, "train/estimated_remaining_minutes": 0.6408438729536203}
4
+ {"step": 80, "epoch": 0.006740815638692282, "timestamp": 1787159750.3386374, "loss": 6.480591583251953, "grad_norm": 1.2109375, "learning_rate": 0.000395, "train/total_time_seconds": 3.086422026157379, "train/time_per_step_avg": 0.03858027532696724, "train/epoch_time_elapsed": 8.245873432606459, "train/estimated_remaining_minutes": 0.5915642216801643}
5
+ {"step": 100, "epoch": 0.008426019548365353, "timestamp": 1787159752.2435765, "loss": 5.919136428833008, "grad_norm": 1.0078125, "learning_rate": 0.000495, "train/total_time_seconds": 3.722423255443573, "train/time_per_step_avg": 0.03722423255443573, "train/epoch_time_elapsed": 10.150811605155468, "train/estimated_remaining_minutes": 0.5583634883165359}
6
+ {"step": 100, "epoch": 0.008426019548365353, "timestamp": 1787159760.215479, "eval_loss": 5.634258270263672, "eval_runtime": 7.97, "eval_samples_per_second": 1195.356, "eval_steps_per_second": 0.878, "train/total_time_seconds": 3.722423255443573, "train/time_per_step_avg": 0.03722423255443573, "train/epoch_time_elapsed": 18.122713901102543, "train/estimated_remaining_minutes": 0.5583634883165359}
7
+ {"step": 120, "epoch": 0.010111223458038422, "timestamp": 1787159762.0986185, "loss": 5.412916946411133, "grad_norm": 1.171875, "learning_rate": 0.0005949999999999999, "train/total_time_seconds": 4.363878160715103, "train/time_per_step_avg": 0.031818754747509954, "train/epoch_time_elapsed": 20.005853559821844, "train/estimated_remaining_minutes": 0.5333628863096237}
8
+ {"step": 140, "epoch": 0.011796427367711493, "timestamp": 1787159763.9296083, "loss": 5.0327880859375, "grad_norm": 1.515625, "learning_rate": 0.000695, "train/total_time_seconds": 5.001869980245829, "train/time_per_step_avg": 0.03183204688131809, "train/epoch_time_elapsed": 21.836841892451048, "train/estimated_remaining_minutes": 0.5120962122632634}
9
+ {"step": 160, "epoch": 0.013481631277384564, "timestamp": 1787159765.7692258, "loss": 4.69476089477539, "grad_norm": 0.9765625, "learning_rate": 0.000795, "train/total_time_seconds": 5.6416174322366714, "train/time_per_step_avg": 0.03187321748584509, "train/epoch_time_elapsed": 23.676462575793266, "train/estimated_remaining_minutes": 0.4936415253207088}
10
+ {"step": 180, "epoch": 0.015166835187057633, "timestamp": 1787159767.6319602, "loss": 4.4246673583984375, "grad_norm": 0.8828125, "learning_rate": 0.0008950000000000001, "train/total_time_seconds": 6.286054410040379, "train/time_per_step_avg": 0.03199632383882999, "train/epoch_time_elapsed": 25.539197202771902, "train/estimated_remaining_minutes": 0.4772745015030658}
11
+ {"step": 200, "epoch": 0.016852039096730706, "timestamp": 1787159769.4737778, "loss": 4.204695129394532, "grad_norm": 0.87890625, "learning_rate": 0.000995, "train/total_time_seconds": 6.9344093687832355, "train/time_per_step_avg": 0.03211986113339663, "train/epoch_time_elapsed": 27.38101428002119, "train/estimated_remaining_minutes": 0.4622939579188824}
12
+ {"step": 200, "epoch": 0.016852039096730706, "timestamp": 1787159777.4086134, "eval_loss": 4.133424282073975, "eval_runtime": 7.9329, "eval_samples_per_second": 1200.942, "eval_steps_per_second": 0.882, "train/total_time_seconds": 6.9344093687832355, "train/time_per_step_avg": 0.03211986113339663, "train/epoch_time_elapsed": 35.31584985554218, "train/estimated_remaining_minutes": 0.4622939579188824}
13
+ {"step": 220, "epoch": 0.018537243006403775, "timestamp": 1787159779.2836711, "loss": 4.048733520507812, "grad_norm": 0.6875, "learning_rate": 0.001, "train/total_time_seconds": 7.581706669181585, "train/time_per_step_avg": 0.032178285084664825, "train/epoch_time_elapsed": 37.19090821221471, "train/estimated_remaining_minutes": 0.4480099395425482}
14
+ {"step": 240, "epoch": 0.020222446916076844, "timestamp": 1787159781.146838, "loss": 3.9190834045410154, "grad_norm": 0.73828125, "learning_rate": 0.001, "train/total_time_seconds": 8.228051170706749, "train/time_per_step_avg": 0.0322618119046092, "train/epoch_time_elapsed": 39.054074600338936, "train/estimated_remaining_minutes": 0.434258256231745}
15
+ {"step": 260, "epoch": 0.021907650825749917, "timestamp": 1787159782.9797618, "loss": 3.7833847045898437, "grad_norm": 0.68359375, "learning_rate": 0.001, "train/total_time_seconds": 8.87311216443777, "train/time_per_step_avg": 0.03231494732201099, "train/epoch_time_elapsed": 40.886998776346445, "train/estimated_remaining_minutes": 0.420904038569484}
16
+ {"step": 280, "epoch": 0.023592854735422986, "timestamp": 1787159784.913182, "loss": 3.700960159301758, "grad_norm": 0.82421875, "learning_rate": 0.001, "train/total_time_seconds": 9.516451723873615, "train/time_per_step_avg": 0.032303973138332366, "train/epoch_time_elapsed": 42.82041900604963, "train/estimated_remaining_minutes": 0.4078479310231549}
17
+ {"step": 300, "epoch": 0.025278058645096056, "timestamp": 1787159786.7454824, "loss": 3.6359439849853517, "grad_norm": 0.984375, "learning_rate": 0.001, "train/total_time_seconds": 10.157686203718185, "train/time_per_step_avg": 0.032232768349349496, "train/epoch_time_elapsed": 44.65271816775203, "train/estimated_remaining_minutes": 0.3950211301445961}
18
+ {"step": 300, "epoch": 0.025278058645096056, "timestamp": 1787159794.7201555, "eval_loss": 3.5975606441497803, "eval_runtime": 7.973, "eval_samples_per_second": 1194.91, "eval_steps_per_second": 0.878, "train/total_time_seconds": 10.157686203718185, "train/time_per_step_avg": 0.032232768349349496, "train/epoch_time_elapsed": 52.627391546964645, "train/estimated_remaining_minutes": 0.3950211301445961}
19
+ {"step": 320, "epoch": 0.026963262554769128, "timestamp": 1787159796.601689, "loss": 3.5393722534179686, "grad_norm": 0.80859375, "learning_rate": 0.001, "train/total_time_seconds": 10.796786732971668, "train/time_per_step_avg": 0.03215080063790083, "train/epoch_time_elapsed": 54.50892523676157, "train/estimated_remaining_minutes": 0.3823861967927466}
20
+ {"step": 340, "epoch": 0.028648466464442197, "timestamp": 1787159798.453831, "loss": 3.492219924926758, "grad_norm": 0.72265625, "learning_rate": 0.001, "train/total_time_seconds": 11.440050598233938, "train/time_per_step_avg": 0.032119994275271894, "train/epoch_time_elapsed": 56.36106700450182, "train/estimated_remaining_minutes": 0.3701192840605098}
21
+ {"step": 360, "epoch": 0.030333670374115267, "timestamp": 1787159800.2839177, "loss": 3.4430694580078125, "grad_norm": 0.8203125, "learning_rate": 0.001, "train/total_time_seconds": 12.078172758221626, "train/time_per_step_avg": 0.03205060593783855, "train/epoch_time_elapsed": 58.19115437567234, "train/estimated_remaining_minutes": 0.35787178542878895}
22
+ {"step": 380, "epoch": 0.032018874283788336, "timestamp": 1787159802.1708894, "loss": 3.3989883422851563, "grad_norm": 0.8203125, "learning_rate": 0.001, "train/total_time_seconds": 12.717796016484499, "train/time_per_step_avg": 0.03201344292610884, "train/epoch_time_elapsed": 60.07812675833702, "train/estimated_remaining_minutes": 0.34583480395703464}
23
+ {"step": 400, "epoch": 0.03370407819346141, "timestamp": 1787159804.0363617, "loss": 3.341664123535156, "grad_norm": 0.6484375, "learning_rate": 0.001, "train/total_time_seconds": 13.358520355075598, "train/time_per_step_avg": 0.03200834151357412, "train/epoch_time_elapsed": 61.94359891861677, "train/estimated_remaining_minutes": 0.33396300887688996}
24
+ {"step": 400, "epoch": 0.03370407819346141, "timestamp": 1787159812.2065763, "eval_loss": 3.3295202255249023, "eval_runtime": 8.1687, "eval_samples_per_second": 1166.287, "eval_steps_per_second": 0.857, "train/total_time_seconds": 13.358520355075598, "train/time_per_step_avg": 0.03200834151357412, "train/epoch_time_elapsed": 70.11381255835295, "train/estimated_remaining_minutes": 0.33396300887688996}
25
+ {"step": 420, "epoch": 0.03538928210313448, "timestamp": 1787159814.10439, "loss": 3.3047794342041015, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 14.000833839178085, "train/time_per_step_avg": 0.032040471062064174, "train/epoch_time_elapsed": 72.01162526011467, "train/estimated_remaining_minutes": 0.32224141375886073}
26
+ {"step": 440, "epoch": 0.03707448601280755, "timestamp": 1787159816.047692, "loss": 3.258700942993164, "grad_norm": 0.71875, "learning_rate": 0.001, "train/total_time_seconds": 14.651910278946161, "train/time_per_step_avg": 0.03211859680712223, "train/epoch_time_elapsed": 73.95492927730083, "train/estimated_remaining_minutes": 0.31079809682613063}
27
+ {"step": 460, "epoch": 0.03875968992248062, "timestamp": 1787159817.9195716, "loss": 3.229313278198242, "grad_norm": 0.73046875, "learning_rate": 0.001, "train/total_time_seconds": 15.297841779887676, "train/time_per_step_avg": 0.0321966902166605, "train/epoch_time_elapsed": 75.82680872827768, "train/estimated_remaining_minutes": 0.29930560004128065}
28
+ {"step": 480, "epoch": 0.04044489383215369, "timestamp": 1787159819.7689188, "loss": 3.202067565917969, "grad_norm": 0.703125, "learning_rate": 0.001, "train/total_time_seconds": 15.936651539057493, "train/time_per_step_avg": 0.03218855522572994, "train/epoch_time_elapsed": 77.67615579813719, "train/estimated_remaining_minutes": 0.2877450972329825}
29
+ {"step": 500, "epoch": 0.04213009774182676, "timestamp": 1787159821.6968458, "loss": 3.163772201538086, "grad_norm": 0.7265625, "learning_rate": 0.001, "train/total_time_seconds": 16.570955902338028, "train/time_per_step_avg": 0.0321243554726243, "train/epoch_time_elapsed": 79.60408221185207, "train/estimated_remaining_minutes": 0.27618259837230047}
30
+ {"step": 500, "epoch": 0.04213009774182676, "timestamp": 1787159829.648704, "eval_loss": 3.157355308532715, "eval_runtime": 7.9503, "eval_samples_per_second": 1198.315, "eval_steps_per_second": 0.88, "train/total_time_seconds": 16.570955902338028, "train/time_per_step_avg": 0.0321243554726243, "train/epoch_time_elapsed": 87.5559413023293, "train/estimated_remaining_minutes": 0.27618259837230047}
31
+ {"step": 520, "epoch": 0.043815301651499834, "timestamp": 1787159831.4997501, "loss": 3.1448848724365233, "grad_norm": 0.6953125, "learning_rate": 0.001, "train/total_time_seconds": 17.204289257526398, "train/time_per_step_avg": 0.032034554183483124, "train/epoch_time_elapsed": 89.40698843076825, "train/estimated_remaining_minutes": 0.2646813731927138}
32
+ {"step": 540, "epoch": 0.0455005055611729, "timestamp": 1787159833.3223414, "loss": 3.103527069091797, "grad_norm": 0.828125, "learning_rate": 0.001, "train/total_time_seconds": 17.837791569530964, "train/time_per_step_avg": 0.03185881290584803, "train/epoch_time_elapsed": 91.22957941517234, "train/estimated_remaining_minutes": 0.2532525963575384}
33
+ {"step": 560, "epoch": 0.04718570947084597, "timestamp": 1787159835.1489348, "loss": 3.08404541015625, "grad_norm": 0.7578125, "learning_rate": 0.001, "train/total_time_seconds": 18.47138799726963, "train/time_per_step_avg": 0.03173546217381954, "train/epoch_time_elapsed": 93.05617316067219, "train/estimated_remaining_minutes": 0.241887223773769}
34
+ {"step": 580, "epoch": 0.04887091338051904, "timestamp": 1787159836.9735324, "loss": 3.0501741409301757, "grad_norm": 0.76953125, "learning_rate": 0.001, "train/total_time_seconds": 19.10541071370244, "train/time_per_step_avg": 0.03168759174644947, "train/epoch_time_elapsed": 94.88077012076974, "train/estimated_remaining_minutes": 0.23058254309640874}
35
+ {"step": 600, "epoch": 0.05055611729019211, "timestamp": 1787159839.033019, "loss": 3.0376760482788088, "grad_norm": 0.96875, "learning_rate": 0.001, "train/total_time_seconds": 19.750634588301182, "train/time_per_step_avg": 0.03179678685963154, "train/epoch_time_elapsed": 96.94025699794292, "train/estimated_remaining_minutes": 0.2194514954255687}
36
+ {"step": 600, "epoch": 0.05055611729019211, "timestamp": 1787159847.154113, "eval_loss": 3.0310890674591064, "eval_runtime": 8.1195, "eval_samples_per_second": 1173.346, "eval_steps_per_second": 0.862, "train/total_time_seconds": 19.750634588301182, "train/time_per_step_avg": 0.03179678685963154, "train/epoch_time_elapsed": 105.06134972721338, "train/estimated_remaining_minutes": 0.2194514954255687}
37
+ {"step": 620, "epoch": 0.05224132119986518, "timestamp": 1787159849.0143704, "loss": 3.0314205169677733, "grad_norm": 0.82421875, "learning_rate": 0.001, "train/total_time_seconds": 20.382843017578125, "train/time_per_step_avg": 0.03178553760051727, "train/epoch_time_elapsed": 106.92160803079605, "train/estimated_remaining_minutes": 0.2082118372763357}
38
+ {"step": 640, "epoch": 0.053926525109538256, "timestamp": 1787159850.828231, "loss": 3.0014928817749023, "grad_norm": 0.640625, "learning_rate": 0.001, "train/total_time_seconds": 21.01229700446129, "train/time_per_step_avg": 0.03174505434930325, "train/epoch_time_elapsed": 108.73546917364001, "train/estimated_remaining_minutes": 0.19699028441682456}
39
+ {"step": 660, "epoch": 0.055611729019211326, "timestamp": 1787159852.7167454, "loss": 2.9963293075561523, "grad_norm": 0.70703125, "learning_rate": 0.001, "train/total_time_seconds": 21.644355565309525, "train/time_per_step_avg": 0.031729675680398944, "train/epoch_time_elapsed": 110.62398328632116, "train/estimated_remaining_minutes": 0.18583537606578884}
40
+ {"step": 680, "epoch": 0.057296932928884395, "timestamp": 1787159854.5207703, "loss": 2.9568761825561523, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 22.27377400547266, "train/time_per_step_avg": 0.031683632917702195, "train/epoch_time_elapsed": 112.42800731211901, "train/estimated_remaining_minutes": 0.1746962667095895}
41
+ {"step": 700, "epoch": 0.058982136838557464, "timestamp": 1787159856.334935, "loss": 2.9395275115966797, "grad_norm": 0.69921875, "learning_rate": 0.001, "train/total_time_seconds": 22.903755206614733, "train/time_per_step_avg": 0.03153120618313551, "train/epoch_time_elapsed": 114.24217312037945, "train/estimated_remaining_minutes": 0.16359825147581952}
42
+ {"step": 700, "epoch": 0.058982136838557464, "timestamp": 1787159864.3117049, "eval_loss": 2.939429759979248, "eval_runtime": 7.975, "eval_samples_per_second": 1194.602, "eval_steps_per_second": 0.878, "train/total_time_seconds": 22.903755206614733, "train/time_per_step_avg": 0.03153120618313551, "train/epoch_time_elapsed": 122.21894185245037, "train/estimated_remaining_minutes": 0.16359825147581952}
43
+ {"step": 720, "epoch": 0.06066734074823053, "timestamp": 1787159866.154933, "loss": 2.922671890258789, "grad_norm": 0.6484375, "learning_rate": 0.001, "train/total_time_seconds": 23.537149403244257, "train/time_per_step_avg": 0.03154306385666132, "train/epoch_time_elapsed": 124.06217032298446, "train/estimated_remaining_minutes": 0.15255559798399052}
44
+ {"step": 740, "epoch": 0.06235254465790361, "timestamp": 1787159867.9758906, "loss": 2.9145633697509767, "grad_norm": 0.95703125, "learning_rate": 0.001, "train/total_time_seconds": 24.17463242635131, "train/time_per_step_avg": 0.0316233542189002, "train/epoch_time_elapsed": 125.88312843069434, "train/estimated_remaining_minutes": 0.14156316285701215}
45
+ {"step": 760, "epoch": 0.06403774856757667, "timestamp": 1787159869.9564908, "loss": 2.8876668930053713, "grad_norm": 0.7109375, "learning_rate": 0.001, "train/total_time_seconds": 24.810363072901964, "train/time_per_step_avg": 0.031660075075924395, "train/epoch_time_elapsed": 127.86372835934162, "train/estimated_remaining_minutes": 0.13058085827843138}
46
+ {"step": 780, "epoch": 0.06572295247724974, "timestamp": 1787159871.839654, "loss": 2.8692466735839846, "grad_norm": 0.69140625, "learning_rate": 0.001, "train/total_time_seconds": 25.449912142008543, "train/time_per_step_avg": 0.03176138136535883, "train/epoch_time_elapsed": 129.74689135327935, "train/estimated_remaining_minutes": 0.11963633912909998}
47
+ {"step": 800, "epoch": 0.06740815638692282, "timestamp": 1787159873.668295, "loss": 2.8788633346557617, "grad_norm": 0.68359375, "learning_rate": 0.001, "train/total_time_seconds": 26.088915783911943, "train/time_per_step_avg": 0.03185160577297211, "train/epoch_time_elapsed": 131.57553120702505, "train/estimated_remaining_minutes": 0.10870381576629977}
48
+ {"step": 800, "epoch": 0.06740815638692282, "timestamp": 1787159881.9473228, "eval_loss": 2.8632190227508545, "eval_runtime": 8.2773, "eval_samples_per_second": 1150.973, "eval_steps_per_second": 0.846, "train/total_time_seconds": 26.088915783911943, "train/time_per_step_avg": 0.03185160577297211, "train/epoch_time_elapsed": 139.8545593805611, "train/estimated_remaining_minutes": 0.10870381576629977}
49
+ {"step": 820, "epoch": 0.0690933602965959, "timestamp": 1787159883.8665261, "loss": 2.8496471405029298, "grad_norm": 0.6796875, "learning_rate": 0.001, "train/total_time_seconds": 26.722276385873556, "train/time_per_step_avg": 0.031851269826292994, "train/epoch_time_elapsed": 141.77376406639814, "train/estimated_remaining_minutes": 0.09776442580197643}
50
+ {"step": 840, "epoch": 0.07077856420626896, "timestamp": 1787159885.6946013, "loss": 2.847636604309082, "grad_norm": 0.62109375, "learning_rate": 0.001, "train/total_time_seconds": 27.35644706711173, "train/time_per_step_avg": 0.03181814640760422, "train/epoch_time_elapsed": 143.60183906927705, "train/estimated_remaining_minutes": 0.0868458637051166}
51
+ {"step": 860, "epoch": 0.07246376811594203, "timestamp": 1787159887.564113, "loss": 2.8339916229248048, "grad_norm": 0.73046875, "learning_rate": 0.001, "train/total_time_seconds": 27.989167381078005, "train/time_per_step_avg": 0.031788043081760406, "train/epoch_time_elapsed": 145.47135097533464, "train/estimated_remaining_minutes": 0.07593960142152947}
52
+ {"step": 880, "epoch": 0.0741489720256151, "timestamp": 1787159889.387239, "loss": 2.8253028869628904, "grad_norm": 0.69921875, "learning_rate": 0.001, "train/total_time_seconds": 28.62036318704486, "train/time_per_step_avg": 0.03170451045036316, "train/epoch_time_elapsed": 147.29447646439075, "train/estimated_remaining_minutes": 0.0650462799705565}
53
+ {"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159891.2191205, "loss": 2.7853120803833007, "grad_norm": 0.6875, "learning_rate": 0.001, "train/total_time_seconds": 29.254599027335644, "train/time_per_step_avg": 0.031656832434237, "train/epoch_time_elapsed": 149.1263581365347, "train/estimated_remaining_minutes": 0.0541751833839549}
54
+ {"step": 900, "epoch": 0.07583417593528817, "timestamp": 1787159899.179946, "eval_loss": 2.8034849166870117, "eval_runtime": 7.9593, "eval_samples_per_second": 1196.967, "eval_steps_per_second": 0.879, "train/total_time_seconds": 29.254599027335644, "train/time_per_step_avg": 0.031656832434237, "train/epoch_time_elapsed": 157.0871831253171, "train/estimated_remaining_minutes": 0.0541751833839549}
55
+ {"step": 920, "epoch": 0.07751937984496124, "timestamp": 1787159901.2376366, "loss": 2.794095993041992, "grad_norm": 0.609375, "learning_rate": 0.001, "train/total_time_seconds": 29.887034360319376, "train/time_per_step_avg": 0.0316475797444582, "train/epoch_time_elapsed": 159.14487295597792, "train/estimated_remaining_minutes": 0.0433145425511875}
56
+ {"step": 940, "epoch": 0.07920458375463431, "timestamp": 1787159903.067668, "loss": 2.7890634536743164, "grad_norm": 0.671875, "learning_rate": 0.001, "train/total_time_seconds": 30.5212767906487, "train/time_per_step_avg": 0.03164829723536968, "train/epoch_time_elapsed": 160.9749058149755, "train/estimated_remaining_minutes": 0.03246944339430713}
57
+ {"step": 960, "epoch": 0.08088978766430738, "timestamp": 1787159904.9205842, "loss": 2.784823989868164, "grad_norm": 0.58203125, "learning_rate": 0.001, "train/total_time_seconds": 31.156463339924812, "train/time_per_step_avg": 0.03167295958846807, "train/epoch_time_elapsed": 162.82782202214003, "train/estimated_remaining_minutes": 0.021636432874947785}
58
+ {"step": 980, "epoch": 0.08257499157398045, "timestamp": 1787159906.7382162, "loss": 2.775278663635254, "grad_norm": 0.60546875, "learning_rate": 0.001, "train/total_time_seconds": 31.790426205843687, "train/time_per_step_avg": 0.03170063018798828, "train/epoch_time_elapsed": 164.64545308053493, "train/estimated_remaining_minutes": 0.010813070138042068}
59
+ {"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159908.6763556, "loss": 2.753749656677246, "grad_norm": 0.78125, "learning_rate": 0.001, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 166.58359253406525, "train/estimated_remaining_minutes": 0.0}
60
+ {"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159916.6100245, "eval_loss": 2.7597591876983643, "eval_runtime": 7.9273, "eval_samples_per_second": 1201.797, "eval_steps_per_second": 0.883, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 174.5172589495778, "train/estimated_remaining_minutes": 0.0}
61
+ {"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159916.6630983, "train_runtime": 175.4068, "train_samples_per_second": 456.083, "train_steps_per_second": 5.701, "total_flos": 362985553920000.0, "train_loss": 3.6923015975952147, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 174.57033431902528, "train/estimated_remaining_minutes": 0.0}
62
+ {"step": 1000, "epoch": 0.08426019548365352, "timestamp": 1787159924.561638, "eval_loss": 2.7597591876983643, "eval_runtime": 7.8957, "eval_samples_per_second": 1206.607, "eval_steps_per_second": 0.887, "train/total_time_seconds": 32.43894848972559, "train/time_per_step_avg": 0.03184349462389946, "train/epoch_time_elapsed": 182.46887450292706, "train/estimated_remaining_minutes": 0.0}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
zain/Activation/out/sweep_summary.json CHANGED
@@ -1,9 +1,9 @@
1
  [
2
  {
3
- "variant": "mlp-linear-3L",
4
- "eval_loss": 2.841240167617798,
5
- "out": "out/mlp-linear-3L_run",
6
- "run_name": "LM-mlp-linear-3L-1.0M-20260819-171214",
7
  "status": "success"
8
  }
9
  ]
 
1
  [
2
  {
3
+ "variant": "mlp-linear-9L",
4
+ "eval_loss": 2.7597591876983643,
5
+ "out": "out/mlp-linear-9L_run",
6
+ "run_name": "LM-mlp-linear-9L-2.0M-20260819-171540",
7
  "status": "success"
8
  }
9
  ]
zain/Activation/wandb/debug-internal.log CHANGED
@@ -7,3 +7,35 @@
7
  {"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
8
  {"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
  {"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7
  {"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
8
  {"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
  {"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
+ {"time":"2026-08-19T17:15:57.093327808Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":5,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":9,"uploaded_len":2}
11
+ {"time":"2026-08-19T17:15:57.208750755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
+ {"time":"2026-08-19T17:16:12.09424474Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":6,"events_offset":1,"events_lines":2,"console_offset":2,"console_lines":1}
13
+ {"time":"2026-08-19T17:16:12.254954463Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
14
+ {"time":"2026-08-19T17:16:27.093741092Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":6,"events_offset":3,"events_lines":2,"console_offset":8,"console_lines":22}
15
+ {"time":"2026-08-19T17:16:27.218480793Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
16
+ {"time":"2026-08-19T17:16:42.09555922Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":4,"events_offset":5,"events_lines":2,"console_offset":24,"console_lines":1}
17
+ {"time":"2026-08-19T17:16:42.264597826Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
18
+ {"time":"2026-08-19T17:16:57.09386987Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":5,"events_offset":7,"events_lines":2,"console_offset":30,"console_lines":19}
19
+ {"time":"2026-08-19T17:16:57.209800875Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
20
+ {"time":"2026-08-19T17:17:12.094049325Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":5,"events_offset":9,"events_lines":2,"console_offset":46,"console_lines":1}
21
+ {"time":"2026-08-19T17:17:12.252024387Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
22
+ {"time":"2026-08-19T17:17:27.093470433Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":4,"events_offset":11,"events_lines":2,"console_offset":49,"console_lines":15}
23
+ {"time":"2026-08-19T17:17:27.202702289Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
24
+ {"time":"2026-08-19T17:17:42.093533111Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":57,"console_lines":1}
25
+ {"time":"2026-08-19T17:17:42.240909652Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
26
+ {"time":"2026-08-19T17:17:57.093850562Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":63,"console_lines":23}
27
+ {"time":"2026-08-19T17:17:57.20571167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
28
+ {"time":"2026-08-19T17:18:12.094059821Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":17,"events_lines":2,"console_offset":79,"console_lines":1}
29
+ {"time":"2026-08-19T17:18:12.283285285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
30
+ {"time":"2026-08-19T17:18:27.0939628Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":5,"events_offset":19,"events_lines":2,"console_offset":85,"console_lines":21}
31
+ {"time":"2026-08-19T17:18:27.220681676Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
32
+ {"time":"2026-08-19T17:18:42.09374958Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":3,"events_offset":21,"events_lines":2,"console_offset":101,"console_lines":1}
33
+ {"time":"2026-08-19T17:18:42.218715049Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
34
+ {"time":"2026-08-19T17:18:45.014142777Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
35
+ {"time":"2026-08-19T17:18:45.014366349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":23,"events_lines":1,"console_offset":106,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
36
+ {"time":"2026-08-19T17:18:45.153108654Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
37
+ {"time":"2026-08-19T17:18:45.154305649Z","level":"INFO","msg":"handler: operation stats","stats":{}}
38
+ {"time":"2026-08-19T17:18:45.157232952Z","level":"INFO","msg":"stream: finishing up"}
39
+ {"time":"2026-08-19T17:18:45.157265556Z","level":"INFO","msg":"handler: closed"}
40
+ {"time":"2026-08-19T17:18:45.161743858Z","level":"INFO","msg":"sender: closed"}
41
+ {"time":"2026-08-19T17:18:45.161763043Z","level":"INFO","msg":"stream: all finished"}
zain/Activation/wandb/debug.log CHANGED
@@ -21,3 +21,8 @@ config: {'_wandb': {}}
21
  2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
22
  2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
23
  2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
 
 
 
 
 
 
21
  2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
22
  2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
23
  2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
24
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate-checking/bhq23mh5
25
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
26
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2570] restore
27
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2576] restore done
28
+ 2026-08-19 17:18:45,156 INFO MainThread:3953242 [wandb_run.py:_footer_sync_info():3993] logging synced files
zain/Activation/wandb/run-20260819_163845-74tq2syl/logs/debug-core.log CHANGED
@@ -65,3 +65,12 @@
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
 
 
 
 
 
 
 
 
 
 
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_164534-glh82iyh/logs/debug-core.log CHANGED
@@ -65,3 +65,12 @@
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
 
 
 
 
 
 
 
 
 
 
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_164553-44x03ghu/logs/debug-core.log CHANGED
@@ -65,3 +65,12 @@
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
 
 
 
 
 
 
 
 
 
 
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_170854-zhg4u13t/logs/debug-core.log CHANGED
@@ -65,3 +65,12 @@
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
 
 
 
 
 
 
 
 
 
 
65
  {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
  {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
  {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_171215-2nil4rqn/logs/debug-core.log CHANGED
@@ -62,3 +62,15 @@
62
  {"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
63
  {"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
64
  {"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  {"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
63
  {"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
64
  {"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
65
+ {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
+ {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
+ {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/config.yaml ADDED
@@ -0,0 +1,435 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _name_or_path:
2
+ value: ""
3
+ _wandb:
4
+ value:
5
+ cli_version: 0.28.1
6
+ e:
7
+ 2l4ndnmtei7frfwveku3d762xxak1bdj:
8
+ args:
9
+ - --config
10
+ - /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml
11
+ - --variants
12
+ - mlp-linear-9L
13
+ codePath: sweep.py
14
+ codePathLocal: sweep.py
15
+ cpu_count: 112
16
+ cpu_count_logical: 224
17
+ cudaVersion: "12.4"
18
+ disk:
19
+ /:
20
+ total: "1560765693952"
21
+ used: "708485218304"
22
+ email: deepnevro@gmail.com
23
+ executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python
24
+ git:
25
+ commit: 463c9961366755fc55f02df9a0d471b3cbcb025e
26
+ remote: https://github.com/w-ahmad1a10/Activation.git
27
+ gpu: NVIDIA H100 80GB HBM3
28
+ gpu_count: 8
29
+ gpu_nvidia:
30
+ - architecture: Hopper
31
+ cudaCores: 16896
32
+ memoryTotal: "85520809984"
33
+ name: NVIDIA H100 80GB HBM3
34
+ uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae
35
+ - architecture: Hopper
36
+ cudaCores: 16896
37
+ memoryTotal: "85520809984"
38
+ name: NVIDIA H100 80GB HBM3
39
+ uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3
40
+ - architecture: Hopper
41
+ cudaCores: 16896
42
+ memoryTotal: "85520809984"
43
+ name: NVIDIA H100 80GB HBM3
44
+ uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab
45
+ - architecture: Hopper
46
+ cudaCores: 16896
47
+ memoryTotal: "85520809984"
48
+ name: NVIDIA H100 80GB HBM3
49
+ uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864
50
+ - architecture: Hopper
51
+ cudaCores: 16896
52
+ memoryTotal: "85520809984"
53
+ name: NVIDIA H100 80GB HBM3
54
+ uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef
55
+ - architecture: Hopper
56
+ cudaCores: 16896
57
+ memoryTotal: "85520809984"
58
+ name: NVIDIA H100 80GB HBM3
59
+ uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54
60
+ - architecture: Hopper
61
+ cudaCores: 16896
62
+ memoryTotal: "85520809984"
63
+ name: NVIDIA H100 80GB HBM3
64
+ uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9
65
+ - architecture: Hopper
66
+ cudaCores: 16896
67
+ memoryTotal: "85520809984"
68
+ name: NVIDIA H100 80GB HBM3
69
+ uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea
70
+ host: deeplens-k3s-node1
71
+ memory:
72
+ total: "2164089937920"
73
+ os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35
74
+ program: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py
75
+ python: CPython 3.11.15
76
+ root: /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation
77
+ startedAt: "2026-08-19T17:15:41.408834Z"
78
+ writerId: 2l4ndnmtei7frfwveku3d762xxak1bdj
79
+ m:
80
+ - "1": train/global_step
81
+ "6":
82
+ - 3
83
+ "7": []
84
+ - "2": '*'
85
+ "5": 1
86
+ "6":
87
+ - 1
88
+ "7": []
89
+ python_version: 3.11.15
90
+ t:
91
+ "1":
92
+ - 1
93
+ - 5
94
+ - 11
95
+ - 41
96
+ - 49
97
+ - 51
98
+ - 53
99
+ - 71
100
+ "2":
101
+ - 1
102
+ - 5
103
+ - 11
104
+ - 41
105
+ - 49
106
+ - 51
107
+ - 53
108
+ - 71
109
+ "3":
110
+ - 2
111
+ - 7
112
+ - 13
113
+ - 19
114
+ - 62
115
+ - 66
116
+ "4": 3.11.15
117
+ "5": 0.28.1
118
+ "6": 5.16.0.dev0
119
+ "9":
120
+ "1": transformers_trainer
121
+ "12": 0.28.1
122
+ "13": linux-x86_64
123
+ accelerator_config:
124
+ value:
125
+ dispatch_batches: null
126
+ even_batches: true
127
+ gradient_accumulation_kwargs: null
128
+ non_blocking: false
129
+ split_batches: false
130
+ use_seedable_sampler: true
131
+ activation:
132
+ value: linear
133
+ adam_beta1:
134
+ value: 0.9
135
+ adam_beta2:
136
+ value: 0.999
137
+ adam_epsilon:
138
+ value: 1e-08
139
+ architectures:
140
+ value: null
141
+ attention_bias:
142
+ value: false
143
+ attention_dropout:
144
+ value: 0
145
+ auto_find_batch_size:
146
+ value: false
147
+ average_tokens_across_devices:
148
+ value: true
149
+ batch_eval_metrics:
150
+ value: false
151
+ bf16:
152
+ value: true
153
+ bf16_full_eval:
154
+ value: false
155
+ bos_token_id:
156
+ value: 1
157
+ chunk_size_feed_forward:
158
+ value: 0
159
+ data_seed:
160
+ value: 42
161
+ dataloader_drop_last:
162
+ value: false
163
+ dataloader_in_order:
164
+ value: true
165
+ dataloader_multiprocessing_context:
166
+ value: null
167
+ dataloader_num_workers:
168
+ value: 0
169
+ dataloader_persistent_workers:
170
+ value: false
171
+ dataloader_pin_memory:
172
+ value: true
173
+ dataloader_prefetch_factor:
174
+ value: null
175
+ ddp_backend:
176
+ value: null
177
+ ddp_broadcast_buffers:
178
+ value: null
179
+ ddp_bucket_cap_mb:
180
+ value: null
181
+ ddp_find_unused_parameters:
182
+ value: null
183
+ ddp_static_graph:
184
+ value: null
185
+ ddp_timeout:
186
+ value: 1800
187
+ debug:
188
+ value: []
189
+ deepspeed:
190
+ value: null
191
+ disable_tqdm:
192
+ value: false
193
+ do_eval:
194
+ value: true
195
+ do_predict:
196
+ value: false
197
+ do_train:
198
+ value: false
199
+ dtype:
200
+ value: null
201
+ enable_jit_checkpoint:
202
+ value: false
203
+ eos_token_id:
204
+ value: 2
205
+ eval_accumulation_steps:
206
+ value: null
207
+ eval_delay:
208
+ value: 0
209
+ eval_do_concat_batches:
210
+ value: true
211
+ eval_on_start:
212
+ value: false
213
+ eval_steps:
214
+ value: 100
215
+ eval_strategy:
216
+ value: steps
217
+ eval_use_gather_object:
218
+ value: false
219
+ fp16:
220
+ value: false
221
+ fp16_full_eval:
222
+ value: false
223
+ fsdp:
224
+ value: null
225
+ fsdp_config:
226
+ value: null
227
+ full_determinism:
228
+ value: false
229
+ gradient_accumulation_steps:
230
+ value: 1
231
+ gradient_checkpointing:
232
+ value: false
233
+ gradient_checkpointing_kwargs:
234
+ value: null
235
+ greater_is_better:
236
+ value: null
237
+ head_dim:
238
+ value: 32
239
+ hidden_act:
240
+ value: silu
241
+ hidden_size:
242
+ value: 128
243
+ hub_always_push:
244
+ value: false
245
+ hub_model_id:
246
+ value: w-ahmad/6L-mlp-linear-9L
247
+ hub_private_repo:
248
+ value: null
249
+ hub_revision:
250
+ value: null
251
+ hub_strategy:
252
+ value: every_save
253
+ hub_token:
254
+ value: <HUB_TOKEN>
255
+ id2label:
256
+ value:
257
+ "0": LABEL_0
258
+ "1": LABEL_1
259
+ ignore_data_skip:
260
+ value: false
261
+ include_for_metrics:
262
+ value: []
263
+ include_num_input_tokens_seen:
264
+ value: "no"
265
+ initializer_range:
266
+ value: 0.02
267
+ intermediate_size:
268
+ value: 256
269
+ is_encoder_decoder:
270
+ value: false
271
+ label_names:
272
+ value: null
273
+ label_smoothing_factor:
274
+ value: 0
275
+ label2id:
276
+ value:
277
+ LABEL_0: 0
278
+ LABEL_1: 1
279
+ learning_rate:
280
+ value: 0.001
281
+ length_column_name:
282
+ value: length
283
+ liger_kernel_config:
284
+ value: null
285
+ load_best_model_at_end:
286
+ value: false
287
+ local_rank:
288
+ value: -1
289
+ log_level:
290
+ value: passive
291
+ log_level_replica:
292
+ value: warning
293
+ log_on_each_node:
294
+ value: true
295
+ logging_first_step:
296
+ value: false
297
+ logging_nan_inf_filter:
298
+ value: true
299
+ logging_steps:
300
+ value: 20
301
+ logging_strategy:
302
+ value: steps
303
+ lr_scheduler_kwargs:
304
+ value: null
305
+ lr_scheduler_type:
306
+ value: constant_with_warmup
307
+ max_grad_norm:
308
+ value: 1
309
+ max_position_embeddings:
310
+ value: 512
311
+ max_steps:
312
+ value: 1000
313
+ metric_for_best_model:
314
+ value: null
315
+ mlp_bias:
316
+ value: false
317
+ mlp_type:
318
+ value: mlp
319
+ model/num_parameters:
320
+ value: 2001280
321
+ model_type:
322
+ value: tiny_llama
323
+ neftune_noise_alpha:
324
+ value: null
325
+ num_attention_heads:
326
+ value: 4
327
+ num_hidden_layers:
328
+ value: 9
329
+ num_key_value_heads:
330
+ value: 4
331
+ num_train_epochs:
332
+ value: 1
333
+ optim:
334
+ value: adamw_torch_fused
335
+ optim_args:
336
+ value: null
337
+ optim_target_modules:
338
+ value: null
339
+ output_attentions:
340
+ value: false
341
+ output_dir:
342
+ value: out/mlp-linear-9L_run
343
+ output_hidden_states:
344
+ value: false
345
+ pad_token_id:
346
+ value: 0
347
+ parallelism_config:
348
+ value: null
349
+ per_device_eval_batch_size:
350
+ value: 1500
351
+ per_device_train_batch_size:
352
+ value: 80
353
+ powlu_m:
354
+ value: 3
355
+ prediction_loss_only:
356
+ value: false
357
+ pretraining_tp:
358
+ value: 1
359
+ problem_type:
360
+ value: null
361
+ project:
362
+ value: huggingface
363
+ push_to_hub:
364
+ value: false
365
+ remove_unused_columns:
366
+ value: false
367
+ report_to:
368
+ value:
369
+ - wandb
370
+ restore_callback_states_from_checkpoint:
371
+ value: false
372
+ resume_from_checkpoint:
373
+ value: null
374
+ return_dict:
375
+ value: true
376
+ rms_norm_eps:
377
+ value: 1e-06
378
+ rope_parameters:
379
+ value:
380
+ rope_theta: 10000
381
+ rope_type: default
382
+ run_name:
383
+ value: LM-mlp-linear-9L-2.0M-20260819-171540
384
+ save_on_each_node:
385
+ value: false
386
+ save_only_model:
387
+ value: false
388
+ save_steps:
389
+ value: 100
390
+ save_strategy:
391
+ value: steps
392
+ save_total_limit:
393
+ value: null
394
+ seed:
395
+ value: 42
396
+ skip_memory_metrics:
397
+ value: true
398
+ tf32:
399
+ value: null
400
+ tie_word_embeddings:
401
+ value: true
402
+ tokenizer_name:
403
+ value: w-ahmad/tiny-stories-tokenizer
404
+ torch_compile:
405
+ value: false
406
+ torch_compile_backend:
407
+ value: null
408
+ torch_compile_mode:
409
+ value: null
410
+ torch_empty_cache_steps:
411
+ value: null
412
+ trackio_bucket_id:
413
+ value: null
414
+ trackio_space_id:
415
+ value: null
416
+ trackio_static_space_id:
417
+ value: null
418
+ train_sampling_strategy:
419
+ value: random
420
+ transformers_version:
421
+ value: 5.16.0.dev0
422
+ use_cache:
423
+ value: false
424
+ use_cpu:
425
+ value: false
426
+ use_liger_kernel:
427
+ value: false
428
+ vocab_size:
429
+ value: 4096
430
+ waleed_beta:
431
+ value: 10
432
+ warmup_steps:
433
+ value: 200
434
+ weight_decay:
435
+ value: 0
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/output.log ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 0%| | 0/1000 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
2
+ [INFO] Causal mask (float with -inf) applied to all attention layers.
3
+ 10%|█ | 100/1000 [00:18<01:34, 9.55[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
4
+ {'loss': '8.201', 'grad_norm': '1.242', 'learning_rate': '9.5e-05', 'epoch': '0.001685', 'train/total_time_seconds': '1.182', 'train/time_per_step_avg': '0.0591', 'train/epoch_time_elapsed': '2.652', 'train/estimated_remaining_minutes': '0.9653'}
5
+ {'loss': '7.765', 'grad_norm': '1.242', 'learning_rate': '0.000195', 'epoch': '0.00337', 'train/total_time_seconds': '1.819', 'train/time_per_step_avg': '0.04547', 'train/epoch_time_elapsed': '4.584', 'train/estimated_remaining_minutes': '0.7275'}
6
+ {'loss': '7.16', 'grad_norm': '1.227', 'learning_rate': '0.000295', 'epoch': '0.005056', 'train/total_time_seconds': '2.454', 'train/time_per_step_avg': '0.0409', 'train/epoch_time_elapsed': '6.432', 'train/estimated_remaining_minutes': '0.6408'}
7
+ {'loss': '6.481', 'grad_norm': '1.211', 'learning_rate': '0.000395', 'epoch': '0.006741', 'train/total_time_seconds': '3.086', 'train/time_per_step_avg': '0.03858', 'train/epoch_time_elapsed': '8.246', 'train/estimated_remaining_minutes': '0.5916'}
8
+ {'loss': '5.919', 'grad_norm': '1.008', 'learning_rate': '0.000495', 'epoch': '0.008426', 'train/total_time_seconds': '3.722', 'train/time_per_step_avg': '0.03722', 'train/epoch_time_elapsed': '10.15', 'train/estimated_remaining_minutes': '0.5584'}
9
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
10
+ {'eval_loss': '5.634', 'eval_runtime': '7.97', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.008426', 'train/total_time_seconds': '3.722', 'train/time_per_step_avg': '0.03722', 'train/epoch_time_elapsed': '18.12', 'train/estimated_remaining_minutes': '0.5584'}
11
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
12
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
13
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 93.80it/s]
14
+ 20%|██ | 200/1000 [00:35<01:13, 10.84[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
15
+ {'loss': '5.413', 'grad_norm': '1.172', 'learning_rate': '0.000595', 'epoch': '0.01011', 'train/total_time_seconds': '4.364', 'train/time_per_step_avg': '0.03182', 'train/epoch_time_elapsed': '20.01', 'train/estimated_remaining_minutes': '0.5334'}
16
+ {'loss': '5.033', 'grad_norm': '1.516', 'learning_rate': '0.000695', 'epoch': '0.0118', 'train/total_time_seconds': '5.002', 'train/time_per_step_avg': '0.03183', 'train/epoch_time_elapsed': '21.84', 'train/estimated_remaining_minutes': '0.5121'}
17
+ {'loss': '4.695', 'grad_norm': '0.9766', 'learning_rate': '0.000795', 'epoch': '0.01348', 'train/total_time_seconds': '5.642', 'train/time_per_step_avg': '0.03187', 'train/epoch_time_elapsed': '23.68', 'train/estimated_remaining_minutes': '0.4936'}
18
+ {'loss': '4.425', 'grad_norm': '0.8828', 'learning_rate': '0.000895', 'epoch': '0.01517', 'train/total_time_seconds': '6.286', 'train/time_per_step_avg': '0.032', 'train/epoch_time_elapsed': '25.54', 'train/estimated_remaining_minutes': '0.4773'}
19
+ {'loss': '4.205', 'grad_norm': '0.8789', 'learning_rate': '0.000995', 'epoch': '0.01685', 'train/total_time_seconds': '6.934', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '27.38', 'train/estimated_remaining_minutes': '0.4623'}
20
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
21
+ {'eval_loss': '4.133', 'eval_runtime': '7.933', 'eval_samples_per_second': '1201', 'eval_steps_per_second': '0.882', 'epoch': '0.01685', 'train/total_time_seconds': '6.934', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '35.32', 'train/estimated_remaining_minutes': '0.4623'}
22
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
23
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
24
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 114.05it/s]
25
+ 30%|███ | 300/1000 [00:52<01:04, 10.91[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
26
+ {'loss': '4.049', 'grad_norm': '0.6875', 'learning_rate': '0.001', 'epoch': '0.01854', 'train/total_time_seconds': '7.582', 'train/time_per_step_avg': '0.03218', 'train/epoch_time_elapsed': '37.19', 'train/estimated_remaining_minutes': '0.448'}
27
+ {'loss': '3.919', 'grad_norm': '0.7383', 'learning_rate': '0.001', 'epoch': '0.02022', 'train/total_time_seconds': '8.228', 'train/time_per_step_avg': '0.03226', 'train/epoch_time_elapsed': '39.05', 'train/estimated_remaining_minutes': '0.4343'}
28
+ {'loss': '3.783', 'grad_norm': '0.6836', 'learning_rate': '0.001', 'epoch': '0.02191', 'train/total_time_seconds': '8.873', 'train/time_per_step_avg': '0.03231', 'train/epoch_time_elapsed': '40.89', 'train/estimated_remaining_minutes': '0.4209'}
29
+ {'loss': '3.701', 'grad_norm': '0.8242', 'learning_rate': '0.001', 'epoch': '0.02359', 'train/total_time_seconds': '9.516', 'train/time_per_step_avg': '0.0323', 'train/epoch_time_elapsed': '42.82', 'train/estimated_remaining_minutes': '0.4078'}
30
+ {'loss': '3.636', 'grad_norm': '0.9844', 'learning_rate': '0.001', 'epoch': '0.02528', 'train/total_time_seconds': '10.16', 'train/time_per_step_avg': '0.03223', 'train/epoch_time_elapsed': '44.65', 'train/estimated_remaining_minutes': '0.395'}
31
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
32
+ {'eval_loss': '3.598', 'eval_runtime': '7.973', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.02528', 'train/total_time_seconds': '10.16', 'train/time_per_step_avg': '0.03223', 'train/epoch_time_elapsed': '52.63', 'train/estimated_remaining_minutes': '0.395'}
33
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
34
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
35
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 170.80it/s]
36
+ 40%|████ | 400/1000 [01:10<00:56, 10.56[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
37
+ {'loss': '3.539', 'grad_norm': '0.8086', 'learning_rate': '0.001', 'epoch': '0.02696', 'train/total_time_seconds': '10.8', 'train/time_per_step_avg': '0.03215', 'train/epoch_time_elapsed': '54.51', 'train/estimated_remaining_minutes': '0.3824'}
38
+ {'loss': '3.492', 'grad_norm': '0.7227', 'learning_rate': '0.001', 'epoch': '0.02865', 'train/total_time_seconds': '11.44', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '56.36', 'train/estimated_remaining_minutes': '0.3701'}
39
+ {'loss': '3.443', 'grad_norm': '0.8203', 'learning_rate': '0.001', 'epoch': '0.03033', 'train/total_time_seconds': '12.08', 'train/time_per_step_avg': '0.03205', 'train/epoch_time_elapsed': '58.19', 'train/estimated_remaining_minutes': '0.3579'}
40
+ {'loss': '3.399', 'grad_norm': '0.8203', 'learning_rate': '0.001', 'epoch': '0.03202', 'train/total_time_seconds': '12.72', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '60.08', 'train/estimated_remaining_minutes': '0.3458'}
41
+ {'loss': '3.342', 'grad_norm': '0.6484', 'learning_rate': '0.001', 'epoch': '0.0337', 'train/total_time_seconds': '13.36', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '61.94', 'train/estimated_remaining_minutes': '0.334'}
42
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
43
+ {'eval_loss': '3.33', 'eval_runtime': '8.169', 'eval_samples_per_second': '1166', 'eval_steps_per_second': '0.857', 'epoch': '0.0337', 'train/total_time_seconds': '13.36', 'train/time_per_step_avg': '0.03201', 'train/epoch_time_elapsed': '70.11', 'train/estimated_remaining_minutes': '0.334'}
44
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
45
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
46
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 115.56it/s]
47
+ 50%|█████ | 500/1000 [01:27<00:46, 10.76[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
48
+ {'loss': '3.305', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.03539', 'train/total_time_seconds': '14', 'train/time_per_step_avg': '0.03204', 'train/epoch_time_elapsed': '72.01', 'train/estimated_remaining_minutes': '0.3222'}
49
+ {'loss': '3.259', 'grad_norm': '0.7188', 'learning_rate': '0.001', 'epoch': '0.03707', 'train/total_time_seconds': '14.65', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '73.95', 'train/estimated_remaining_minutes': '0.3108'}
50
+ {'loss': '3.229', 'grad_norm': '0.7305', 'learning_rate': '0.001', 'epoch': '0.03876', 'train/total_time_seconds': '15.3', 'train/time_per_step_avg': '0.0322', 'train/epoch_time_elapsed': '75.83', 'train/estimated_remaining_minutes': '0.2993'}
51
+ {'loss': '3.202', 'grad_norm': '0.7031', 'learning_rate': '0.001', 'epoch': '0.04044', 'train/total_time_seconds': '15.94', 'train/time_per_step_avg': '0.03219', 'train/epoch_time_elapsed': '77.68', 'train/estimated_remaining_minutes': '0.2877'}
52
+ {'loss': '3.164', 'grad_norm': '0.7266', 'learning_rate': '0.001', 'epoch': '0.04213', 'train/total_time_seconds': '16.57', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '79.6', 'train/estimated_remaining_minutes': '0.2762'}
53
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
54
+ {'eval_loss': '3.157', 'eval_runtime': '7.95', 'eval_samples_per_second': '1198', 'eval_steps_per_second': '0.88', 'epoch': '0.04213', 'train/total_time_seconds': '16.57', 'train/time_per_step_avg': '0.03212', 'train/epoch_time_elapsed': '87.56', 'train/estimated_remaining_minutes': '0.2762'}
55
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
56
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
57
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 173.21it/s]
58
+ 60%|██████ | 600/1000 [01:45<00:39, 10.00[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
59
+ {'loss': '3.145', 'grad_norm': '0.6953', 'learning_rate': '0.001', 'epoch': '0.04382', 'train/total_time_seconds': '17.2', 'train/time_per_step_avg': '0.03203', 'train/epoch_time_elapsed': '89.41', 'train/estimated_remaining_minutes': '0.2647'}
60
+ {'loss': '3.104', 'grad_norm': '0.8281', 'learning_rate': '0.001', 'epoch': '0.0455', 'train/total_time_seconds': '17.84', 'train/time_per_step_avg': '0.03186', 'train/epoch_time_elapsed': '91.23', 'train/estimated_remaining_minutes': '0.2533'}
61
+ {'loss': '3.084', 'grad_norm': '0.7578', 'learning_rate': '0.001', 'epoch': '0.04719', 'train/total_time_seconds': '18.47', 'train/time_per_step_avg': '0.03174', 'train/epoch_time_elapsed': '93.06', 'train/estimated_remaining_minutes': '0.2419'}
62
+ {'loss': '3.05', 'grad_norm': '0.7695', 'learning_rate': '0.001', 'epoch': '0.04887', 'train/total_time_seconds': '19.11', 'train/time_per_step_avg': '0.03169', 'train/epoch_time_elapsed': '94.88', 'train/estimated_remaining_minutes': '0.2306'}
63
+ {'loss': '3.038', 'grad_norm': '0.9688', 'learning_rate': '0.001', 'epoch': '0.05056', 'train/total_time_seconds': '19.75', 'train/time_per_step_avg': '0.0318', 'train/epoch_time_elapsed': '96.94', 'train/estimated_remaining_minutes': '0.2195'}
64
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
65
+ {'eval_loss': '3.031', 'eval_runtime': '8.12', 'eval_samples_per_second': '1173', 'eval_steps_per_second': '0.862', 'epoch': '0.05056', 'train/total_time_seconds': '19.75', 'train/time_per_step_avg': '0.0318', 'train/epoch_time_elapsed': '105.1', 'train/estimated_remaining_minutes': '0.2195'}
66
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
67
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
68
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 172.53it/s]
69
+ 70%|███████ | 700/1000 [02:02<00:27, 11.05[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
70
+ {'loss': '3.031', 'grad_norm': '0.8242', 'learning_rate': '0.001', 'epoch': '0.05224', 'train/total_time_seconds': '20.38', 'train/time_per_step_avg': '0.03179', 'train/epoch_time_elapsed': '106.9', 'train/estimated_remaining_minutes': '0.2082'}
71
+ {'loss': '3.001', 'grad_norm': '0.6406', 'learning_rate': '0.001', 'epoch': '0.05393', 'train/total_time_seconds': '21.01', 'train/time_per_step_avg': '0.03175', 'train/epoch_time_elapsed': '108.7', 'train/estimated_remaining_minutes': '0.197'}
72
+ {'loss': '2.996', 'grad_norm': '0.707', 'learning_rate': '0.001', 'epoch': '0.05561', 'train/total_time_seconds': '21.64', 'train/time_per_step_avg': '0.03173', 'train/epoch_time_elapsed': '110.6', 'train/estimated_remaining_minutes': '0.1858'}
73
+ {'loss': '2.957', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.0573', 'train/total_time_seconds': '22.27', 'train/time_per_step_avg': '0.03168', 'train/epoch_time_elapsed': '112.4', 'train/estimated_remaining_minutes': '0.1747'}
74
+ {'loss': '2.94', 'grad_norm': '0.6992', 'learning_rate': '0.001', 'epoch': '0.05898', 'train/total_time_seconds': '22.9', 'train/time_per_step_avg': '0.03153', 'train/epoch_time_elapsed': '114.2', 'train/estimated_remaining_minutes': '0.1636'}
75
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
76
+ {'eval_loss': '2.939', 'eval_runtime': '7.975', 'eval_samples_per_second': '1195', 'eval_steps_per_second': '0.878', 'epoch': '0.05898', 'train/total_time_seconds': '22.9', 'train/time_per_step_avg': '0.03153', 'train/epoch_time_elapsed': '122.2', 'train/estimated_remaining_minutes': '0.1636'}
77
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
78
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
79
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 171.02it/s]
80
+ 80%|████████ | 800/1000 [02:19<00:18, 10.93[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
81
+ {'loss': '2.923', 'grad_norm': '0.6484', 'learning_rate': '0.001', 'epoch': '0.06067', 'train/total_time_seconds': '23.54', 'train/time_per_step_avg': '0.03154', 'train/epoch_time_elapsed': '124.1', 'train/estimated_remaining_minutes': '0.1526'}
82
+ {'loss': '2.915', 'grad_norm': '0.957', 'learning_rate': '0.001', 'epoch': '0.06235', 'train/total_time_seconds': '24.17', 'train/time_per_step_avg': '0.03162', 'train/epoch_time_elapsed': '125.9', 'train/estimated_remaining_minutes': '0.1416'}
83
+ {'loss': '2.888', 'grad_norm': '0.7109', 'learning_rate': '0.001', 'epoch': '0.06404', 'train/total_time_seconds': '24.81', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '127.9', 'train/estimated_remaining_minutes': '0.1306'}
84
+ {'loss': '2.869', 'grad_norm': '0.6914', 'learning_rate': '0.001', 'epoch': '0.06572', 'train/total_time_seconds': '25.45', 'train/time_per_step_avg': '0.03176', 'train/epoch_time_elapsed': '129.7', 'train/estimated_remaining_minutes': '0.1196'}
85
+ {'loss': '2.879', 'grad_norm': '0.6836', 'learning_rate': '0.001', 'epoch': '0.06741', 'train/total_time_seconds': '26.09', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '131.6', 'train/estimated_remaining_minutes': '0.1087'}
86
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
87
+ {'eval_loss': '2.863', 'eval_runtime': '8.277', 'eval_samples_per_second': '1151', 'eval_steps_per_second': '0.846', 'epoch': '0.06741', 'train/total_time_seconds': '26.09', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '139.9', 'train/estimated_remaining_minutes': '0.1087'}
88
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
89
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
90
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 118.50it/s]
91
+ 90%|█████████ | 900/1000 [02:37<00:09, 10.89[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
92
+ {'loss': '2.85', 'grad_norm': '0.6797', 'learning_rate': '0.001', 'epoch': '0.06909', 'train/total_time_seconds': '26.72', 'train/time_per_step_avg': '0.03185', 'train/epoch_time_elapsed': '141.8', 'train/estimated_remaining_minutes': '0.09776'}
93
+ {'loss': '2.848', 'grad_norm': '0.6211', 'learning_rate': '0.001', 'epoch': '0.07078', 'train/total_time_seconds': '27.36', 'train/time_per_step_avg': '0.03182', 'train/epoch_time_elapsed': '143.6', 'train/estimated_remaining_minutes': '0.08685'}
94
+ {'loss': '2.834', 'grad_norm': '0.7305', 'learning_rate': '0.001', 'epoch': '0.07246', 'train/total_time_seconds': '27.99', 'train/time_per_step_avg': '0.03179', 'train/epoch_time_elapsed': '145.5', 'train/estimated_remaining_minutes': '0.07594'}
95
+ {'loss': '2.825', 'grad_norm': '0.6992', 'learning_rate': '0.001', 'epoch': '0.07415', 'train/total_time_seconds': '28.62', 'train/time_per_step_avg': '0.0317', 'train/epoch_time_elapsed': '147.3', 'train/estimated_remaining_minutes': '0.06505'}
96
+ {'loss': '2.785', 'grad_norm': '0.6875', 'learning_rate': '0.001', 'epoch': '0.07583', 'train/total_time_seconds': '29.25', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '149.1', 'train/estimated_remaining_minutes': '0.05418'}
97
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
98
+ {'eval_loss': '2.803', 'eval_runtime': '7.959', 'eval_samples_per_second': '1197', 'eval_steps_per_second': '0.879', 'epoch': '0.07583', 'train/total_time_seconds': '29.25', 'train/time_per_step_avg': '0.03166', 'train/epoch_time_elapsed': '157.1', 'train/estimated_remaining_minutes': '0.05418'}
99
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
100
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
101
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 84.10it/s]
102
+ 100%|██████████| 1000/1000 [02:54<00:00, 10.5[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
103
+ {'loss': '2.794', 'grad_norm': '0.6094', 'learning_rate': '0.001', 'epoch': '0.07752', 'train/total_time_seconds': '29.89', 'train/time_per_step_avg': '0.03165', 'train/epoch_time_elapsed': '159.1', 'train/estimated_remaining_minutes': '0.04331'}
104
+ {'loss': '2.789', 'grad_norm': '0.6719', 'learning_rate': '0.001', 'epoch': '0.0792', 'train/total_time_seconds': '30.52', 'train/time_per_step_avg': '0.03165', 'train/epoch_time_elapsed': '161', 'train/estimated_remaining_minutes': '0.03247'}
105
+ {'loss': '2.785', 'grad_norm': '0.582', 'learning_rate': '0.001', 'epoch': '0.08089', 'train/total_time_seconds': '31.16', 'train/time_per_step_avg': '0.03167', 'train/epoch_time_elapsed': '162.8', 'train/estimated_remaining_minutes': '0.02164'}
106
+ {'loss': '2.775', 'grad_norm': '0.6055', 'learning_rate': '0.001', 'epoch': '0.08257', 'train/total_time_seconds': '31.79', 'train/time_per_step_avg': '0.0317', 'train/epoch_time_elapsed': '164.6', 'train/estimated_remaining_minutes': '0.01081'}
107
+ {'loss': '2.754', 'grad_norm': '0.7812', 'learning_rate': '0.001', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '166.6', 'train/estimated_remaining_minutes': '0'}
108
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
109
+ {'eval_loss': '2.76', 'eval_runtime': '7.927', 'eval_samples_per_second': '1202', 'eval_steps_per_second': '0.883', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '174.5', 'train/estimated_remaining_minutes': '0'}
110
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
111
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
112
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 114.15it/s]
113
+ 100%|███���██████| 1000/1000 [02:54<00:00, 5.73it/s], ?it/s]
114
+ {'train_runtime': '175.4', 'train_samples_per_second': '456.1', 'train_steps_per_second': '5.701', 'train_loss': '3.692', 'epoch': '0.08426', 'train/total_time_seconds': '32.44', 'train/time_per_step_avg': '0.03184', 'train/epoch_time_elapsed': '174.6', 'train/estimated_remaining_minutes': '0'}
115
+ 100%|██████████| 7/7 [00:05<00:00, 1.22it/s]
116
+ [transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
117
+ - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
118
+ - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
119
+ - If you are not the owner of the model architecture class, please contact the model code owner to update it.
120
+ Writing model shards: 100%|██████████| 1/1 [00:00<00:00, 172.45it/s]
121
+ >>> FINISHED mlp-linear-9L successfully
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/requirements.txt ADDED
@@ -0,0 +1,149 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ asttokens==3.0.1
2
+ comm==0.2.3
3
+ debugpy==1.8.21
4
+ decorator==5.3.1
5
+ executing==2.2.1
6
+ nest-asyncio==1.6.0
7
+ parso==0.8.7
8
+ platformdirs==4.11.0
9
+ psutil==7.2.2
10
+ ptyprocess==0.7.0
11
+ pure_eval==0.2.3
12
+ Pygments==2.20.0
13
+ pyzmq==27.1.0
14
+ setuptools==83.0.0
15
+ six==1.17.0
16
+ tornado==6.5.7
17
+ traitlets==5.15.0
18
+ fsspec==2026.4.0
19
+ wcwidth==0.8.2
20
+ ipython_pygments_lexers==1.1.1
21
+ jedi==0.20.0
22
+ jupyter_core==5.9.1
23
+ matplotlib-inline==0.2.2
24
+ pexpect==4.9.0
25
+ prompt_toolkit==3.0.53
26
+ python-dateutil==2.9.0.post0
27
+ stack_data==0.6.3
28
+ wheel==0.47.0
29
+ jupyter_client==8.9.1
30
+ pip==26.1.2
31
+ ipython==9.15.0
32
+ ipykernel==7.2.0
33
+ threadpoolctl==3.6.0
34
+ pyparsing==3.3.2
35
+ typing_extensions==4.15.0
36
+ Jinja2==3.1.6
37
+ narwhals==2.24.0
38
+ kiwisolver==1.5.0
39
+ joblib==1.5.3
40
+ fonttools==4.63.0
41
+ cycler==0.12.1
42
+ scipy==1.17.1
43
+ pandas==3.0.5
44
+ contourpy==1.3.3
45
+ scikit-learn==1.9.0
46
+ matplotlib==3.11.1
47
+ urllib3==2.7.0
48
+ tqdm==4.70.0
49
+ idna==3.18
50
+ charset-normalizer==3.4.9
51
+ certifi==2026.7.22
52
+ requests==2.34.2
53
+ seaborn==0.13.2
54
+ uv==0.12.0
55
+ shellingham==1.5.4
56
+ mpmath==1.3.0
57
+ attrs==26.1.0
58
+ hf-xet==1.5.2
59
+ nvidia-nccl-cu12==2.21.5
60
+ MarkupSafe==3.0.3
61
+ regex==2026.7.19
62
+ importlib_metadata==9.0.0
63
+ httpcore==1.0.9
64
+ annotated-doc==0.0.5
65
+ multidict==6.7.1
66
+ aiohttp==3.14.3
67
+ aiosignal==1.4.0
68
+ xxhash==3.8.1
69
+ aiohappyeyeballs==2.7.1
70
+ mdurl==0.1.2
71
+ cuda-toolkit==13.0.3.0
72
+ networkx==3.6.1
73
+ PyYAML==6.0.3
74
+ nvidia-cufile==1.15.1.6
75
+ typer==0.27.0
76
+ torchaudio==2.6.0+cu124
77
+ rich==15.0.0
78
+ nvidia-cufft-cu12==11.2.1.3
79
+ h11==0.16.0
80
+ dill==0.4.1
81
+ cuda-pathfinder==1.6.0
82
+ filelock==3.29.0
83
+ nvidia-nvtx-cu12==12.4.127
84
+ httpx==0.28.1
85
+ anyio==4.14.2
86
+ numpy==2.4.4
87
+ yarl==1.24.5
88
+ click==8.4.2
89
+ triton==3.2.0
90
+ frozenlist==1.8.0
91
+ zipp==4.1.0
92
+ propcache==0.5.2
93
+ markdown-it-py==4.2.0
94
+ nvidia-cuda-runtime==13.0.96
95
+ cuda-bindings==13.3.1
96
+ nvidia-cuda-cupti==13.0.85
97
+ torch==2.6.0+cu124
98
+ multiprocess==0.70.19
99
+ pillow==12.2.0
100
+ transformers==5.16.0.dev0
101
+ wandb==0.28.1
102
+ nvidia-curand==10.4.0.35
103
+ sympy==1.13.1
104
+ nvidia-cusparse==12.6.3.3
105
+ nvidia-cuda-nvrtc==13.0.88
106
+ typing-inspection==0.4.2
107
+ nvidia-cusolver==12.0.4.66
108
+ nvidia-cufft==12.0.0.61
109
+ nvidia-cudnn-cu13==9.20.0.48
110
+ nvidia-cublas==13.1.1.3
111
+ pyarrow==25.0.0
112
+ evaluate==0.4.6
113
+ diffusers==0.39.0
114
+ pydantic==2.13.4
115
+ annotated-types==0.8.0
116
+ protobuf==7.35.1
117
+ sentry-sdk==2.66.1
118
+ einops==0.8.2
119
+ packaging==26.2
120
+ nvidia-nvjitlink-cu12==12.4.127
121
+ nvidia-curand-cu12==10.3.5.147
122
+ nvidia-cusparselt-cu12==0.6.2
123
+ nvidia-cusparse-cu12==12.3.1.170
124
+ nvidia-cuda-runtime-cu12==12.4.127
125
+ torchvision==0.21.0+cu124
126
+ nvidia-cuda-nvrtc-cu12==12.4.127
127
+ nvidia-cuda-cupti-cu12==12.4.127
128
+ nvidia-cusolver-cu12==11.6.1.9
129
+ nvidia-cublas-cu12==12.4.5.8
130
+ nvidia-cudnn-cu12==9.1.0.70
131
+ huggingface_hub==1.26.0
132
+ datasets==5.0.1
133
+ safetensors==0.8.0
134
+ accelerate==1.14.0
135
+ pydantic_core==2.46.4
136
+ ninja==1.13.0
137
+ tokenizers==0.23.1
138
+ autocommand==2.2.2
139
+ backports.tarfile==1.2.0
140
+ importlib_metadata==8.7.1
141
+ jaraco.text==4.0.0
142
+ jaraco.context==6.1.0
143
+ jaraco.functools==4.4.0
144
+ more-itertools==10.8.0
145
+ packaging==26.0
146
+ platformdirs==4.4.0
147
+ tomli==2.4.0
148
+ wheel==0.46.3
149
+ zipp==3.23.0
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-metadata.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35",
3
+ "python": "CPython 3.11.15",
4
+ "startedAt": "2026-08-19T17:15:41.408834Z",
5
+ "args": [
6
+ "--config",
7
+ "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/configs/baseline100L.yaml",
8
+ "--variants",
9
+ "mlp-linear-9L"
10
+ ],
11
+ "program": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/sweep.py",
12
+ "codePath": "sweep.py",
13
+ "codePathLocal": "sweep.py",
14
+ "git": {
15
+ "remote": "https://github.com/w-ahmad1a10/Activation.git",
16
+ "commit": "463c9961366755fc55f02df9a0d471b3cbcb025e"
17
+ },
18
+ "email": "deepnevro@gmail.com",
19
+ "root": "/mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation",
20
+ "host": "deeplens-k3s-node1",
21
+ "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python",
22
+ "cpu_count": 112,
23
+ "cpu_count_logical": 224,
24
+ "gpu": "NVIDIA H100 80GB HBM3",
25
+ "gpu_count": 8,
26
+ "disk": {
27
+ "/": {
28
+ "total": "1560765693952",
29
+ "used": "708485218304"
30
+ }
31
+ },
32
+ "memory": {
33
+ "total": "2164089937920"
34
+ },
35
+ "gpu_nvidia": [
36
+ {
37
+ "name": "NVIDIA H100 80GB HBM3",
38
+ "memoryTotal": "85520809984",
39
+ "cudaCores": 16896,
40
+ "architecture": "Hopper",
41
+ "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae"
42
+ },
43
+ {
44
+ "name": "NVIDIA H100 80GB HBM3",
45
+ "memoryTotal": "85520809984",
46
+ "cudaCores": 16896,
47
+ "architecture": "Hopper",
48
+ "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3"
49
+ },
50
+ {
51
+ "name": "NVIDIA H100 80GB HBM3",
52
+ "memoryTotal": "85520809984",
53
+ "cudaCores": 16896,
54
+ "architecture": "Hopper",
55
+ "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper",
62
+ "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864"
63
+ },
64
+ {
65
+ "name": "NVIDIA H100 80GB HBM3",
66
+ "memoryTotal": "85520809984",
67
+ "cudaCores": 16896,
68
+ "architecture": "Hopper",
69
+ "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef"
70
+ },
71
+ {
72
+ "name": "NVIDIA H100 80GB HBM3",
73
+ "memoryTotal": "85520809984",
74
+ "cudaCores": 16896,
75
+ "architecture": "Hopper",
76
+ "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54"
77
+ },
78
+ {
79
+ "name": "NVIDIA H100 80GB HBM3",
80
+ "memoryTotal": "85520809984",
81
+ "cudaCores": 16896,
82
+ "architecture": "Hopper",
83
+ "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9"
84
+ },
85
+ {
86
+ "name": "NVIDIA H100 80GB HBM3",
87
+ "memoryTotal": "85520809984",
88
+ "cudaCores": 16896,
89
+ "architecture": "Hopper",
90
+ "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea"
91
+ }
92
+ ],
93
+ "cudaVersion": "12.4",
94
+ "writerId": "2l4ndnmtei7frfwveku3d762xxak1bdj"
95
+ }
zain/Activation/wandb/run-20260819_171541-bhq23mh5/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"train/global_step":1000,"train_steps_per_second":5.701,"eval/samples_per_second":1206.607,"train/learning_rate":0.001,"eval/loss":2.7597591876983643,"eval/steps_per_second":0.887,"_runtime":182,"_wandb":{"runtime":182},"_timestamp":1.7871599245612671e+09,"eval/runtime":7.8957,"train/grad_norm":0.78125,"train/epoch":0.08426019548365352,"_step":61,"total_flos":3.6298555392e+14,"train_loss":3.6923015975952147,"train_runtime":175.4068,"train/loss":2.753749656677246,"train_samples_per_second":456.083}
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-core.log ADDED
@@ -0,0 +1,76 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-08-19T16:38:32.839280488Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpn7u591yp/port-3590110.txt","pid":3590110,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false}
2
+ {"time":"2026-08-19T16:38:32.840574816Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":3590110}
3
+ {"time":"2026-08-19T16:38:32.84057072Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-3590110-3591582-3840970873/socket","Net":"unix"}}
4
+ {"time":"2026-08-19T16:38:33.017601922Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"}
5
+ {"time":"2026-08-19T16:38:45.523223978Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"}
6
+ {"time":"2026-08-19T16:38:45.593765665Z","level":"INFO","msg":"handleInformInit: received","streamId":"74tq2syl","id":"2(@)"}
7
+ {"time":"2026-08-19T16:38:45.85943264Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"74tq2syl","id":"2(@)"}
8
+ {"time":"2026-08-19T16:38:51.253972428Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
9
+ {"time":"2026-08-19T16:43:55.859246567Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
10
+ {"time":"2026-08-19T16:43:56.562384948Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"s41hhf712g5d"}
11
+ {"time":"2026-08-19T16:43:56.564075521Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"74tq2syl","id":"2(@)"}
12
+ {"time":"2026-08-19T16:43:56.56470085Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"74tq2syl","id":"2(@)"}
13
+ {"time":"2026-08-19T16:43:58.189628272Z","level":"INFO","msg":"connection: closing","id":"2(@)"}
14
+ {"time":"2026-08-19T16:43:58.189706084Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"}
15
+ {"time":"2026-08-19T16:43:58.189644595Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"}
16
+ {"time":"2026-08-19T16:43:58.189713806Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"}
17
+ {"time":"2026-08-19T16:45:34.355733713Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"}
18
+ {"time":"2026-08-19T16:45:34.483865974Z","level":"INFO","msg":"handleInformInit: received","streamId":"glh82iyh","id":"3(@)"}
19
+ {"time":"2026-08-19T16:45:34.744778959Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"glh82iyh","id":"3(@)"}
20
+ {"time":"2026-08-19T16:45:40.185561802Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
21
+ {"time":"2026-08-19T16:45:40.195891916Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
22
+ {"time":"2026-08-19T16:45:40.960129401Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"d1gg1cuklin0"}
23
+ {"time":"2026-08-19T16:45:40.961603019Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"glh82iyh","id":"3(@)"}
24
+ {"time":"2026-08-19T16:45:40.96217202Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"glh82iyh","id":"3(@)"}
25
+ {"time":"2026-08-19T16:45:42.643090062Z","level":"INFO","msg":"connection: closing","id":"3(@)"}
26
+ {"time":"2026-08-19T16:45:42.643171917Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"}
27
+ {"time":"2026-08-19T16:45:42.643096466Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"}
28
+ {"time":"2026-08-19T16:45:42.643183903Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"}
29
+ {"time":"2026-08-19T16:45:53.775307513Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"}
30
+ {"time":"2026-08-19T16:45:53.86012332Z","level":"INFO","msg":"handleInformInit: received","streamId":"44x03ghu","id":"4(@)"}
31
+ {"time":"2026-08-19T16:45:54.119528402Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"44x03ghu","id":"4(@)"}
32
+ {"time":"2026-08-19T16:45:59.764320578Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
33
+ {"time":"2026-08-19T16:50:40.211693707Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
34
+ {"time":"2026-08-19T16:50:40.749569572Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"u6tmkghavygl"}
35
+ {"time":"2026-08-19T16:50:40.751009589Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"44x03ghu","id":"4(@)"}
36
+ {"time":"2026-08-19T16:50:40.751589921Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"44x03ghu","id":"4(@)"}
37
+ {"time":"2026-08-19T16:50:42.667568611Z","level":"INFO","msg":"connection: closing","id":"4(@)"}
38
+ {"time":"2026-08-19T16:50:42.66767627Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"}
39
+ {"time":"2026-08-19T16:50:42.667582108Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"}
40
+ {"time":"2026-08-19T16:50:42.667687704Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"}
41
+ {"time":"2026-08-19T17:08:54.476108098Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"}
42
+ {"time":"2026-08-19T17:08:54.716542574Z","level":"INFO","msg":"handleInformInit: received","streamId":"zhg4u13t","id":"5(@)"}
43
+ {"time":"2026-08-19T17:08:54.983870419Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"zhg4u13t","id":"5(@)"}
44
+ {"time":"2026-08-19T17:09:00.342597647Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
45
+ {"time":"2026-08-19T17:11:36.384264189Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
46
+ {"time":"2026-08-19T17:11:36.932623627Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"qy20ie44zvce"}
47
+ {"time":"2026-08-19T17:11:36.934203156Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"zhg4u13t","id":"5(@)"}
48
+ {"time":"2026-08-19T17:11:36.934659413Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"zhg4u13t","id":"5(@)"}
49
+ {"time":"2026-08-19T17:11:38.381994049Z","level":"INFO","msg":"connection: closing","id":"5(@)"}
50
+ {"time":"2026-08-19T17:11:38.382078588Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"}
51
+ {"time":"2026-08-19T17:11:38.382003656Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"}
52
+ {"time":"2026-08-19T17:11:38.382088954Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"}
53
+ {"time":"2026-08-19T17:12:15.205004959Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"6(@)"}
54
+ {"time":"2026-08-19T17:12:15.319455303Z","level":"INFO","msg":"handleInformInit: received","streamId":"2nil4rqn","id":"6(@)"}
55
+ {"time":"2026-08-19T17:12:15.582732337Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"2nil4rqn","id":"6(@)"}
56
+ {"time":"2026-08-19T17:12:21.153010638Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
57
+ {"time":"2026-08-19T17:14:57.367689028Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
58
+ {"time":"2026-08-19T17:14:58.279859891Z","level":"INFO","msg":"connection: cancelling request","id":"6(@)","requestId":"g8ym1re70rep"}
59
+ {"time":"2026-08-19T17:14:58.28171116Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"2nil4rqn","id":"6(@)"}
60
+ {"time":"2026-08-19T17:14:58.282403203Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"2nil4rqn","id":"6(@)"}
61
+ {"time":"2026-08-19T17:15:00.260323511Z","level":"INFO","msg":"connection: closing","id":"6(@)"}
62
+ {"time":"2026-08-19T17:15:00.26039463Z","level":"INFO","msg":"connection: closed successfully","id":"6(@)"}
63
+ {"time":"2026-08-19T17:15:00.260332968Z","level":"INFO","msg":"processOutgoingData: finished","id":"6(@)"}
64
+ {"time":"2026-08-19T17:15:00.260403826Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"6(@)"}
65
+ {"time":"2026-08-19T17:15:41.262714425Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"7(@)"}
66
+ {"time":"2026-08-19T17:15:41.411904785Z","level":"INFO","msg":"handleInformInit: received","streamId":"bhq23mh5","id":"7(@)"}
67
+ {"time":"2026-08-19T17:15:41.67518661Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"bhq23mh5","id":"7(@)"}
68
+ {"time":"2026-08-19T17:15:47.115142635Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
69
+ {"time":"2026-08-19T17:18:44.577098246Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
70
+ {"time":"2026-08-19T17:18:45.155577147Z","level":"INFO","msg":"connection: cancelling request","id":"7(@)","requestId":"dp9grmbn4dao"}
71
+ {"time":"2026-08-19T17:18:45.157199333Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"bhq23mh5","id":"7(@)"}
72
+ {"time":"2026-08-19T17:18:45.162184308Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"bhq23mh5","id":"7(@)"}
73
+ {"time":"2026-08-19T17:18:47.12414285Z","level":"INFO","msg":"processOutgoingData: finished","id":"7(@)"}
74
+ {"time":"2026-08-19T17:18:47.124138531Z","level":"INFO","msg":"connection: closing","id":"7(@)"}
75
+ {"time":"2026-08-19T17:18:47.124233886Z","level":"INFO","msg":"connection: closed successfully","id":"7(@)"}
76
+ {"time":"2026-08-19T17:18:47.124238443Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"7(@)"}
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-08-19T17:15:41.412064429Z","level":"INFO","msg":"wandb-core"}
2
+ {"time":"2026-08-19T17:15:41.41218811Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"}
3
+ {"time":"2026-08-19T17:15:41.674976743Z","level":"INFO","msg":"stream: created new stream","id":"bhq23mh5"}
4
+ {"time":"2026-08-19T17:15:41.675059529Z","level":"INFO","msg":"handler: started"}
5
+ {"time":"2026-08-19T17:15:41.675177927Z","level":"INFO","msg":"stream: started"}
6
+ {"time":"2026-08-19T17:15:41.675184514Z","level":"INFO","msg":"writer: started","stream_id":"bhq23mh5"}
7
+ {"time":"2026-08-19T17:15:41.675204094Z","level":"INFO","msg":"sender: started"}
8
+ {"time":"2026-08-19T17:15:42.093210867Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1}
9
+ {"time":"2026-08-19T17:15:42.190260489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
10
+ {"time":"2026-08-19T17:15:57.093327808Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":5,"events_offset":0,"events_lines":1,"console_offset":0,"console_lines":9,"uploaded_len":2}
11
+ {"time":"2026-08-19T17:15:57.208750755Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
12
+ {"time":"2026-08-19T17:16:12.09424474Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":6,"events_offset":1,"events_lines":2,"console_offset":2,"console_lines":1}
13
+ {"time":"2026-08-19T17:16:12.254954463Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
14
+ {"time":"2026-08-19T17:16:27.093741092Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":6,"events_offset":3,"events_lines":2,"console_offset":8,"console_lines":22}
15
+ {"time":"2026-08-19T17:16:27.218480793Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
16
+ {"time":"2026-08-19T17:16:42.09555922Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":4,"events_offset":5,"events_lines":2,"console_offset":24,"console_lines":1}
17
+ {"time":"2026-08-19T17:16:42.264597826Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
18
+ {"time":"2026-08-19T17:16:57.09386987Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":5,"events_offset":7,"events_lines":2,"console_offset":30,"console_lines":19}
19
+ {"time":"2026-08-19T17:16:57.209800875Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
20
+ {"time":"2026-08-19T17:17:12.094049325Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":5,"events_offset":9,"events_lines":2,"console_offset":46,"console_lines":1}
21
+ {"time":"2026-08-19T17:17:12.252024387Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
22
+ {"time":"2026-08-19T17:17:27.093470433Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":4,"events_offset":11,"events_lines":2,"console_offset":49,"console_lines":15}
23
+ {"time":"2026-08-19T17:17:27.202702289Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
24
+ {"time":"2026-08-19T17:17:42.093533111Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":6,"events_offset":13,"events_lines":2,"console_offset":57,"console_lines":1}
25
+ {"time":"2026-08-19T17:17:42.240909652Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
26
+ {"time":"2026-08-19T17:17:57.093850562Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":6,"events_offset":15,"events_lines":2,"console_offset":63,"console_lines":23}
27
+ {"time":"2026-08-19T17:17:57.20571167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
28
+ {"time":"2026-08-19T17:18:12.094059821Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":6,"events_offset":17,"events_lines":2,"console_offset":79,"console_lines":1}
29
+ {"time":"2026-08-19T17:18:12.283285285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
30
+ {"time":"2026-08-19T17:18:27.0939628Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":5,"events_offset":19,"events_lines":2,"console_offset":85,"console_lines":21}
31
+ {"time":"2026-08-19T17:18:27.220681676Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
32
+ {"time":"2026-08-19T17:18:42.09374958Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":3,"events_offset":21,"events_lines":2,"console_offset":101,"console_lines":1}
33
+ {"time":"2026-08-19T17:18:42.218715049Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
34
+ {"time":"2026-08-19T17:18:45.014142777Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
35
+ {"time":"2026-08-19T17:18:45.014366349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":23,"events_lines":1,"console_offset":106,"console_lines":15,"uploaded_len":3,"complete":true,"exit_code":0}
36
+ {"time":"2026-08-19T17:18:45.153108654Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"}
37
+ {"time":"2026-08-19T17:18:45.154305649Z","level":"INFO","msg":"handler: operation stats","stats":{}}
38
+ {"time":"2026-08-19T17:18:45.157232952Z","level":"INFO","msg":"stream: finishing up"}
39
+ {"time":"2026-08-19T17:18:45.157265556Z","level":"INFO","msg":"handler: closed"}
40
+ {"time":"2026-08-19T17:18:45.161743858Z","level":"INFO","msg":"sender: closed"}
41
+ {"time":"2026-08-19T17:18:45.161763043Z","level":"INFO","msg":"stream: all finished"}
zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1
2
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Configure stats pid to 3953242
3
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_setup.py:_flush():81] Loading settings from environment variables
4
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug.log
5
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/zain/Activation/wandb/run-20260819_171541-bhq23mh5/logs/debug-internal.log
6
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():772] calling init triggers
7
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():777] wandb.init called with sweep_config: {}
8
+ config: {'_wandb': {}}
9
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():820] starting backend
10
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE
11
+ 2026-08-19 17:15:41,410 INFO MainThread:3953242 [wandb_init.py:init():835] sending inform_init request
12
+ 2026-08-19 17:15:41,675 INFO MainThread:3953242 [wandb_init.py:init():840] backend started and connected
13
+ 2026-08-19 17:15:41,679 INFO MainThread:3953242 [wandb_init.py:init():910] updated telemetry
14
+ 2026-08-19 17:15:41,686 INFO MainThread:3953242 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout
15
+ 2026-08-19 17:15:42,011 INFO MainThread:3953242 [wandb_init.py:init():978] starting run threads in backend
16
+ 2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_console_start():2621] atexit reg
17
+ 2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2471] redirect: wrap_raw
18
+ 2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2540] Wrapping output streams.
19
+ 2026-08-19 17:15:42,086 INFO MainThread:3953242 [wandb_run.py:_redirect():2563] Redirects installed.
20
+ 2026-08-19 17:15:42,089 INFO MainThread:3953242 [wandb_init.py:init():1016] run started, returning control to user process
21
+ 2026-08-19 17:15:42,090 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.16.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'waleed_beta': 10.0, 'powlu_m': 3.0, 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 1000, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant_with_warmup', 'lr_scheduler_kwargs': None, 'warmup_steps': 200, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.0, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 1, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260819-171540', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 100, 'eval_delay': 0, 'per_device_eval_batch_size': 1500, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': False, 'hub_token': '<HUB_TOKEN>', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/6L-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1}
22
+ 2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - <bound method Run._config_callback of <wandb.sdk.wandb_run.Run object at 0x15079cfad610>>
23
+ 2026-08-19 17:15:42,091 INFO MainThread:3953242 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None
24
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/research-ultimate-checking/bhq23mh5
25
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0
26
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2570] restore
27
+ 2026-08-19 17:18:44,576 INFO MainThread:3953242 [wandb_run.py:_restore():2576] restore done
28
+ 2026-08-19 17:18:45,156 INFO MainThread:3953242 [wandb_run.py:_footer_sync_info():3993] logging synced files
zain/Activation/wandb/run-20260819_171541-bhq23mh5/run-bhq23mh5.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:896db157ba2ca29ceb3d6b02e8ae3f8c75a1864cc416843cde980ad34c0a27a7
3
+ size 251327