w-ahmad commited on
Commit
393c89f
·
verified ·
1 Parent(s): 99243f9

Auto upload zain 2026-08-18T15:43:05.347262

Browse files
zain/Activation/out/mlp-gelu-9L_run/checkpoint-600/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4500605646f57ecbaab3673eee29e548e0f21b262ac52c8de486c2731674412e
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2e975634df489a36f32879561ecc0bc3604a638ee0d3042123eb8170cbc3fb3f
3
  size 4010544
zain/Activation/out/mlp-gelu-9L_run/checkpoint-600/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9ca8629aa0ae12792a389d2b05653c5d547427e3fb44e9d88a052e21869c803d
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a82464792de05e600d7649e002ffc0a88f8abb9a3c597b9a161bb66426a7092b
3
  size 8068282
zain/Activation/out/mlp-gelu-9L_run/checkpoint-600/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43a21235ddd7c8c0020d887652dce105923f094e2bc54766bffa71522037a007
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:295d5fdc0e0bbae04d3db91b9c6051a7cccac49c004495947e95dcd6c59adda3
3
  size 1064
zain/Activation/out/mlp-gelu-9L_run/checkpoint-600/trainer_state.json CHANGED
@@ -11,260 +11,260 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.8984375,
15
- "learning_rate": 2.66e-05,
16
- "loss": 8.290375518798829,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.3125,
22
- "learning_rate": 5.46e-05,
23
- "loss": 8.103549194335937,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.2734375,
29
- "learning_rate": 8.259999999999999e-05,
30
- "loss": 7.879036712646484,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
- "grad_norm": 1.2734375,
36
- "learning_rate": 0.0001106,
37
- "loss": 7.590550231933594,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
- "learning_rate": 0.0001386,
44
- "loss": 7.234850311279297,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
- "eval_loss": 7.028295993804932,
50
- "eval_runtime": 7.9767,
51
- "eval_samples_per_second": 1194.359,
52
- "eval_steps_per_second": 1.254,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
- "grad_norm": 1.7421875,
58
- "learning_rate": 0.00016659999999999998,
59
- "loss": 6.841970825195313,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.1484375,
65
- "learning_rate": 0.00019460000000000001,
66
- "loss": 6.4576164245605465,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
- "grad_norm": 1.0859375,
72
- "learning_rate": 0.0002226,
73
- "loss": 6.111285781860351,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
- "grad_norm": 1.109375,
79
- "learning_rate": 0.00025059999999999997,
80
- "loss": 5.793069076538086,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
- "grad_norm": 0.85546875,
86
- "learning_rate": 0.0002786,
87
- "loss": 5.492683410644531,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
- "eval_loss": 5.342745304107666,
93
- "eval_runtime": 7.9623,
94
- "eval_samples_per_second": 1196.508,
95
- "eval_steps_per_second": 1.256,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
- "grad_norm": 0.87890625,
101
- "learning_rate": 0.00030659999999999997,
102
- "loss": 5.233791732788086,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
- "grad_norm": 0.78515625,
108
- "learning_rate": 0.0003346,
109
- "loss": 4.977091598510742,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
- "grad_norm": 0.87890625,
115
- "learning_rate": 0.00036260000000000003,
116
- "loss": 4.77899055480957,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
- "grad_norm": 0.91015625,
122
- "learning_rate": 0.0003906,
123
- "loss": 4.601796722412109,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
- "grad_norm": 0.87109375,
129
- "learning_rate": 0.0004186,
130
- "loss": 4.450910949707032,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
- "eval_loss": 4.373648643493652,
136
- "eval_runtime": 7.9961,
137
- "eval_samples_per_second": 1191.458,
138
- "eval_steps_per_second": 1.251,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
- "grad_norm": 1.078125,
144
- "learning_rate": 0.0004466,
145
- "loss": 4.315410232543945,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
- "grad_norm": 0.80078125,
151
- "learning_rate": 0.00047460000000000004,
152
- "loss": 4.23107795715332,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
- "grad_norm": 0.91015625,
158
- "learning_rate": 0.0005026,
159
- "loss": 4.112172698974609,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
- "grad_norm": 0.8828125,
165
- "learning_rate": 0.0005306,
166
- "loss": 4.030664443969727,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
- "grad_norm": 0.93359375,
172
- "learning_rate": 0.0005586,
173
- "loss": 3.95709228515625,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
- "eval_loss": 3.9225285053253174,
179
- "eval_runtime": 8.1083,
180
- "eval_samples_per_second": 1174.974,
181
- "eval_steps_per_second": 1.233,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
- "grad_norm": 1.0078125,
187
- "learning_rate": 0.0005866,
188
- "loss": 3.8780284881591798,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
- "grad_norm": 1.0078125,
194
- "learning_rate": 0.0006146,
195
- "loss": 3.837977981567383,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
- "grad_norm": 1.171875,
201
- "learning_rate": 0.0006426,
202
- "loss": 3.777029800415039,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
- "grad_norm": 0.79296875,
208
- "learning_rate": 0.0006705999999999999,
209
- "loss": 3.6980987548828126,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
- "grad_norm": 0.89453125,
215
- "learning_rate": 0.0006986,
216
- "loss": 3.6566497802734377,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
- "eval_loss": 3.6329572200775146,
222
- "eval_runtime": 7.8721,
223
- "eval_samples_per_second": 1210.221,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
- "grad_norm": 0.8984375,
230
- "learning_rate": 0.0007,
231
- "loss": 3.620241165161133,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
- "grad_norm": 1.140625,
237
- "learning_rate": 0.0007,
238
- "loss": 3.5605819702148436,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
- "grad_norm": 0.84375,
244
- "learning_rate": 0.0007,
245
- "loss": 3.515232467651367,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
- "grad_norm": 0.9609375,
251
- "learning_rate": 0.0007,
252
- "loss": 3.472977066040039,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
- "grad_norm": 0.8828125,
258
- "learning_rate": 0.0007,
259
- "loss": 3.442094421386719,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
- "eval_loss": 3.4318618774414062,
265
- "eval_runtime": 7.8952,
266
- "eval_samples_per_second": 1206.68,
267
- "eval_steps_per_second": 1.267,
268
  "step": 600
269
  }
270
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
+ "grad_norm": 2.328125,
15
+ "learning_rate": 1.14e-05,
16
+ "loss": 8.309149932861327,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.734375,
22
+ "learning_rate": 2.34e-05,
23
+ "loss": 8.2459228515625,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
+ "grad_norm": 1.296875,
29
+ "learning_rate": 3.539999999999999e-05,
30
+ "loss": 8.09324951171875,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
+ "grad_norm": 1.28125,
36
+ "learning_rate": 4.7399999999999993e-05,
37
+ "loss": 7.956021881103515,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
+ "learning_rate": 5.94e-05,
44
+ "loss": 7.7899658203125,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
+ "eval_loss": 7.690263748168945,
50
+ "eval_runtime": 7.9534,
51
+ "eval_samples_per_second": 1197.859,
52
+ "eval_steps_per_second": 1.257,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
+ "grad_norm": 1.25,
58
+ "learning_rate": 7.139999999999999e-05,
59
+ "loss": 7.591774749755859,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
+ "grad_norm": 1.265625,
65
+ "learning_rate": 8.34e-05,
66
+ "loss": 7.362733459472656,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
+ "grad_norm": 1.2734375,
72
+ "learning_rate": 9.539999999999999e-05,
73
+ "loss": 7.124365234375,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
+ "grad_norm": 1.2265625,
79
+ "learning_rate": 0.00010739999999999998,
80
+ "loss": 6.878346252441406,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
+ "grad_norm": 1.8359375,
86
+ "learning_rate": 0.0001194,
87
+ "loss": 6.63219985961914,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
+ "eval_loss": 6.522833824157715,
93
+ "eval_runtime": 7.8828,
94
+ "eval_samples_per_second": 1208.576,
95
+ "eval_steps_per_second": 1.269,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
+ "grad_norm": 1.703125,
101
+ "learning_rate": 0.0001314,
102
+ "loss": 6.431203460693359,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
+ "grad_norm": 1.1796875,
108
+ "learning_rate": 0.0001434,
109
+ "loss": 6.207939529418946,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
+ "grad_norm": 1.125,
115
+ "learning_rate": 0.00015539999999999998,
116
+ "loss": 5.998645782470703,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
+ "grad_norm": 1.0625,
122
+ "learning_rate": 0.0001674,
123
+ "loss": 5.795352172851563,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
+ "grad_norm": 0.93359375,
129
+ "learning_rate": 0.00017939999999999997,
130
+ "loss": 5.597444915771485,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
+ "eval_loss": 5.480377674102783,
136
+ "eval_runtime": 7.8653,
137
+ "eval_samples_per_second": 1211.274,
138
+ "eval_steps_per_second": 1.271,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
+ "grad_norm": 0.94140625,
144
+ "learning_rate": 0.0001914,
145
+ "loss": 5.3849952697753904,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
+ "grad_norm": 0.98828125,
151
+ "learning_rate": 0.00020339999999999998,
152
+ "loss": 5.2166282653808596,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
+ "grad_norm": 1.078125,
158
+ "learning_rate": 0.00021539999999999998,
159
+ "loss": 5.036893081665039,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
+ "grad_norm": 1.2109375,
165
+ "learning_rate": 0.00022739999999999997,
166
+ "loss": 4.901935195922851,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
+ "grad_norm": 0.984375,
172
+ "learning_rate": 0.0002394,
173
+ "loss": 4.777506256103516,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
+ "eval_loss": 4.718365669250488,
179
+ "eval_runtime": 7.988,
180
+ "eval_samples_per_second": 1192.66,
181
+ "eval_steps_per_second": 1.252,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
+ "grad_norm": 1.2109375,
187
+ "learning_rate": 0.0002514,
188
+ "loss": 4.656603622436523,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
+ "grad_norm": 1.71875,
194
+ "learning_rate": 0.00026339999999999995,
195
+ "loss": 4.587258148193359,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
+ "grad_norm": 1.21875,
201
+ "learning_rate": 0.00027539999999999997,
202
+ "loss": 4.507755279541016,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
+ "grad_norm": 1.25,
208
+ "learning_rate": 0.00028739999999999994,
209
+ "loss": 4.410206985473633,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
+ "grad_norm": 1.3203125,
215
+ "learning_rate": 0.00029939999999999996,
216
+ "loss": 4.3579551696777346,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
+ "eval_loss": 4.317867755889893,
222
+ "eval_runtime": 7.8743,
223
+ "eval_samples_per_second": 1209.882,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
+ "grad_norm": 1.1484375,
230
+ "learning_rate": 0.0003,
231
+ "loss": 4.301048278808594,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
+ "grad_norm": 1.3046875,
237
+ "learning_rate": 0.0003,
238
+ "loss": 4.22619743347168,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
+ "grad_norm": 1.1328125,
244
+ "learning_rate": 0.0003,
245
+ "loss": 4.173627090454102,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
+ "grad_norm": 0.97265625,
251
+ "learning_rate": 0.0003,
252
+ "loss": 4.122076797485351,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
+ "grad_norm": 1.2109375,
258
+ "learning_rate": 0.0003,
259
+ "loss": 4.078117370605469,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
+ "eval_loss": 4.0582146644592285,
265
+ "eval_runtime": 7.9148,
266
+ "eval_samples_per_second": 1203.7,
267
+ "eval_steps_per_second": 1.263,
268
  "step": 600
269
  }
270
  ],
zain/Activation/out/mlp-gelu-9L_run/checkpoint-600/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:79e0def26beb7a751361df77096e4e9d429ab64974840e7eb3dcf84a49f4a583
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:282bf64fd9cb5669a85393aac3b68bd51ade27ff8218886467eed07f2725602f
3
  size 4920
zain/Activation/out/mlp-gelu-9L_run/checkpoint-700/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:daf72d92b87327e97db75c0c6c47e7b61fdeea058860c623997d66d12c7264c7
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:13d7540500b76ad43155714e067c3a713d09b1ea639ba52982f613513c3be914
3
  size 4010544
zain/Activation/out/mlp-gelu-9L_run/checkpoint-700/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b83926bf5585093d371040082feb25ee0e434be1c9b62aa3ae0118705382a737
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fac036442baf982ac1611f727ee849a27fe6b26708473bf3f744affbfa5b74c3
3
  size 8068282
zain/Activation/out/mlp-gelu-9L_run/checkpoint-700/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8aac662062525611dd137fd05108444d17dfbd69f137e40e8a3651bb69d2ba7e
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6c38b4e6128ebabd81a22bafe942090015968e75f90e5e9ef25cd5ebdef8357
3
  size 1064
zain/Activation/out/mlp-gelu-9L_run/checkpoint-700/trainer_state.json CHANGED
@@ -11,303 +11,303 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.8984375,
15
- "learning_rate": 2.66e-05,
16
- "loss": 8.290375518798829,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.3125,
22
- "learning_rate": 5.46e-05,
23
- "loss": 8.103549194335937,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.2734375,
29
- "learning_rate": 8.259999999999999e-05,
30
- "loss": 7.879036712646484,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
- "grad_norm": 1.2734375,
36
- "learning_rate": 0.0001106,
37
- "loss": 7.590550231933594,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
- "learning_rate": 0.0001386,
44
- "loss": 7.234850311279297,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
- "eval_loss": 7.028295993804932,
50
- "eval_runtime": 7.9767,
51
- "eval_samples_per_second": 1194.359,
52
- "eval_steps_per_second": 1.254,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
- "grad_norm": 1.7421875,
58
- "learning_rate": 0.00016659999999999998,
59
- "loss": 6.841970825195313,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.1484375,
65
- "learning_rate": 0.00019460000000000001,
66
- "loss": 6.4576164245605465,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
- "grad_norm": 1.0859375,
72
- "learning_rate": 0.0002226,
73
- "loss": 6.111285781860351,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
- "grad_norm": 1.109375,
79
- "learning_rate": 0.00025059999999999997,
80
- "loss": 5.793069076538086,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
- "grad_norm": 0.85546875,
86
- "learning_rate": 0.0002786,
87
- "loss": 5.492683410644531,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
- "eval_loss": 5.342745304107666,
93
- "eval_runtime": 7.9623,
94
- "eval_samples_per_second": 1196.508,
95
- "eval_steps_per_second": 1.256,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
- "grad_norm": 0.87890625,
101
- "learning_rate": 0.00030659999999999997,
102
- "loss": 5.233791732788086,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
- "grad_norm": 0.78515625,
108
- "learning_rate": 0.0003346,
109
- "loss": 4.977091598510742,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
- "grad_norm": 0.87890625,
115
- "learning_rate": 0.00036260000000000003,
116
- "loss": 4.77899055480957,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
- "grad_norm": 0.91015625,
122
- "learning_rate": 0.0003906,
123
- "loss": 4.601796722412109,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
- "grad_norm": 0.87109375,
129
- "learning_rate": 0.0004186,
130
- "loss": 4.450910949707032,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
- "eval_loss": 4.373648643493652,
136
- "eval_runtime": 7.9961,
137
- "eval_samples_per_second": 1191.458,
138
- "eval_steps_per_second": 1.251,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
- "grad_norm": 1.078125,
144
- "learning_rate": 0.0004466,
145
- "loss": 4.315410232543945,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
- "grad_norm": 0.80078125,
151
- "learning_rate": 0.00047460000000000004,
152
- "loss": 4.23107795715332,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
- "grad_norm": 0.91015625,
158
- "learning_rate": 0.0005026,
159
- "loss": 4.112172698974609,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
- "grad_norm": 0.8828125,
165
- "learning_rate": 0.0005306,
166
- "loss": 4.030664443969727,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
- "grad_norm": 0.93359375,
172
- "learning_rate": 0.0005586,
173
- "loss": 3.95709228515625,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
- "eval_loss": 3.9225285053253174,
179
- "eval_runtime": 8.1083,
180
- "eval_samples_per_second": 1174.974,
181
- "eval_steps_per_second": 1.233,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
- "grad_norm": 1.0078125,
187
- "learning_rate": 0.0005866,
188
- "loss": 3.8780284881591798,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
- "grad_norm": 1.0078125,
194
- "learning_rate": 0.0006146,
195
- "loss": 3.837977981567383,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
- "grad_norm": 1.171875,
201
- "learning_rate": 0.0006426,
202
- "loss": 3.777029800415039,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
- "grad_norm": 0.79296875,
208
- "learning_rate": 0.0006705999999999999,
209
- "loss": 3.6980987548828126,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
- "grad_norm": 0.89453125,
215
- "learning_rate": 0.0006986,
216
- "loss": 3.6566497802734377,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
- "eval_loss": 3.6329572200775146,
222
- "eval_runtime": 7.8721,
223
- "eval_samples_per_second": 1210.221,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
- "grad_norm": 0.8984375,
230
- "learning_rate": 0.0007,
231
- "loss": 3.620241165161133,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
- "grad_norm": 1.140625,
237
- "learning_rate": 0.0007,
238
- "loss": 3.5605819702148436,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
- "grad_norm": 0.84375,
244
- "learning_rate": 0.0007,
245
- "loss": 3.515232467651367,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
- "grad_norm": 0.9609375,
251
- "learning_rate": 0.0007,
252
- "loss": 3.472977066040039,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
- "grad_norm": 0.8828125,
258
- "learning_rate": 0.0007,
259
- "loss": 3.442094421386719,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
- "eval_loss": 3.4318618774414062,
265
- "eval_runtime": 7.8952,
266
- "eval_samples_per_second": 1206.68,
267
- "eval_steps_per_second": 1.267,
268
  "step": 600
269
  },
270
  {
271
  "epoch": 0.020896528479946073,
272
- "grad_norm": 0.91015625,
273
- "learning_rate": 0.0007,
274
- "loss": 3.3882373809814452,
275
  "step": 620
276
  },
277
  {
278
  "epoch": 0.021570610043815303,
279
- "grad_norm": 0.921875,
280
- "learning_rate": 0.0007,
281
- "loss": 3.363667297363281,
282
  "step": 640
283
  },
284
  {
285
  "epoch": 0.022244691607684528,
286
- "grad_norm": 0.77734375,
287
- "learning_rate": 0.0007,
288
- "loss": 3.3378562927246094,
289
  "step": 660
290
  },
291
  {
292
  "epoch": 0.022918773171553757,
293
- "grad_norm": 0.859375,
294
- "learning_rate": 0.0007,
295
- "loss": 3.2941909790039063,
296
  "step": 680
297
  },
298
  {
299
  "epoch": 0.023592854735422986,
300
- "grad_norm": 0.7265625,
301
- "learning_rate": 0.0007,
302
- "loss": 3.2900840759277346,
303
  "step": 700
304
  },
305
  {
306
  "epoch": 0.023592854735422986,
307
- "eval_loss": 3.277914047241211,
308
- "eval_runtime": 7.9495,
309
- "eval_samples_per_second": 1198.439,
310
- "eval_steps_per_second": 1.258,
311
  "step": 700
312
  }
313
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
+ "grad_norm": 2.328125,
15
+ "learning_rate": 1.14e-05,
16
+ "loss": 8.309149932861327,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.734375,
22
+ "learning_rate": 2.34e-05,
23
+ "loss": 8.2459228515625,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
+ "grad_norm": 1.296875,
29
+ "learning_rate": 3.539999999999999e-05,
30
+ "loss": 8.09324951171875,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
+ "grad_norm": 1.28125,
36
+ "learning_rate": 4.7399999999999993e-05,
37
+ "loss": 7.956021881103515,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
+ "learning_rate": 5.94e-05,
44
+ "loss": 7.7899658203125,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
+ "eval_loss": 7.690263748168945,
50
+ "eval_runtime": 7.9534,
51
+ "eval_samples_per_second": 1197.859,
52
+ "eval_steps_per_second": 1.257,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
+ "grad_norm": 1.25,
58
+ "learning_rate": 7.139999999999999e-05,
59
+ "loss": 7.591774749755859,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
+ "grad_norm": 1.265625,
65
+ "learning_rate": 8.34e-05,
66
+ "loss": 7.362733459472656,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
+ "grad_norm": 1.2734375,
72
+ "learning_rate": 9.539999999999999e-05,
73
+ "loss": 7.124365234375,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
+ "grad_norm": 1.2265625,
79
+ "learning_rate": 0.00010739999999999998,
80
+ "loss": 6.878346252441406,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
+ "grad_norm": 1.8359375,
86
+ "learning_rate": 0.0001194,
87
+ "loss": 6.63219985961914,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
+ "eval_loss": 6.522833824157715,
93
+ "eval_runtime": 7.8828,
94
+ "eval_samples_per_second": 1208.576,
95
+ "eval_steps_per_second": 1.269,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
+ "grad_norm": 1.703125,
101
+ "learning_rate": 0.0001314,
102
+ "loss": 6.431203460693359,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
+ "grad_norm": 1.1796875,
108
+ "learning_rate": 0.0001434,
109
+ "loss": 6.207939529418946,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
+ "grad_norm": 1.125,
115
+ "learning_rate": 0.00015539999999999998,
116
+ "loss": 5.998645782470703,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
+ "grad_norm": 1.0625,
122
+ "learning_rate": 0.0001674,
123
+ "loss": 5.795352172851563,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
+ "grad_norm": 0.93359375,
129
+ "learning_rate": 0.00017939999999999997,
130
+ "loss": 5.597444915771485,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
+ "eval_loss": 5.480377674102783,
136
+ "eval_runtime": 7.8653,
137
+ "eval_samples_per_second": 1211.274,
138
+ "eval_steps_per_second": 1.271,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
+ "grad_norm": 0.94140625,
144
+ "learning_rate": 0.0001914,
145
+ "loss": 5.3849952697753904,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
+ "grad_norm": 0.98828125,
151
+ "learning_rate": 0.00020339999999999998,
152
+ "loss": 5.2166282653808596,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
+ "grad_norm": 1.078125,
158
+ "learning_rate": 0.00021539999999999998,
159
+ "loss": 5.036893081665039,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
+ "grad_norm": 1.2109375,
165
+ "learning_rate": 0.00022739999999999997,
166
+ "loss": 4.901935195922851,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
+ "grad_norm": 0.984375,
172
+ "learning_rate": 0.0002394,
173
+ "loss": 4.777506256103516,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
+ "eval_loss": 4.718365669250488,
179
+ "eval_runtime": 7.988,
180
+ "eval_samples_per_second": 1192.66,
181
+ "eval_steps_per_second": 1.252,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
+ "grad_norm": 1.2109375,
187
+ "learning_rate": 0.0002514,
188
+ "loss": 4.656603622436523,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
+ "grad_norm": 1.71875,
194
+ "learning_rate": 0.00026339999999999995,
195
+ "loss": 4.587258148193359,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
+ "grad_norm": 1.21875,
201
+ "learning_rate": 0.00027539999999999997,
202
+ "loss": 4.507755279541016,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
+ "grad_norm": 1.25,
208
+ "learning_rate": 0.00028739999999999994,
209
+ "loss": 4.410206985473633,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
+ "grad_norm": 1.3203125,
215
+ "learning_rate": 0.00029939999999999996,
216
+ "loss": 4.3579551696777346,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
+ "eval_loss": 4.317867755889893,
222
+ "eval_runtime": 7.8743,
223
+ "eval_samples_per_second": 1209.882,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
+ "grad_norm": 1.1484375,
230
+ "learning_rate": 0.0003,
231
+ "loss": 4.301048278808594,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
+ "grad_norm": 1.3046875,
237
+ "learning_rate": 0.0003,
238
+ "loss": 4.22619743347168,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
+ "grad_norm": 1.1328125,
244
+ "learning_rate": 0.0003,
245
+ "loss": 4.173627090454102,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
+ "grad_norm": 0.97265625,
251
+ "learning_rate": 0.0003,
252
+ "loss": 4.122076797485351,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
+ "grad_norm": 1.2109375,
258
+ "learning_rate": 0.0003,
259
+ "loss": 4.078117370605469,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
+ "eval_loss": 4.0582146644592285,
265
+ "eval_runtime": 7.9148,
266
+ "eval_samples_per_second": 1203.7,
267
+ "eval_steps_per_second": 1.263,
268
  "step": 600
269
  },
270
  {
271
  "epoch": 0.020896528479946073,
272
+ "grad_norm": 1.234375,
273
+ "learning_rate": 0.0003,
274
+ "loss": 4.019903182983398,
275
  "step": 620
276
  },
277
  {
278
  "epoch": 0.021570610043815303,
279
+ "grad_norm": 1.421875,
280
+ "learning_rate": 0.0003,
281
+ "loss": 3.995362091064453,
282
  "step": 640
283
  },
284
  {
285
  "epoch": 0.022244691607684528,
286
+ "grad_norm": 1.1875,
287
+ "learning_rate": 0.0003,
288
+ "loss": 3.961619186401367,
289
  "step": 660
290
  },
291
  {
292
  "epoch": 0.022918773171553757,
293
+ "grad_norm": 1.2578125,
294
+ "learning_rate": 0.0003,
295
+ "loss": 3.922329330444336,
296
  "step": 680
297
  },
298
  {
299
  "epoch": 0.023592854735422986,
300
+ "grad_norm": 1.34375,
301
+ "learning_rate": 0.0003,
302
+ "loss": 3.915304946899414,
303
  "step": 700
304
  },
305
  {
306
  "epoch": 0.023592854735422986,
307
+ "eval_loss": 3.899839401245117,
308
+ "eval_runtime": 7.9432,
309
+ "eval_samples_per_second": 1199.387,
310
+ "eval_steps_per_second": 1.259,
311
  "step": 700
312
  }
313
  ],
zain/Activation/out/mlp-gelu-9L_run/checkpoint-700/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:79e0def26beb7a751361df77096e4e9d429ab64974840e7eb3dcf84a49f4a583
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:282bf64fd9cb5669a85393aac3b68bd51ade27ff8218886467eed07f2725602f
3
  size 4920
zain/Activation/out/mlp-gelu-9L_run/checkpoint-800/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f3d3613fc8d170ff8709329a4777d28e49a7f2ed4d7efd845881d72494af6c20
3
  size 4010544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d3f04f78afdee67e816c503c9704cbb548b0c1e34d83274b2db79860fcb271c
3
  size 4010544
zain/Activation/out/mlp-gelu-9L_run/checkpoint-800/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8cc15549707ce57325fd1535acadf307c7372cae5db1384d6e19ece30d7b7852
3
  size 8068282
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:84ec3365760c8c9d65abea8de19c2c704a7b06e6d5e4d868dd08f176d493ded6
3
  size 8068282
zain/Activation/out/mlp-gelu-9L_run/checkpoint-800/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b4ae273f6e52d9470fdb0cf70fb543f74693bb1ddd058314f4d5c7d3c01c667e
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2224470b783e3f05273b6a4ac609fa6ba76b1099ae64a843a85fea2d2bf1e0dd
3
  size 1064
zain/Activation/out/mlp-gelu-9L_run/checkpoint-800/trainer_state.json CHANGED
@@ -11,346 +11,346 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
- "grad_norm": 1.8984375,
15
- "learning_rate": 2.66e-05,
16
- "loss": 8.290375518798829,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
- "grad_norm": 1.3125,
22
- "learning_rate": 5.46e-05,
23
- "loss": 8.103549194335937,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
- "grad_norm": 1.2734375,
29
- "learning_rate": 8.259999999999999e-05,
30
- "loss": 7.879036712646484,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
- "grad_norm": 1.2734375,
36
- "learning_rate": 0.0001106,
37
- "loss": 7.590550231933594,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
- "learning_rate": 0.0001386,
44
- "loss": 7.234850311279297,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
- "eval_loss": 7.028295993804932,
50
- "eval_runtime": 7.9767,
51
- "eval_samples_per_second": 1194.359,
52
- "eval_steps_per_second": 1.254,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
- "grad_norm": 1.7421875,
58
- "learning_rate": 0.00016659999999999998,
59
- "loss": 6.841970825195313,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
- "grad_norm": 1.1484375,
65
- "learning_rate": 0.00019460000000000001,
66
- "loss": 6.4576164245605465,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
- "grad_norm": 1.0859375,
72
- "learning_rate": 0.0002226,
73
- "loss": 6.111285781860351,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
- "grad_norm": 1.109375,
79
- "learning_rate": 0.00025059999999999997,
80
- "loss": 5.793069076538086,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
- "grad_norm": 0.85546875,
86
- "learning_rate": 0.0002786,
87
- "loss": 5.492683410644531,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
- "eval_loss": 5.342745304107666,
93
- "eval_runtime": 7.9623,
94
- "eval_samples_per_second": 1196.508,
95
- "eval_steps_per_second": 1.256,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
- "grad_norm": 0.87890625,
101
- "learning_rate": 0.00030659999999999997,
102
- "loss": 5.233791732788086,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
- "grad_norm": 0.78515625,
108
- "learning_rate": 0.0003346,
109
- "loss": 4.977091598510742,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
- "grad_norm": 0.87890625,
115
- "learning_rate": 0.00036260000000000003,
116
- "loss": 4.77899055480957,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
- "grad_norm": 0.91015625,
122
- "learning_rate": 0.0003906,
123
- "loss": 4.601796722412109,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
- "grad_norm": 0.87109375,
129
- "learning_rate": 0.0004186,
130
- "loss": 4.450910949707032,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
- "eval_loss": 4.373648643493652,
136
- "eval_runtime": 7.9961,
137
- "eval_samples_per_second": 1191.458,
138
- "eval_steps_per_second": 1.251,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
- "grad_norm": 1.078125,
144
- "learning_rate": 0.0004466,
145
- "loss": 4.315410232543945,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
- "grad_norm": 0.80078125,
151
- "learning_rate": 0.00047460000000000004,
152
- "loss": 4.23107795715332,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
- "grad_norm": 0.91015625,
158
- "learning_rate": 0.0005026,
159
- "loss": 4.112172698974609,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
- "grad_norm": 0.8828125,
165
- "learning_rate": 0.0005306,
166
- "loss": 4.030664443969727,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
- "grad_norm": 0.93359375,
172
- "learning_rate": 0.0005586,
173
- "loss": 3.95709228515625,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
- "eval_loss": 3.9225285053253174,
179
- "eval_runtime": 8.1083,
180
- "eval_samples_per_second": 1174.974,
181
- "eval_steps_per_second": 1.233,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
- "grad_norm": 1.0078125,
187
- "learning_rate": 0.0005866,
188
- "loss": 3.8780284881591798,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
- "grad_norm": 1.0078125,
194
- "learning_rate": 0.0006146,
195
- "loss": 3.837977981567383,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
- "grad_norm": 1.171875,
201
- "learning_rate": 0.0006426,
202
- "loss": 3.777029800415039,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
- "grad_norm": 0.79296875,
208
- "learning_rate": 0.0006705999999999999,
209
- "loss": 3.6980987548828126,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
- "grad_norm": 0.89453125,
215
- "learning_rate": 0.0006986,
216
- "loss": 3.6566497802734377,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
- "eval_loss": 3.6329572200775146,
222
- "eval_runtime": 7.8721,
223
- "eval_samples_per_second": 1210.221,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
- "grad_norm": 0.8984375,
230
- "learning_rate": 0.0007,
231
- "loss": 3.620241165161133,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
- "grad_norm": 1.140625,
237
- "learning_rate": 0.0007,
238
- "loss": 3.5605819702148436,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
- "grad_norm": 0.84375,
244
- "learning_rate": 0.0007,
245
- "loss": 3.515232467651367,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
- "grad_norm": 0.9609375,
251
- "learning_rate": 0.0007,
252
- "loss": 3.472977066040039,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
- "grad_norm": 0.8828125,
258
- "learning_rate": 0.0007,
259
- "loss": 3.442094421386719,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
- "eval_loss": 3.4318618774414062,
265
- "eval_runtime": 7.8952,
266
- "eval_samples_per_second": 1206.68,
267
- "eval_steps_per_second": 1.267,
268
  "step": 600
269
  },
270
  {
271
  "epoch": 0.020896528479946073,
272
- "grad_norm": 0.91015625,
273
- "learning_rate": 0.0007,
274
- "loss": 3.3882373809814452,
275
  "step": 620
276
  },
277
  {
278
  "epoch": 0.021570610043815303,
279
- "grad_norm": 0.921875,
280
- "learning_rate": 0.0007,
281
- "loss": 3.363667297363281,
282
  "step": 640
283
  },
284
  {
285
  "epoch": 0.022244691607684528,
286
- "grad_norm": 0.77734375,
287
- "learning_rate": 0.0007,
288
- "loss": 3.3378562927246094,
289
  "step": 660
290
  },
291
  {
292
  "epoch": 0.022918773171553757,
293
- "grad_norm": 0.859375,
294
- "learning_rate": 0.0007,
295
- "loss": 3.2941909790039063,
296
  "step": 680
297
  },
298
  {
299
  "epoch": 0.023592854735422986,
300
- "grad_norm": 0.7265625,
301
- "learning_rate": 0.0007,
302
- "loss": 3.2900840759277346,
303
  "step": 700
304
  },
305
  {
306
  "epoch": 0.023592854735422986,
307
- "eval_loss": 3.277914047241211,
308
- "eval_runtime": 7.9495,
309
- "eval_samples_per_second": 1198.439,
310
- "eval_steps_per_second": 1.258,
311
  "step": 700
312
  },
313
  {
314
  "epoch": 0.024266936299292215,
315
- "grad_norm": 0.8203125,
316
- "learning_rate": 0.0007,
317
- "loss": 3.2640079498291015,
318
  "step": 720
319
  },
320
  {
321
  "epoch": 0.02494101786316144,
322
- "grad_norm": 0.796875,
323
- "learning_rate": 0.0007,
324
- "loss": 3.2568283081054688,
325
  "step": 740
326
  },
327
  {
328
  "epoch": 0.02561509942703067,
329
- "grad_norm": 0.828125,
330
- "learning_rate": 0.0007,
331
- "loss": 3.206019973754883,
332
  "step": 760
333
  },
334
  {
335
  "epoch": 0.0262891809908999,
336
- "grad_norm": 0.80078125,
337
- "learning_rate": 0.0007,
338
- "loss": 3.164164924621582,
339
  "step": 780
340
  },
341
  {
342
  "epoch": 0.026963262554769128,
343
- "grad_norm": 0.85546875,
344
- "learning_rate": 0.0007,
345
- "loss": 3.1771799087524415,
346
  "step": 800
347
  },
348
  {
349
  "epoch": 0.026963262554769128,
350
- "eval_loss": 3.1695895195007324,
351
- "eval_runtime": 7.9915,
352
- "eval_samples_per_second": 1192.141,
353
- "eval_steps_per_second": 1.251,
354
  "step": 800
355
  }
356
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.0006740815638692282,
14
+ "grad_norm": 2.328125,
15
+ "learning_rate": 1.14e-05,
16
+ "loss": 8.309149932861327,
17
  "step": 20
18
  },
19
  {
20
  "epoch": 0.0013481631277384564,
21
+ "grad_norm": 1.734375,
22
+ "learning_rate": 2.34e-05,
23
+ "loss": 8.2459228515625,
24
  "step": 40
25
  },
26
  {
27
  "epoch": 0.0020222446916076846,
28
+ "grad_norm": 1.296875,
29
+ "learning_rate": 3.539999999999999e-05,
30
+ "loss": 8.09324951171875,
31
  "step": 60
32
  },
33
  {
34
  "epoch": 0.002696326255476913,
35
+ "grad_norm": 1.28125,
36
+ "learning_rate": 4.7399999999999993e-05,
37
+ "loss": 7.956021881103515,
38
  "step": 80
39
  },
40
  {
41
  "epoch": 0.003370407819346141,
42
  "grad_norm": 1.25,
43
+ "learning_rate": 5.94e-05,
44
+ "loss": 7.7899658203125,
45
  "step": 100
46
  },
47
  {
48
  "epoch": 0.003370407819346141,
49
+ "eval_loss": 7.690263748168945,
50
+ "eval_runtime": 7.9534,
51
+ "eval_samples_per_second": 1197.859,
52
+ "eval_steps_per_second": 1.257,
53
  "step": 100
54
  },
55
  {
56
  "epoch": 0.004044489383215369,
57
+ "grad_norm": 1.25,
58
+ "learning_rate": 7.139999999999999e-05,
59
+ "loss": 7.591774749755859,
60
  "step": 120
61
  },
62
  {
63
  "epoch": 0.0047185709470845974,
64
+ "grad_norm": 1.265625,
65
+ "learning_rate": 8.34e-05,
66
+ "loss": 7.362733459472656,
67
  "step": 140
68
  },
69
  {
70
  "epoch": 0.005392652510953826,
71
+ "grad_norm": 1.2734375,
72
+ "learning_rate": 9.539999999999999e-05,
73
+ "loss": 7.124365234375,
74
  "step": 160
75
  },
76
  {
77
  "epoch": 0.006066734074823054,
78
+ "grad_norm": 1.2265625,
79
+ "learning_rate": 0.00010739999999999998,
80
+ "loss": 6.878346252441406,
81
  "step": 180
82
  },
83
  {
84
  "epoch": 0.006740815638692282,
85
+ "grad_norm": 1.8359375,
86
+ "learning_rate": 0.0001194,
87
+ "loss": 6.63219985961914,
88
  "step": 200
89
  },
90
  {
91
  "epoch": 0.006740815638692282,
92
+ "eval_loss": 6.522833824157715,
93
+ "eval_runtime": 7.8828,
94
+ "eval_samples_per_second": 1208.576,
95
+ "eval_steps_per_second": 1.269,
96
  "step": 200
97
  },
98
  {
99
  "epoch": 0.00741489720256151,
100
+ "grad_norm": 1.703125,
101
+ "learning_rate": 0.0001314,
102
+ "loss": 6.431203460693359,
103
  "step": 220
104
  },
105
  {
106
  "epoch": 0.008088978766430738,
107
+ "grad_norm": 1.1796875,
108
+ "learning_rate": 0.0001434,
109
+ "loss": 6.207939529418946,
110
  "step": 240
111
  },
112
  {
113
  "epoch": 0.008763060330299966,
114
+ "grad_norm": 1.125,
115
+ "learning_rate": 0.00015539999999999998,
116
+ "loss": 5.998645782470703,
117
  "step": 260
118
  },
119
  {
120
  "epoch": 0.009437141894169195,
121
+ "grad_norm": 1.0625,
122
+ "learning_rate": 0.0001674,
123
+ "loss": 5.795352172851563,
124
  "step": 280
125
  },
126
  {
127
  "epoch": 0.010111223458038422,
128
+ "grad_norm": 0.93359375,
129
+ "learning_rate": 0.00017939999999999997,
130
+ "loss": 5.597444915771485,
131
  "step": 300
132
  },
133
  {
134
  "epoch": 0.010111223458038422,
135
+ "eval_loss": 5.480377674102783,
136
+ "eval_runtime": 7.8653,
137
+ "eval_samples_per_second": 1211.274,
138
+ "eval_steps_per_second": 1.271,
139
  "step": 300
140
  },
141
  {
142
  "epoch": 0.010785305021907651,
143
+ "grad_norm": 0.94140625,
144
+ "learning_rate": 0.0001914,
145
+ "loss": 5.3849952697753904,
146
  "step": 320
147
  },
148
  {
149
  "epoch": 0.011459386585776879,
150
+ "grad_norm": 0.98828125,
151
+ "learning_rate": 0.00020339999999999998,
152
+ "loss": 5.2166282653808596,
153
  "step": 340
154
  },
155
  {
156
  "epoch": 0.012133468149646108,
157
+ "grad_norm": 1.078125,
158
+ "learning_rate": 0.00021539999999999998,
159
+ "loss": 5.036893081665039,
160
  "step": 360
161
  },
162
  {
163
  "epoch": 0.012807549713515335,
164
+ "grad_norm": 1.2109375,
165
+ "learning_rate": 0.00022739999999999997,
166
+ "loss": 4.901935195922851,
167
  "step": 380
168
  },
169
  {
170
  "epoch": 0.013481631277384564,
171
+ "grad_norm": 0.984375,
172
+ "learning_rate": 0.0002394,
173
+ "loss": 4.777506256103516,
174
  "step": 400
175
  },
176
  {
177
  "epoch": 0.013481631277384564,
178
+ "eval_loss": 4.718365669250488,
179
+ "eval_runtime": 7.988,
180
+ "eval_samples_per_second": 1192.66,
181
+ "eval_steps_per_second": 1.252,
182
  "step": 400
183
  },
184
  {
185
  "epoch": 0.014155712841253791,
186
+ "grad_norm": 1.2109375,
187
+ "learning_rate": 0.0002514,
188
+ "loss": 4.656603622436523,
189
  "step": 420
190
  },
191
  {
192
  "epoch": 0.01482979440512302,
193
+ "grad_norm": 1.71875,
194
+ "learning_rate": 0.00026339999999999995,
195
+ "loss": 4.587258148193359,
196
  "step": 440
197
  },
198
  {
199
  "epoch": 0.015503875968992248,
200
+ "grad_norm": 1.21875,
201
+ "learning_rate": 0.00027539999999999997,
202
+ "loss": 4.507755279541016,
203
  "step": 460
204
  },
205
  {
206
  "epoch": 0.016177957532861477,
207
+ "grad_norm": 1.25,
208
+ "learning_rate": 0.00028739999999999994,
209
+ "loss": 4.410206985473633,
210
  "step": 480
211
  },
212
  {
213
  "epoch": 0.016852039096730706,
214
+ "grad_norm": 1.3203125,
215
+ "learning_rate": 0.00029939999999999996,
216
+ "loss": 4.3579551696777346,
217
  "step": 500
218
  },
219
  {
220
  "epoch": 0.016852039096730706,
221
+ "eval_loss": 4.317867755889893,
222
+ "eval_runtime": 7.8743,
223
+ "eval_samples_per_second": 1209.882,
224
  "eval_steps_per_second": 1.27,
225
  "step": 500
226
  },
227
  {
228
  "epoch": 0.01752612066059993,
229
+ "grad_norm": 1.1484375,
230
+ "learning_rate": 0.0003,
231
+ "loss": 4.301048278808594,
232
  "step": 520
233
  },
234
  {
235
  "epoch": 0.01820020222446916,
236
+ "grad_norm": 1.3046875,
237
+ "learning_rate": 0.0003,
238
+ "loss": 4.22619743347168,
239
  "step": 540
240
  },
241
  {
242
  "epoch": 0.01887428378833839,
243
+ "grad_norm": 1.1328125,
244
+ "learning_rate": 0.0003,
245
+ "loss": 4.173627090454102,
246
  "step": 560
247
  },
248
  {
249
  "epoch": 0.01954836535220762,
250
+ "grad_norm": 0.97265625,
251
+ "learning_rate": 0.0003,
252
+ "loss": 4.122076797485351,
253
  "step": 580
254
  },
255
  {
256
  "epoch": 0.020222446916076844,
257
+ "grad_norm": 1.2109375,
258
+ "learning_rate": 0.0003,
259
+ "loss": 4.078117370605469,
260
  "step": 600
261
  },
262
  {
263
  "epoch": 0.020222446916076844,
264
+ "eval_loss": 4.0582146644592285,
265
+ "eval_runtime": 7.9148,
266
+ "eval_samples_per_second": 1203.7,
267
+ "eval_steps_per_second": 1.263,
268
  "step": 600
269
  },
270
  {
271
  "epoch": 0.020896528479946073,
272
+ "grad_norm": 1.234375,
273
+ "learning_rate": 0.0003,
274
+ "loss": 4.019903182983398,
275
  "step": 620
276
  },
277
  {
278
  "epoch": 0.021570610043815303,
279
+ "grad_norm": 1.421875,
280
+ "learning_rate": 0.0003,
281
+ "loss": 3.995362091064453,
282
  "step": 640
283
  },
284
  {
285
  "epoch": 0.022244691607684528,
286
+ "grad_norm": 1.1875,
287
+ "learning_rate": 0.0003,
288
+ "loss": 3.961619186401367,
289
  "step": 660
290
  },
291
  {
292
  "epoch": 0.022918773171553757,
293
+ "grad_norm": 1.2578125,
294
+ "learning_rate": 0.0003,
295
+ "loss": 3.922329330444336,
296
  "step": 680
297
  },
298
  {
299
  "epoch": 0.023592854735422986,
300
+ "grad_norm": 1.34375,
301
+ "learning_rate": 0.0003,
302
+ "loss": 3.915304946899414,
303
  "step": 700
304
  },
305
  {
306
  "epoch": 0.023592854735422986,
307
+ "eval_loss": 3.899839401245117,
308
+ "eval_runtime": 7.9432,
309
+ "eval_samples_per_second": 1199.387,
310
+ "eval_steps_per_second": 1.259,
311
  "step": 700
312
  },
313
  {
314
  "epoch": 0.024266936299292215,
315
+ "grad_norm": 1.265625,
316
+ "learning_rate": 0.0003,
317
+ "loss": 3.8832908630371095,
318
  "step": 720
319
  },
320
  {
321
  "epoch": 0.02494101786316144,
322
+ "grad_norm": 1.2734375,
323
+ "learning_rate": 0.0003,
324
+ "loss": 3.873937225341797,
325
  "step": 740
326
  },
327
  {
328
  "epoch": 0.02561509942703067,
329
+ "grad_norm": 1.5,
330
+ "learning_rate": 0.0003,
331
+ "loss": 3.8289634704589846,
332
  "step": 760
333
  },
334
  {
335
  "epoch": 0.0262891809908999,
336
+ "grad_norm": 1.265625,
337
+ "learning_rate": 0.0003,
338
+ "loss": 3.78992919921875,
339
  "step": 780
340
  },
341
  {
342
  "epoch": 0.026963262554769128,
343
+ "grad_norm": 1.1015625,
344
+ "learning_rate": 0.0003,
345
+ "loss": 3.7991443634033204,
346
  "step": 800
347
  },
348
  {
349
  "epoch": 0.026963262554769128,
350
+ "eval_loss": 3.794043779373169,
351
+ "eval_runtime": 7.9171,
352
+ "eval_samples_per_second": 1203.347,
353
+ "eval_steps_per_second": 1.263,
354
  "step": 800
355
  }
356
  ],
zain/Activation/out/mlp-gelu-9L_run/checkpoint-800/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:79e0def26beb7a751361df77096e4e9d429ab64974840e7eb3dcf84a49f4a583
3
  size 4920
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:282bf64fd9cb5669a85393aac3b68bd51ade27ff8218886467eed07f2725602f
3
  size 4920
zain/Activation/out/mlp-gelu-9L_run/training_log.jsonl CHANGED
@@ -116,3 +116,18 @@
116
  {"step": 560, "epoch": 0.01887428378833839, "timestamp": 1787067750.2357593, "loss": 4.173627090454102, "grad_norm": 1.1328125, "learning_rate": 0.0003, "train/total_time_seconds": 12.387610416859388, "train/time_per_step_avg": 0.021234799884259702, "train/epoch_time_elapsed": 66.3985048159957, "train/estimated_remaining_minutes": 0.7152370300210479}
117
  {"step": 580, "epoch": 0.01954836535220762, "timestamp": 1787067751.1435575, "loss": 4.122076797485351, "grad_norm": 0.97265625, "learning_rate": 0.0003, "train/total_time_seconds": 12.805801656097174, "train/time_per_step_avg": 0.02109844535589218, "train/epoch_time_elapsed": 67.30630354955792, "train/estimated_remaining_minutes": 0.7065269879226027}
118
  {"step": 600, "epoch": 0.020222446916076844, "timestamp": 1787067752.0671544, "loss": 4.078117370605469, "grad_norm": 1.2109375, "learning_rate": 0.0003, "train/total_time_seconds": 13.223767712712288, "train/time_per_step_avg": 0.02113275308161974, "train/epoch_time_elapsed": 68.22989998012781, "train/estimated_remaining_minutes": 0.6979210737264818}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
116
  {"step": 560, "epoch": 0.01887428378833839, "timestamp": 1787067750.2357593, "loss": 4.173627090454102, "grad_norm": 1.1328125, "learning_rate": 0.0003, "train/total_time_seconds": 12.387610416859388, "train/time_per_step_avg": 0.021234799884259702, "train/epoch_time_elapsed": 66.3985048159957, "train/estimated_remaining_minutes": 0.7152370300210479}
117
  {"step": 580, "epoch": 0.01954836535220762, "timestamp": 1787067751.1435575, "loss": 4.122076797485351, "grad_norm": 0.97265625, "learning_rate": 0.0003, "train/total_time_seconds": 12.805801656097174, "train/time_per_step_avg": 0.02109844535589218, "train/epoch_time_elapsed": 67.30630354955792, "train/estimated_remaining_minutes": 0.7065269879226027}
118
  {"step": 600, "epoch": 0.020222446916076844, "timestamp": 1787067752.0671544, "loss": 4.078117370605469, "grad_norm": 1.2109375, "learning_rate": 0.0003, "train/total_time_seconds": 13.223767712712288, "train/time_per_step_avg": 0.02113275308161974, "train/epoch_time_elapsed": 68.22989998012781, "train/estimated_remaining_minutes": 0.6979210737264818}
119
+ {"step": 600, "epoch": 0.020222446916076844, "timestamp": 1787067759.983548, "eval_loss": 4.0582146644592285, "eval_runtime": 7.9148, "eval_samples_per_second": 1203.7, "eval_steps_per_second": 1.263, "train/total_time_seconds": 13.223767712712288, "train/time_per_step_avg": 0.02113275308161974, "train/epoch_time_elapsed": 76.14629316329956, "train/estimated_remaining_minutes": 0.6979210737264818}
120
+ {"step": 620, "epoch": 0.020896528479946073, "timestamp": 1787067760.930155, "loss": 4.019903182983398, "grad_norm": 1.234375, "learning_rate": 0.0003, "train/total_time_seconds": 13.63673610985279, "train/time_per_step_avg": 0.020922258235514163, "train/epoch_time_elapsed": 77.09290066733956, "train/estimated_remaining_minutes": 0.6891683840463239}
121
+ {"step": 640, "epoch": 0.021570610043815303, "timestamp": 1787067761.8332198, "loss": 3.995362091064453, "grad_norm": 1.421875, "learning_rate": 0.0003, "train/total_time_seconds": 14.049914173781872, "train/time_per_step_avg": 0.020797862485051156, "train/epoch_time_elapsed": 77.99596579000354, "train/estimated_remaining_minutes": 0.6805427177925594}
122
+ {"step": 660, "epoch": 0.022244691607684528, "timestamp": 1787067762.7551243, "loss": 3.961619186401367, "grad_norm": 1.1875, "learning_rate": 0.0003, "train/total_time_seconds": 14.477505411952734, "train/time_per_step_avg": 0.020898949950933457, "train/epoch_time_elapsed": 78.91787022352219, "train/estimated_remaining_minutes": 0.6726921706563896}
123
+ {"step": 680, "epoch": 0.022918773171553757, "timestamp": 1787067763.759098, "loss": 3.922329330444336, "grad_norm": 1.2578125, "learning_rate": 0.0003, "train/total_time_seconds": 14.928573966026306, "train/time_per_step_avg": 0.021227723099291326, "train/epoch_time_elapsed": 79.9218439757824, "train/estimated_remaining_minutes": 0.6659314857394087}
124
+ {"step": 700, "epoch": 0.023592854735422986, "timestamp": 1787067764.7285035, "loss": 3.915304946899414, "grad_norm": 1.34375, "learning_rate": 0.0003, "train/total_time_seconds": 15.398288868367672, "train/time_per_step_avg": 0.02174521155655384, "train/epoch_time_elapsed": 80.89124953746796, "train/estimated_remaining_minutes": 0.659926665787186}
125
+ {"step": 700, "epoch": 0.023592854735422986, "timestamp": 1787067772.6732872, "eval_loss": 3.899839401245117, "eval_runtime": 7.9432, "eval_samples_per_second": 1199.387, "eval_steps_per_second": 1.259, "train/total_time_seconds": 15.398288868367672, "train/time_per_step_avg": 0.02174521155655384, "train/epoch_time_elapsed": 88.83603245019913, "train/estimated_remaining_minutes": 0.659926665787186}
126
+ {"step": 720, "epoch": 0.024266936299292215, "timestamp": 1787067773.6728122, "loss": 3.8832908630371095, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 15.860269576311111, "train/time_per_step_avg": 0.022235334664583207, "train/epoch_time_elapsed": 89.83555833995342, "train/estimated_remaining_minutes": 0.6535018482831894}
127
+ {"step": 740, "epoch": 0.02494101786316144, "timestamp": 1787067774.6581304, "loss": 3.873937225341797, "grad_norm": 1.2734375, "learning_rate": 0.0003, "train/total_time_seconds": 16.31060327962041, "train/time_per_step_avg": 0.022606891058385373, "train/epoch_time_elapsed": 90.82087651267648, "train/estimated_remaining_minutes": 0.6465464363092774}
128
+ {"step": 760, "epoch": 0.02561509942703067, "timestamp": 1787067775.5529265, "loss": 3.8289634704589846, "grad_norm": 1.5, "learning_rate": 0.0003, "train/total_time_seconds": 16.720402892678976, "train/time_per_step_avg": 0.02242897480726242, "train/epoch_time_elapsed": 91.71567215025425, "train/estimated_remaining_minutes": 0.6380153735364346}
129
+ {"step": 780, "epoch": 0.0262891809908999, "timestamp": 1787067776.4514818, "loss": 3.78992919921875, "grad_norm": 1.265625, "learning_rate": 0.0003, "train/total_time_seconds": 17.129779014736414, "train/time_per_step_avg": 0.022012050487101077, "train/epoch_time_elapsed": 92.61422762274742, "train/estimated_remaining_minutes": 0.6295559808834751}
130
+ {"step": 800, "epoch": 0.026963262554769128, "timestamp": 1787067777.3482258, "loss": 3.7991443634033204, "grad_norm": 1.1015625, "learning_rate": 0.0003, "train/total_time_seconds": 17.54112770035863, "train/time_per_step_avg": 0.021428388319909574, "train/epoch_time_elapsed": 93.51097171381116, "train/estimated_remaining_minutes": 0.6212482727210348}
131
+ {"step": 800, "epoch": 0.026963262554769128, "timestamp": 1787067785.2667353, "eval_loss": 3.794043779373169, "eval_runtime": 7.9171, "eval_samples_per_second": 1203.347, "eval_steps_per_second": 1.263, "train/total_time_seconds": 17.54112770035863, "train/time_per_step_avg": 0.021428388319909574, "train/epoch_time_elapsed": 101.4294798001647, "train/estimated_remaining_minutes": 0.6212482727210348}
132
+ {"step": 820, "epoch": 0.027637344118638354, "timestamp": 1787067786.2143607, "loss": 3.7606269836425783, "grad_norm": 1.140625, "learning_rate": 0.0003, "train/total_time_seconds": 17.96183804422617, "train/time_per_step_avg": 0.021015684679150583, "train/epoch_time_elapsed": 102.37710624560714, "train/estimated_remaining_minutes": 0.6133310551686986}
133
+ {"step": 840, "epoch": 0.028311425682507583, "timestamp": 1787067787.139812, "loss": 3.7543869018554688, "grad_norm": 1.234375, "learning_rate": 0.0003, "train/total_time_seconds": 18.38495372235775, "train/time_per_step_avg": 0.02074350442737341, "train/epoch_time_elapsed": 103.30255784094334, "train/estimated_remaining_minutes": 0.6055361741887672}