CodeIsAbstract commited on
Commit
ad4280e
·
verified ·
1 Parent(s): 0f79d91

Training in progress, step 300, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:461a8fe0bbd13663ab3dd972387d88703c64c3540b767828d8a9ef1357b29a8f
3
  size 847599616
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b7a8bb8a4ae15ac03afd44a3577747a17cf8ae073c8e0a12f668e60bf4269c9
3
  size 847599616
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fdccc7eb9376504ec64e1323859d828791b50f3f8842beac2c59f63105017379
3
- size 350603
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d51750dfe21f1524b4e65e279067be3a9cf4264ff3dc3e8c306e71ffc8db8015
3
+ size 1386414411
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:65c0ddce770129db8af78a8281e61b32c4aba3200b93375ffc428f456531d83d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c587bc9603de883e44d0b68949a09abf17c1a2e4c3bc62449c91a13da9d5d454
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ab66db99117673b461ceb8f3865fe0441f45aa01085a419aba1737c22700a3ed
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c18d3014a31b7d0b10d3631683a1f76c56eca77072280718fe48627ba3a1c6f4
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,335 +2,457 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.0,
6
- "eval_steps": 50,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.025,
14
- "grad_norm": 0.28224506974220276,
15
- "learning_rate": 1.6e-06,
16
- "loss": 10.72566909790039,
17
  "step": 5
18
  },
19
  {
20
- "epoch": 0.05,
21
- "grad_norm": 0.3270684778690338,
22
- "learning_rate": 3.6e-06,
23
- "loss": 9.934751892089844,
24
  "step": 10
25
  },
26
  {
27
- "epoch": 0.075,
28
- "grad_norm": 0.16773350536823273,
29
- "learning_rate": 3.995627254437549e-06,
30
- "loss": 9.331498718261718,
31
  "step": 15
32
  },
33
  {
34
- "epoch": 0.1,
35
- "grad_norm": 0.14517103135585785,
36
- "learning_rate": 3.9778957412029366e-06,
37
- "loss": 9.084286499023438,
38
  "step": 20
39
  },
40
  {
41
- "epoch": 0.125,
42
- "grad_norm": 0.15857668220996857,
43
- "learning_rate": 3.9466531944960116e-06,
44
- "loss": 8.825327301025391,
45
  "step": 25
46
  },
47
  {
48
- "epoch": 0.15,
49
- "grad_norm": 0.12968920171260834,
50
- "learning_rate": 3.902113032590307e-06,
51
- "loss": 8.735884094238282,
52
  "step": 30
53
  },
54
  {
55
- "epoch": 0.175,
56
- "grad_norm": 0.11413325369358063,
57
- "learning_rate": 3.844579509954769e-06,
58
- "loss": 8.606085968017577,
59
  "step": 35
60
  },
61
  {
62
- "epoch": 0.2,
63
- "grad_norm": 0.17233271896839142,
64
- "learning_rate": 3.7744456388872234e-06,
65
- "loss": 8.629763793945312,
66
  "step": 40
67
  },
68
  {
69
- "epoch": 0.225,
70
- "grad_norm": 0.11048762500286102,
71
- "learning_rate": 3.6921905048418706e-06,
72
- "loss": 8.442276000976562,
73
  "step": 45
74
  },
75
  {
76
- "epoch": 0.25,
77
- "grad_norm": 0.08961036801338196,
78
- "learning_rate": 3.598375993789849e-06,
79
- "loss": 8.383564758300782,
80
  "step": 50
81
  },
82
  {
83
- "epoch": 0.25,
84
- "eval_accuracy": 0.11198778998778999,
85
- "eval_loss": 8.327370643615723,
86
- "eval_runtime": 10.4343,
87
- "eval_samples_per_second": 9.584,
88
- "eval_steps_per_second": 1.629,
89
- "step": 50
90
- },
91
- {
92
- "epoch": 0.275,
93
- "grad_norm": 0.08887746930122375,
94
- "learning_rate": 3.4936429539683075e-06,
95
- "loss": 8.346420288085938,
96
  "step": 55
97
  },
98
  {
99
- "epoch": 0.3,
100
- "grad_norm": 0.0796026661992073,
101
- "learning_rate": 3.3787068182371314e-06,
102
- "loss": 8.269767761230469,
103
  "step": 60
104
  },
105
  {
106
- "epoch": 0.325,
107
- "grad_norm": 0.09607570618391037,
108
- "learning_rate": 3.254352716947074e-06,
109
- "loss": 8.19412841796875,
110
  "step": 65
111
  },
112
  {
113
- "epoch": 0.35,
114
- "grad_norm": 0.1087556928396225,
115
- "learning_rate": 3.1214301147033453e-06,
116
- "loss": 8.142083740234375,
117
  "step": 70
118
  },
119
  {
120
- "epoch": 0.375,
121
- "grad_norm": 0.08590473234653473,
122
- "learning_rate": 2.9808470076610163e-06,
123
- "loss": 8.119364929199218,
124
  "step": 75
125
  },
126
  {
127
- "epoch": 0.4,
128
- "grad_norm": 0.10873208194971085,
129
- "learning_rate": 2.833563720990581e-06,
130
- "loss": 8.084170532226562,
131
  "step": 80
132
  },
133
  {
134
- "epoch": 0.425,
135
- "grad_norm": 0.08005784451961517,
136
- "learning_rate": 2.680586348883286e-06,
137
- "loss": 8.164724731445313,
138
  "step": 85
139
  },
140
  {
141
- "epoch": 0.45,
142
- "grad_norm": 0.07923667132854462,
143
- "learning_rate": 2.5229598819076393e-06,
144
- "loss": 8.08359603881836,
145
  "step": 90
146
  },
147
  {
148
- "epoch": 0.475,
149
- "grad_norm": 0.08169477432966232,
150
- "learning_rate": 2.36176106866422e-06,
151
- "loss": 7.9432830810546875,
152
  "step": 95
153
  },
154
  {
155
- "epoch": 0.5,
156
- "grad_norm": 0.0804147943854332,
157
- "learning_rate": 2.198091060500926e-06,
158
- "loss": 7.901922607421875,
159
  "step": 100
160
  },
161
  {
162
- "epoch": 0.5,
163
- "eval_accuracy": 0.1186935286935287,
164
- "eval_loss": 7.899251937866211,
165
- "eval_runtime": 10.4815,
166
- "eval_samples_per_second": 9.541,
167
- "eval_steps_per_second": 1.622,
168
- "step": 100
169
- },
170
- {
171
- "epoch": 0.525,
172
- "grad_norm": 0.07347024232149124,
173
- "learning_rate": 2.033067889532717e-06,
174
- "loss": 7.909750366210938,
175
  "step": 105
176
  },
177
  {
178
- "epoch": 0.55,
179
- "grad_norm": 0.07354505360126495,
180
- "learning_rate": 1.8678188313486012e-06,
181
- "loss": 7.931227111816407,
182
  "step": 110
183
  },
184
  {
185
- "epoch": 0.575,
186
- "grad_norm": 0.09041173756122589,
187
- "learning_rate": 1.7034727045763157e-06,
188
- "loss": 7.868826293945313,
189
  "step": 115
190
  },
191
  {
192
- "epoch": 0.6,
193
- "grad_norm": 0.08841845393180847,
194
- "learning_rate": 1.541152159906497e-06,
195
- "loss": 7.831324768066406,
196
  "step": 120
197
  },
198
  {
199
- "epoch": 0.625,
200
- "grad_norm": 0.06918840855360031,
201
- "learning_rate": 1.3819660112501052e-06,
202
- "loss": 7.8350990295410154,
203
  "step": 125
204
  },
205
  {
206
- "epoch": 0.65,
207
- "grad_norm": 0.08589889854192734,
208
- "learning_rate": 1.227001661415096e-06,
209
- "loss": 7.875128173828125,
210
  "step": 130
211
  },
212
  {
213
- "epoch": 0.675,
214
- "grad_norm": 0.07827843725681305,
215
- "learning_rate": 1.0773176740426246e-06,
216
- "loss": 7.833936309814453,
217
  "step": 135
218
  },
219
  {
220
- "epoch": 0.7,
221
- "grad_norm": 0.0692669078707695,
222
- "learning_rate": 9.339365425440129e-07,
223
- "loss": 7.741961669921875,
224
  "step": 140
225
  },
226
  {
227
- "epoch": 0.725,
228
- "grad_norm": 0.10289043933153152,
229
- "learning_rate": 7.978377054339498e-07,
230
- "loss": 7.69259033203125,
231
  "step": 145
232
  },
233
  {
234
- "epoch": 0.75,
235
- "grad_norm": 0.09380868077278137,
236
- "learning_rate": 6.699508557723032e-07,
237
- "loss": 7.753170776367187,
238
  "step": 150
239
  },
240
  {
241
- "epoch": 0.75,
242
- "eval_accuracy": 0.12066422466422466,
243
- "eval_loss": 7.652101516723633,
244
- "eval_runtime": 10.959,
245
- "eval_samples_per_second": 9.125,
246
- "eval_steps_per_second": 1.551,
247
  "step": 150
248
  },
249
  {
250
- "epoch": 0.775,
251
- "grad_norm": 0.07780186086893082,
252
- "learning_rate": 5.511495904178221e-07,
253
- "loss": 7.669155883789062,
254
  "step": 155
255
  },
256
  {
257
- "epoch": 0.8,
258
- "grad_norm": 0.08477064967155457,
259
- "learning_rate": 4.4224544247577535e-07,
260
- "loss": 7.680642700195312,
261
  "step": 160
262
  },
263
  {
264
- "epoch": 0.825,
265
- "grad_norm": 0.09825273603200912,
266
- "learning_rate": 3.439823377039599e-07,
267
- "loss": 7.696351623535156,
268
  "step": 165
269
  },
270
  {
271
- "epoch": 0.85,
272
- "grad_norm": 0.0829058587551117,
273
- "learning_rate": 2.570315127454452e-07,
274
- "loss": 7.609683227539063,
275
  "step": 170
276
  },
277
  {
278
- "epoch": 0.875,
279
- "grad_norm": 0.159183531999588,
280
- "learning_rate": 1.8198692990167497e-07,
281
- "loss": 7.513552856445313,
282
  "step": 175
283
  },
284
  {
285
- "epoch": 0.9,
286
- "grad_norm": 0.06830265372991562,
287
- "learning_rate": 1.1936121976767767e-07,
288
- "loss": 7.567562103271484,
289
  "step": 180
290
  },
291
  {
292
- "epoch": 0.925,
293
- "grad_norm": 0.12620584666728973,
294
- "learning_rate": 6.958217944530287e-08,
295
- "loss": 7.754322814941406,
296
  "step": 185
297
  },
298
  {
299
- "epoch": 0.95,
300
- "grad_norm": 0.08359432220458984,
301
- "learning_rate": 3.2989850255235265e-08,
302
- "loss": 7.56811294555664,
303
  "step": 190
304
  },
305
  {
306
- "epoch": 0.975,
307
- "grad_norm": 0.07398480176925659,
308
- "learning_rate": 9.834194909977168e-09,
309
- "loss": 7.568279266357422,
310
  "step": 195
311
  },
312
  {
313
- "epoch": 1.0,
314
- "grad_norm": 0.07088548690080643,
315
- "learning_rate": 2.7339001506199164e-10,
316
- "loss": 7.527226257324219,
317
  "step": 200
318
  },
319
  {
320
- "epoch": 1.0,
321
- "eval_accuracy": 0.1237924297924298,
322
- "eval_loss": 7.467564582824707,
323
- "eval_runtime": 10.43,
324
- "eval_samples_per_second": 9.588,
325
- "eval_steps_per_second": 1.63,
326
- "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
  }
328
  ],
329
  "logging_steps": 5,
330
- "max_steps": 200,
331
  "num_input_tokens_seen": 0,
332
  "num_train_epochs": 9223372036854775807,
333
- "save_steps": 100,
334
  "stateful_callbacks": {
335
  "TrainerControl": {
336
  "args": {
@@ -338,7 +460,7 @@
338
  "should_evaluate": false,
339
  "should_log": false,
340
  "should_save": true,
341
- "should_training_stop": true
342
  },
343
  "attributes": {}
344
  }
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.06,
6
+ "eval_steps": 150,
7
+ "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.001,
14
+ "grad_norm": 2.5244109630584717,
15
+ "learning_rate": 6.4e-08,
16
+ "loss": 10.941880798339843,
17
  "step": 5
18
  },
19
  {
20
+ "epoch": 0.002,
21
+ "grad_norm": 2.4931907653808594,
22
+ "learning_rate": 1.44e-07,
23
+ "loss": 10.940663146972657,
24
  "step": 10
25
  },
26
  {
27
+ "epoch": 0.003,
28
+ "grad_norm": 2.397547721862793,
29
+ "learning_rate": 2.24e-07,
30
+ "loss": 10.932537841796876,
31
  "step": 15
32
  },
33
  {
34
+ "epoch": 0.004,
35
+ "grad_norm": 2.362260103225708,
36
+ "learning_rate": 3.0399999999999997e-07,
37
+ "loss": 10.920089721679688,
38
  "step": 20
39
  },
40
  {
41
+ "epoch": 0.005,
42
+ "grad_norm": 2.5390405654907227,
43
+ "learning_rate": 3.84e-07,
44
+ "loss": 10.902597045898437,
45
  "step": 25
46
  },
47
  {
48
+ "epoch": 0.006,
49
+ "grad_norm": 2.4956705570220947,
50
+ "learning_rate": 4.64e-07,
51
+ "loss": 10.886011505126953,
52
  "step": 30
53
  },
54
  {
55
+ "epoch": 0.007,
56
+ "grad_norm": 2.5255372524261475,
57
+ "learning_rate": 5.44e-07,
58
+ "loss": 10.858057403564453,
59
  "step": 35
60
  },
61
  {
62
+ "epoch": 0.008,
63
+ "grad_norm": 2.4960005283355713,
64
+ "learning_rate": 6.24e-07,
65
+ "loss": 10.832856750488281,
66
  "step": 40
67
  },
68
  {
69
+ "epoch": 0.009,
70
+ "grad_norm": 2.6297495365142822,
71
+ "learning_rate": 7.04e-07,
72
+ "loss": 10.798202514648438,
73
  "step": 45
74
  },
75
  {
76
+ "epoch": 0.01,
77
+ "grad_norm": 2.4866788387298584,
78
+ "learning_rate": 7.84e-07,
79
+ "loss": 10.764683532714844,
80
  "step": 50
81
  },
82
  {
83
+ "epoch": 0.011,
84
+ "grad_norm": 2.472733736038208,
85
+ "learning_rate": 8.639999999999999e-07,
86
+ "loss": 10.72347640991211,
 
 
 
 
 
 
 
 
 
87
  "step": 55
88
  },
89
  {
90
+ "epoch": 0.012,
91
+ "grad_norm": 2.432396650314331,
92
+ "learning_rate": 9.439999999999999e-07,
93
+ "loss": 10.679368591308593,
94
  "step": 60
95
  },
96
  {
97
+ "epoch": 0.013,
98
+ "grad_norm": 2.444960355758667,
99
+ "learning_rate": 1.024e-06,
100
+ "loss": 10.622978973388673,
101
  "step": 65
102
  },
103
  {
104
+ "epoch": 0.014,
105
+ "grad_norm": 2.3638627529144287,
106
+ "learning_rate": 1.1040000000000001e-06,
107
+ "loss": 10.573637390136719,
108
  "step": 70
109
  },
110
  {
111
+ "epoch": 0.015,
112
+ "grad_norm": 2.4232065677642822,
113
+ "learning_rate": 1.1839999999999998e-06,
114
+ "loss": 10.517623901367188,
115
  "step": 75
116
  },
117
  {
118
+ "epoch": 0.016,
119
+ "grad_norm": 2.4094903469085693,
120
+ "learning_rate": 1.2639999999999999e-06,
121
+ "loss": 10.454130554199219,
122
  "step": 80
123
  },
124
  {
125
+ "epoch": 0.017,
126
+ "grad_norm": 2.3315868377685547,
127
+ "learning_rate": 1.344e-06,
128
+ "loss": 10.414918518066406,
129
  "step": 85
130
  },
131
  {
132
+ "epoch": 0.018,
133
+ "grad_norm": 2.5305004119873047,
134
+ "learning_rate": 1.4239999999999998e-06,
135
+ "loss": 10.341883087158203,
136
  "step": 90
137
  },
138
  {
139
+ "epoch": 0.019,
140
+ "grad_norm": 2.3920392990112305,
141
+ "learning_rate": 1.504e-06,
142
+ "loss": 10.270912170410156,
143
  "step": 95
144
  },
145
  {
146
+ "epoch": 0.02,
147
+ "grad_norm": 2.678619861602783,
148
+ "learning_rate": 1.584e-06,
149
+ "loss": 10.162237548828125,
150
  "step": 100
151
  },
152
  {
153
+ "epoch": 0.021,
154
+ "grad_norm": 2.3744475841522217,
155
+ "learning_rate": 1.6639999999999999e-06,
156
+ "loss": 10.103695678710938,
 
 
 
 
 
 
 
 
 
157
  "step": 105
158
  },
159
  {
160
+ "epoch": 0.022,
161
+ "grad_norm": 2.334637403488159,
162
+ "learning_rate": 1.744e-06,
163
+ "loss": 10.062642669677734,
164
  "step": 110
165
  },
166
  {
167
+ "epoch": 0.023,
168
+ "grad_norm": 2.4162437915802,
169
+ "learning_rate": 1.824e-06,
170
+ "loss": 10.005216979980469,
171
  "step": 115
172
  },
173
  {
174
+ "epoch": 0.024,
175
+ "grad_norm": 2.38637113571167,
176
+ "learning_rate": 1.904e-06,
177
+ "loss": 9.93200454711914,
178
  "step": 120
179
  },
180
  {
181
+ "epoch": 0.025,
182
+ "grad_norm": 2.2896499633789062,
183
+ "learning_rate": 1.984e-06,
184
+ "loss": 9.873931121826171,
185
  "step": 125
186
  },
187
  {
188
+ "epoch": 0.026,
189
+ "grad_norm": 2.198643684387207,
190
+ "learning_rate": 2.064e-06,
191
+ "loss": 9.849259185791016,
192
  "step": 130
193
  },
194
  {
195
+ "epoch": 0.027,
196
+ "grad_norm": 2.191349983215332,
197
+ "learning_rate": 2.144e-06,
198
+ "loss": 9.790320587158202,
199
  "step": 135
200
  },
201
  {
202
+ "epoch": 0.028,
203
+ "grad_norm": 2.097088575363159,
204
+ "learning_rate": 2.2240000000000002e-06,
205
+ "loss": 9.746687316894532,
206
  "step": 140
207
  },
208
  {
209
+ "epoch": 0.029,
210
+ "grad_norm": 2.1079649925231934,
211
+ "learning_rate": 2.304e-06,
212
+ "loss": 9.693339538574218,
213
  "step": 145
214
  },
215
  {
216
+ "epoch": 0.03,
217
+ "grad_norm": 2.0016579627990723,
218
+ "learning_rate": 2.384e-06,
219
+ "loss": 9.655043792724609,
220
  "step": 150
221
  },
222
  {
223
+ "epoch": 0.03,
224
+ "eval_accuracy": 0.0701001221001221,
225
+ "eval_loss": 9.63985824584961,
226
+ "eval_runtime": 10.5292,
227
+ "eval_samples_per_second": 9.497,
228
+ "eval_steps_per_second": 1.615,
229
  "step": 150
230
  },
231
  {
232
+ "epoch": 0.031,
233
+ "grad_norm": 1.9648433923721313,
234
+ "learning_rate": 2.464e-06,
235
+ "loss": 9.62866439819336,
236
  "step": 155
237
  },
238
  {
239
+ "epoch": 0.032,
240
+ "grad_norm": 2.0246658325195312,
241
+ "learning_rate": 2.544e-06,
242
+ "loss": 9.588599395751952,
243
  "step": 160
244
  },
245
  {
246
+ "epoch": 0.033,
247
+ "grad_norm": 1.941605806350708,
248
+ "learning_rate": 2.624e-06,
249
+ "loss": 9.559522247314453,
250
  "step": 165
251
  },
252
  {
253
+ "epoch": 0.034,
254
+ "grad_norm": 2.1081161499023438,
255
+ "learning_rate": 2.704e-06,
256
+ "loss": 9.523811340332031,
257
  "step": 170
258
  },
259
  {
260
+ "epoch": 0.035,
261
+ "grad_norm": 1.9597957134246826,
262
+ "learning_rate": 2.7839999999999995e-06,
263
+ "loss": 9.42935791015625,
264
  "step": 175
265
  },
266
  {
267
+ "epoch": 0.036,
268
+ "grad_norm": 2.00820255279541,
269
+ "learning_rate": 2.8639999999999996e-06,
270
+ "loss": 9.449478149414062,
271
  "step": 180
272
  },
273
  {
274
+ "epoch": 0.037,
275
+ "grad_norm": 2.1553680896759033,
276
+ "learning_rate": 2.9439999999999997e-06,
277
+ "loss": 9.507718658447265,
278
  "step": 185
279
  },
280
  {
281
+ "epoch": 0.038,
282
+ "grad_norm": 2.015054225921631,
283
+ "learning_rate": 3.0239999999999998e-06,
284
+ "loss": 9.442604064941406,
285
  "step": 190
286
  },
287
  {
288
+ "epoch": 0.039,
289
+ "grad_norm": 1.9352600574493408,
290
+ "learning_rate": 3.104e-06,
291
+ "loss": 9.40188980102539,
292
  "step": 195
293
  },
294
  {
295
+ "epoch": 0.04,
296
+ "grad_norm": 1.9175664186477661,
297
+ "learning_rate": 3.184e-06,
298
+ "loss": 9.372711944580079,
299
  "step": 200
300
  },
301
  {
302
+ "epoch": 0.041,
303
+ "grad_norm": 2.047138214111328,
304
+ "learning_rate": 3.2639999999999996e-06,
305
+ "loss": 9.297854614257812,
306
+ "step": 205
307
+ },
308
+ {
309
+ "epoch": 0.042,
310
+ "grad_norm": 1.7625023126602173,
311
+ "learning_rate": 3.3439999999999997e-06,
312
+ "loss": 9.354965209960938,
313
+ "step": 210
314
+ },
315
+ {
316
+ "epoch": 0.043,
317
+ "grad_norm": 1.9414070844650269,
318
+ "learning_rate": 3.4239999999999997e-06,
319
+ "loss": 9.30640106201172,
320
+ "step": 215
321
+ },
322
+ {
323
+ "epoch": 0.044,
324
+ "grad_norm": 1.8544589281082153,
325
+ "learning_rate": 3.504e-06,
326
+ "loss": 9.286300659179688,
327
+ "step": 220
328
+ },
329
+ {
330
+ "epoch": 0.045,
331
+ "grad_norm": 1.864773154258728,
332
+ "learning_rate": 3.584e-06,
333
+ "loss": 9.244451141357422,
334
+ "step": 225
335
+ },
336
+ {
337
+ "epoch": 0.046,
338
+ "grad_norm": 1.9134947061538696,
339
+ "learning_rate": 3.664e-06,
340
+ "loss": 9.235625457763671,
341
+ "step": 230
342
+ },
343
+ {
344
+ "epoch": 0.047,
345
+ "grad_norm": 1.959318995475769,
346
+ "learning_rate": 3.744e-06,
347
+ "loss": 9.202496337890626,
348
+ "step": 235
349
+ },
350
+ {
351
+ "epoch": 0.048,
352
+ "grad_norm": 1.792734146118164,
353
+ "learning_rate": 3.823999999999999e-06,
354
+ "loss": 9.204552459716798,
355
+ "step": 240
356
+ },
357
+ {
358
+ "epoch": 0.049,
359
+ "grad_norm": 1.8003584146499634,
360
+ "learning_rate": 3.903999999999999e-06,
361
+ "loss": 9.14830322265625,
362
+ "step": 245
363
+ },
364
+ {
365
+ "epoch": 0.05,
366
+ "grad_norm": 1.941810131072998,
367
+ "learning_rate": 3.9839999999999995e-06,
368
+ "loss": 9.14439697265625,
369
+ "step": 250
370
+ },
371
+ {
372
+ "epoch": 0.051,
373
+ "grad_norm": 1.7709863185882568,
374
+ "learning_rate": 3.99999300106024e-06,
375
+ "loss": 9.078700256347656,
376
+ "step": 255
377
+ },
378
+ {
379
+ "epoch": 0.052,
380
+ "grad_norm": 1.771018385887146,
381
+ "learning_rate": 3.9999645679514235e-06,
382
+ "loss": 9.125299072265625,
383
+ "step": 260
384
+ },
385
+ {
386
+ "epoch": 0.053,
387
+ "grad_norm": 1.6554689407348633,
388
+ "learning_rate": 3.999914263550513e-06,
389
+ "loss": 9.08728485107422,
390
+ "step": 265
391
+ },
392
+ {
393
+ "epoch": 0.054,
394
+ "grad_norm": 1.889380931854248,
395
+ "learning_rate": 3.9998420884076325e-06,
396
+ "loss": 9.034518432617187,
397
+ "step": 270
398
+ },
399
+ {
400
+ "epoch": 0.055,
401
+ "grad_norm": 1.7578399181365967,
402
+ "learning_rate": 3.999748043312075e-06,
403
+ "loss": 9.05845947265625,
404
+ "step": 275
405
+ },
406
+ {
407
+ "epoch": 0.056,
408
+ "grad_norm": 1.7057148218154907,
409
+ "learning_rate": 3.999632129292304e-06,
410
+ "loss": 9.022759246826173,
411
+ "step": 280
412
+ },
413
+ {
414
+ "epoch": 0.057,
415
+ "grad_norm": 1.7159546613693237,
416
+ "learning_rate": 3.9994943476159364e-06,
417
+ "loss": 8.9812255859375,
418
+ "step": 285
419
+ },
420
+ {
421
+ "epoch": 0.058,
422
+ "grad_norm": 1.7239329814910889,
423
+ "learning_rate": 3.999334699789731e-06,
424
+ "loss": 8.976101684570313,
425
+ "step": 290
426
+ },
427
+ {
428
+ "epoch": 0.059,
429
+ "grad_norm": 1.73225736618042,
430
+ "learning_rate": 3.99915318755957e-06,
431
+ "loss": 8.98138885498047,
432
+ "step": 295
433
+ },
434
+ {
435
+ "epoch": 0.06,
436
+ "grad_norm": 1.7068036794662476,
437
+ "learning_rate": 3.9989498129104425e-06,
438
+ "loss": 8.94490966796875,
439
+ "step": 300
440
+ },
441
+ {
442
+ "epoch": 0.06,
443
+ "eval_accuracy": 0.08137484737484738,
444
+ "eval_loss": 8.956033706665039,
445
+ "eval_runtime": 10.504,
446
+ "eval_samples_per_second": 9.52,
447
+ "eval_steps_per_second": 1.618,
448
+ "step": 300
449
  }
450
  ],
451
  "logging_steps": 5,
452
+ "max_steps": 5000,
453
  "num_input_tokens_seen": 0,
454
  "num_train_epochs": 9223372036854775807,
455
+ "save_steps": 300,
456
  "stateful_callbacks": {
457
  "TrainerControl": {
458
  "args": {
 
460
  "should_evaluate": false,
461
  "should_log": false,
462
  "should_save": true,
463
+ "should_training_stop": false
464
  },
465
  "attributes": {}
466
  }
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:917d0d97de493b58757e5f74362d60fd58076fa34ab6735a0440935a012608ef
3
  size 5265
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:781128b2f6799be638b34a7822cb65221aed10bfe95ca287b92a6464f9518c23
3
  size 5265