aixk commited on
Commit
020a922
·
1 Parent(s): f2717ae

System: Prune old checkpoint folder checkpoint-400 to save space

Browse files
checkpoint-400/config.json DELETED
@@ -1,19 +0,0 @@
1
- {
2
- "architectures": [
3
- "FastPlusForCausalLM"
4
- ],
5
- "dtype": "float32",
6
- "hidden_size": 768,
7
- "initializer_range": 0.02,
8
- "intermediate_size": 2304,
9
- "kd_alpha": 0.4,
10
- "kd_temperature": 2.5,
11
- "max_position_embeddings": 512,
12
- "model_type": "fastplus",
13
- "num_attention_heads": 12,
14
- "num_hidden_layers": 12,
15
- "tie_word_embeddings": true,
16
- "transformers_version": "5.12.0",
17
- "use_cache": false,
18
- "vocab_size": 31968
19
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-400/generation_config.json DELETED
@@ -1,6 +0,0 @@
1
- {
2
- "_from_model_config": true,
3
- "output_attentions": false,
4
- "output_hidden_states": false,
5
- "transformers_version": "5.12.0"
6
- }
 
 
 
 
 
 
 
checkpoint-400/trainer_state.json DELETED
@@ -1,354 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.013333333333333334,
6
- "eval_steps": 500,
7
- "global_step": 400,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.0003333333333333333,
14
- "grad_norm": 243.78553771972656,
15
- "learning_rate": 5.95e-06,
16
- "loss": 166.0732177734375,
17
- "original_ce_loss": 10.3789,
18
- "step": 10
19
- },
20
- {
21
- "epoch": 0.0006666666666666666,
22
- "grad_norm": 239.2476348876953,
23
- "learning_rate": 1.445e-05,
24
- "loss": 165.82977294921875,
25
- "original_ce_loss": 10.3637,
26
- "step": 20
27
- },
28
- {
29
- "epoch": 0.001,
30
- "grad_norm": 244.72525024414062,
31
- "learning_rate": 2.295e-05,
32
- "loss": 165.18355712890624,
33
- "original_ce_loss": 10.3233,
34
- "step": 30
35
- },
36
- {
37
- "epoch": 0.0013333333333333333,
38
- "grad_norm": 270.7417297363281,
39
- "learning_rate": 3.145e-05,
40
- "loss": 163.79927978515624,
41
- "original_ce_loss": 10.2368,
42
- "step": 40
43
- },
44
- {
45
- "epoch": 0.0016666666666666668,
46
- "grad_norm": 326.618408203125,
47
- "learning_rate": 3.9949999999999995e-05,
48
- "loss": 160.3599365234375,
49
- "original_ce_loss": 10.0218,
50
- "step": 50
51
- },
52
- {
53
- "epoch": 0.002,
54
- "grad_norm": 166.10595703125,
55
- "learning_rate": 4.845e-05,
56
- "loss": 151.0793212890625,
57
- "original_ce_loss": 9.4418,
58
- "step": 60
59
- },
60
- {
61
- "epoch": 0.0023333333333333335,
62
- "grad_norm": 54.51128387451172,
63
- "learning_rate": 5.695e-05,
64
- "loss": 142.18406982421874,
65
- "original_ce_loss": 8.8858,
66
- "step": 70
67
- },
68
- {
69
- "epoch": 0.0026666666666666666,
70
- "grad_norm": 45.69681930541992,
71
- "learning_rate": 6.544999999999999e-05,
72
- "loss": 135.513818359375,
73
- "original_ce_loss": 8.4689,
74
- "step": 80
75
- },
76
- {
77
- "epoch": 0.003,
78
- "grad_norm": 44.38578414916992,
79
- "learning_rate": 7.395e-05,
80
- "loss": 129.2341796875,
81
- "original_ce_loss": 8.0765,
82
- "step": 90
83
- },
84
- {
85
- "epoch": 0.0033333333333333335,
86
- "grad_norm": 36.00300979614258,
87
- "learning_rate": 8.245e-05,
88
- "loss": 123.0133544921875,
89
- "original_ce_loss": 7.6877,
90
- "step": 100
91
- },
92
- {
93
- "epoch": 0.0036666666666666666,
94
- "grad_norm": 32.789207458496094,
95
- "learning_rate": 9.094999999999999e-05,
96
- "loss": 117.155908203125,
97
- "original_ce_loss": 7.3216,
98
- "step": 110
99
- },
100
- {
101
- "epoch": 0.004,
102
- "grad_norm": 26.692514419555664,
103
- "learning_rate": 9.86e-05,
104
- "loss": 112.15670166015624,
105
- "original_ce_loss": 7.0091,
106
- "step": 120
107
- },
108
- {
109
- "epoch": 0.004333333333333333,
110
- "grad_norm": 20.985740661621094,
111
- "learning_rate": 0.0001071,
112
- "loss": 108.44638671875,
113
- "original_ce_loss": 6.7772,
114
- "step": 130
115
- },
116
- {
117
- "epoch": 0.004666666666666667,
118
- "grad_norm": NaN,
119
- "learning_rate": 0.00011475,
120
- "loss": 105.64476318359375,
121
- "original_ce_loss": 6.6021,
122
- "step": 140
123
- },
124
- {
125
- "epoch": 0.005,
126
- "grad_norm": 47.759765625,
127
- "learning_rate": 0.000119,
128
- "loss": 104.857763671875,
129
- "original_ce_loss": 6.5529,
130
- "step": 150
131
- },
132
- {
133
- "epoch": 0.005333333333333333,
134
- "grad_norm": 164.30889892578125,
135
- "learning_rate": 0.00012749999999999998,
136
- "loss": 103.3579833984375,
137
- "original_ce_loss": 6.4592,
138
- "step": 160
139
- },
140
- {
141
- "epoch": 0.005666666666666667,
142
- "grad_norm": 11.574602127075195,
143
- "learning_rate": 0.000136,
144
- "loss": 103.28526611328125,
145
- "original_ce_loss": 6.4547,
146
- "step": 170
147
- },
148
- {
149
- "epoch": 0.006,
150
- "grad_norm": 13.660222053527832,
151
- "learning_rate": 0.00014450000000000002,
152
- "loss": 103.16549072265624,
153
- "original_ce_loss": 6.4472,
154
- "step": 180
155
- },
156
- {
157
- "epoch": 0.006333333333333333,
158
- "grad_norm": 15.656424522399902,
159
- "learning_rate": 0.00015299999999999998,
160
- "loss": 102.62147216796875,
161
- "original_ce_loss": 6.4132,
162
- "step": 190
163
- },
164
- {
165
- "epoch": 0.006666666666666667,
166
- "grad_norm": 18.646780014038086,
167
- "learning_rate": 0.0001615,
168
- "loss": 101.59368896484375,
169
- "original_ce_loss": 6.3489,
170
- "step": 200
171
- },
172
- {
173
- "epoch": 0.007,
174
- "grad_norm": 39.721866607666016,
175
- "learning_rate": 0.00017,
176
- "loss": 100.74405517578126,
177
- "original_ce_loss": 6.2958,
178
- "step": 210
179
- },
180
- {
181
- "epoch": 0.007333333333333333,
182
- "grad_norm": 16.912254333496094,
183
- "learning_rate": 0.00017849999999999997,
184
- "loss": 99.42055053710938,
185
- "original_ce_loss": 6.2131,
186
- "step": 220
187
- },
188
- {
189
- "epoch": 0.007666666666666666,
190
- "grad_norm": 17.74278450012207,
191
- "learning_rate": 0.000187,
192
- "loss": 97.92646484375,
193
- "original_ce_loss": 6.1197,
194
- "step": 230
195
- },
196
- {
197
- "epoch": 0.008,
198
- "grad_norm": 18.32769775390625,
199
- "learning_rate": 0.0001955,
200
- "loss": 96.8761962890625,
201
- "original_ce_loss": 6.0541,
202
- "step": 240
203
- },
204
- {
205
- "epoch": 0.008333333333333333,
206
- "grad_norm": 16.8792781829834,
207
- "learning_rate": 0.00020399999999999997,
208
- "loss": 95.25513305664063,
209
- "original_ce_loss": 5.9528,
210
- "step": 250
211
- },
212
- {
213
- "epoch": 0.008666666666666666,
214
- "grad_norm": 12.205955505371094,
215
- "learning_rate": 0.0002125,
216
- "loss": 93.572900390625,
217
- "original_ce_loss": 5.8476,
218
- "step": 260
219
- },
220
- {
221
- "epoch": 0.009,
222
- "grad_norm": 13.931303024291992,
223
- "learning_rate": 0.000221,
224
- "loss": 91.66510009765625,
225
- "original_ce_loss": 5.7284,
226
- "step": 270
227
- },
228
- {
229
- "epoch": 0.009333333333333334,
230
- "grad_norm": 26.242258071899414,
231
- "learning_rate": 0.0002295,
232
- "loss": 90.45693969726562,
233
- "original_ce_loss": 5.6529,
234
- "step": 280
235
- },
236
- {
237
- "epoch": 0.009666666666666667,
238
- "grad_norm": 16.811609268188477,
239
- "learning_rate": 0.000238,
240
- "loss": 91.2010009765625,
241
- "original_ce_loss": 5.6994,
242
- "step": 290
243
- },
244
- {
245
- "epoch": 0.01,
246
- "grad_norm": 25.731582641601562,
247
- "learning_rate": 0.00024565,
248
- "loss": 97.92516479492187,
249
- "original_ce_loss": 6.1197,
250
- "step": 300
251
- },
252
- {
253
- "epoch": 0.010333333333333333,
254
- "grad_norm": 36.38114547729492,
255
- "learning_rate": 0.00025414999999999997,
256
- "loss": 98.45838623046875,
257
- "original_ce_loss": 6.153,
258
- "step": 310
259
- },
260
- {
261
- "epoch": 0.010666666666666666,
262
- "grad_norm": NaN,
263
- "learning_rate": 0.00026179999999999997,
264
- "loss": 95.69706420898437,
265
- "original_ce_loss": 5.9804,
266
- "step": 320
267
- },
268
- {
269
- "epoch": 0.011,
270
- "grad_norm": 51.799739837646484,
271
- "learning_rate": 0.00026944999999999996,
272
- "loss": 91.98704223632812,
273
- "original_ce_loss": 5.7485,
274
- "step": 330
275
- },
276
- {
277
- "epoch": 0.011333333333333334,
278
- "grad_norm": 22.107471466064453,
279
- "learning_rate": 0.00027795,
280
- "loss": 85.44754028320312,
281
- "original_ce_loss": 5.3398,
282
- "step": 340
283
- },
284
- {
285
- "epoch": 0.011666666666666667,
286
- "grad_norm": 30.49631118774414,
287
- "learning_rate": 0.00028645,
288
- "loss": 82.21802978515625,
289
- "original_ce_loss": 5.138,
290
- "step": 350
291
- },
292
- {
293
- "epoch": 0.012,
294
- "grad_norm": 46.481483459472656,
295
- "learning_rate": 0.00029495,
296
- "loss": 81.45587768554688,
297
- "original_ce_loss": 5.0904,
298
- "step": 360
299
- },
300
- {
301
- "epoch": 0.012333333333333333,
302
- "grad_norm": 35.435367584228516,
303
- "learning_rate": 0.00030345,
304
- "loss": 80.58915405273437,
305
- "original_ce_loss": 5.0362,
306
- "step": 370
307
- },
308
- {
309
- "epoch": 0.012666666666666666,
310
- "grad_norm": 23.500009536743164,
311
- "learning_rate": 0.00031109999999999997,
312
- "loss": 79.21079711914062,
313
- "original_ce_loss": 4.95,
314
- "step": 380
315
- },
316
- {
317
- "epoch": 0.013,
318
- "grad_norm": 36.313453674316406,
319
- "learning_rate": 0.00031959999999999996,
320
- "loss": 77.30345458984375,
321
- "original_ce_loss": 4.8308,
322
- "step": 390
323
- },
324
- {
325
- "epoch": 0.013333333333333334,
326
- "grad_norm": 22.148427963256836,
327
- "learning_rate": 0.0003281,
328
- "loss": 75.77711181640625,
329
- "original_ce_loss": 4.7354,
330
- "step": 400
331
- }
332
- ],
333
- "logging_steps": 10,
334
- "max_steps": 30000,
335
- "num_input_tokens_seen": 0,
336
- "num_train_epochs": 9223372036854775807,
337
- "save_steps": 200,
338
- "stateful_callbacks": {
339
- "TrainerControl": {
340
- "args": {
341
- "should_epoch_stop": false,
342
- "should_evaluate": false,
343
- "should_log": false,
344
- "should_save": true,
345
- "should_training_stop": false
346
- },
347
- "attributes": {}
348
- }
349
- },
350
- "total_flos": 2.94745836355584e+16,
351
- "train_batch_size": 16,
352
- "trial_name": null,
353
- "trial_params": null
354
- }