3v324v23 commited on
Commit
79cb61b
·
1 Parent(s): 5835438

Model first push

Browse files
checkpoint-17500/config.json ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "yip-i/uaspeech-pretrained",
3
+ "activation_dropout": 0.0,
4
+ "adapter_kernel_size": 3,
5
+ "adapter_stride": 2,
6
+ "add_adapter": false,
7
+ "apply_spec_augment": true,
8
+ "architectures": [
9
+ "Wav2Vec2ForCTC"
10
+ ],
11
+ "attention_dropout": 0.0,
12
+ "bos_token_id": 1,
13
+ "classifier_proj_size": 256,
14
+ "codevector_dim": 256,
15
+ "contrastive_logits_temperature": 0.1,
16
+ "conv_bias": true,
17
+ "conv_dim": [
18
+ 512,
19
+ 512,
20
+ 512,
21
+ 512,
22
+ 512,
23
+ 512,
24
+ 512
25
+ ],
26
+ "conv_kernel": [
27
+ 10,
28
+ 3,
29
+ 3,
30
+ 3,
31
+ 3,
32
+ 2,
33
+ 2
34
+ ],
35
+ "conv_stride": [
36
+ 5,
37
+ 2,
38
+ 2,
39
+ 2,
40
+ 2,
41
+ 2,
42
+ 2
43
+ ],
44
+ "ctc_loss_reduction": "mean",
45
+ "ctc_zero_infinity": false,
46
+ "diversity_loss_weight": 0.1,
47
+ "do_stable_layer_norm": true,
48
+ "eos_token_id": 2,
49
+ "feat_extract_activation": "gelu",
50
+ "feat_extract_dropout": 0.0,
51
+ "feat_extract_norm": "layer",
52
+ "feat_proj_dropout": 0.0,
53
+ "feat_quantizer_dropout": 0.0,
54
+ "final_dropout": 0.0,
55
+ "hidden_act": "gelu",
56
+ "hidden_dropout": 0.0,
57
+ "hidden_dropout_prob": 0.0,
58
+ "hidden_size": 768,
59
+ "initializer_range": 0.02,
60
+ "intermediate_size": 3072,
61
+ "layer_norm_eps": 1e-05,
62
+ "layerdrop": 0.0,
63
+ "mask_feature_length": 10,
64
+ "mask_feature_min_masks": 0,
65
+ "mask_feature_prob": 0.0,
66
+ "mask_time_length": 10,
67
+ "mask_time_min_masks": 2,
68
+ "mask_time_prob": 0.65,
69
+ "model_type": "wav2vec2",
70
+ "num_adapter_layers": 3,
71
+ "num_attention_heads": 12,
72
+ "num_codevector_groups": 2,
73
+ "num_codevectors_per_group": 320,
74
+ "num_conv_pos_embedding_groups": 16,
75
+ "num_conv_pos_embeddings": 128,
76
+ "num_feat_extract_layers": 7,
77
+ "num_hidden_layers": 12,
78
+ "num_negatives": 100,
79
+ "output_hidden_size": 768,
80
+ "pad_token_id": 29,
81
+ "proj_codevector_dim": 256,
82
+ "tdnn_dilation": [
83
+ 1,
84
+ 2,
85
+ 3,
86
+ 1,
87
+ 1
88
+ ],
89
+ "tdnn_dim": [
90
+ 512,
91
+ 512,
92
+ 512,
93
+ 512,
94
+ 1500
95
+ ],
96
+ "tdnn_kernel": [
97
+ 5,
98
+ 3,
99
+ 3,
100
+ 1,
101
+ 1
102
+ ],
103
+ "torch_dtype": "float32",
104
+ "transformers_version": "4.23.1",
105
+ "use_weighted_layer_sum": false,
106
+ "vocab_size": 32,
107
+ "xvector_output_dim": 512
108
+ }
checkpoint-17500/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:536b181a3a322aaf0b4d78d18751437c33b961e6655cc4c049ea3b207faaa6e9
3
+ size 721685265
checkpoint-17500/preprocessor_config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_normalize": true,
3
+ "feature_extractor_type": "Wav2Vec2FeatureExtractor",
4
+ "feature_size": 1,
5
+ "padding_side": "right",
6
+ "padding_value": 0.0,
7
+ "return_attention_mask": false,
8
+ "sampling_rate": 16000
9
+ }
checkpoint-17500/pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1d6ff44d661cae8cb3e5d07e2957fb41f2ea18f4544836f52130d34eddfd5089
3
+ size 377702321
checkpoint-17500/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e06a5780cb4c5a5ab12f02e995c2100f2d68a7b1512c016d743690d9188cb6c
3
+ size 14567
checkpoint-17500/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f78aeb70f5e1d3045d4779dc880fe517a3726f5e89e28a06e1feee3b56dbde0
3
+ size 623
checkpoint-17500/trainer_state.json ADDED
@@ -0,0 +1,541 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 28.40909090909091,
5
+ "global_step": 17500,
6
+ "is_hyper_param_search": false,
7
+ "is_local_process_zero": true,
8
+ "is_world_process_zero": true,
9
+ "log_history": [
10
+ {
11
+ "epoch": 0.81,
12
+ "learning_rate": 5e-05,
13
+ "loss": 8.6952,
14
+ "step": 500
15
+ },
16
+ {
17
+ "epoch": 0.81,
18
+ "eval_loss": 3.0705761909484863,
19
+ "eval_runtime": 8.0191,
20
+ "eval_samples_per_second": 25.813,
21
+ "eval_steps_per_second": 3.242,
22
+ "eval_wer": 1.1315192743764173,
23
+ "step": 500
24
+ },
25
+ {
26
+ "epoch": 1.62,
27
+ "learning_rate": 0.0001,
28
+ "loss": 3.2723,
29
+ "step": 1000
30
+ },
31
+ {
32
+ "epoch": 1.62,
33
+ "eval_loss": 2.9638686180114746,
34
+ "eval_runtime": 8.0684,
35
+ "eval_samples_per_second": 25.656,
36
+ "eval_steps_per_second": 3.222,
37
+ "eval_wer": 1.0,
38
+ "step": 1000
39
+ },
40
+ {
41
+ "epoch": 2.44,
42
+ "learning_rate": 9.713958810068651e-05,
43
+ "loss": 3.2191,
44
+ "step": 1500
45
+ },
46
+ {
47
+ "epoch": 2.44,
48
+ "eval_loss": 3.0145716667175293,
49
+ "eval_runtime": 8.0711,
50
+ "eval_samples_per_second": 25.647,
51
+ "eval_steps_per_second": 3.221,
52
+ "eval_wer": 1.0,
53
+ "step": 1500
54
+ },
55
+ {
56
+ "epoch": 3.25,
57
+ "learning_rate": 9.4279176201373e-05,
58
+ "loss": 3.0698,
59
+ "step": 2000
60
+ },
61
+ {
62
+ "epoch": 3.25,
63
+ "eval_loss": 2.9972307682037354,
64
+ "eval_runtime": 8.0326,
65
+ "eval_samples_per_second": 25.77,
66
+ "eval_steps_per_second": 3.237,
67
+ "eval_wer": 1.3356009070294785,
68
+ "step": 2000
69
+ },
70
+ {
71
+ "epoch": 4.06,
72
+ "learning_rate": 9.14187643020595e-05,
73
+ "loss": 3.1918,
74
+ "step": 2500
75
+ },
76
+ {
77
+ "epoch": 4.06,
78
+ "eval_loss": 2.9555039405822754,
79
+ "eval_runtime": 8.0447,
80
+ "eval_samples_per_second": 25.731,
81
+ "eval_steps_per_second": 3.232,
82
+ "eval_wer": 1.0,
83
+ "step": 2500
84
+ },
85
+ {
86
+ "epoch": 4.87,
87
+ "learning_rate": 8.8558352402746e-05,
88
+ "loss": 3.0932,
89
+ "step": 3000
90
+ },
91
+ {
92
+ "epoch": 4.87,
93
+ "eval_loss": 2.9776482582092285,
94
+ "eval_runtime": 8.053,
95
+ "eval_samples_per_second": 25.705,
96
+ "eval_steps_per_second": 3.229,
97
+ "eval_wer": 1.0,
98
+ "step": 3000
99
+ },
100
+ {
101
+ "epoch": 5.68,
102
+ "learning_rate": 8.569794050343249e-05,
103
+ "loss": 3.2271,
104
+ "step": 3500
105
+ },
106
+ {
107
+ "epoch": 5.68,
108
+ "eval_loss": 3.055955410003662,
109
+ "eval_runtime": 8.1309,
110
+ "eval_samples_per_second": 25.458,
111
+ "eval_steps_per_second": 3.198,
112
+ "eval_wer": 1.3378684807256236,
113
+ "step": 3500
114
+ },
115
+ {
116
+ "epoch": 6.49,
117
+ "learning_rate": 8.283752860411899e-05,
118
+ "loss": 3.2925,
119
+ "step": 4000
120
+ },
121
+ {
122
+ "epoch": 6.49,
123
+ "eval_loss": 2.9724416732788086,
124
+ "eval_runtime": 8.0677,
125
+ "eval_samples_per_second": 25.658,
126
+ "eval_steps_per_second": 3.223,
127
+ "eval_wer": 1.3900226757369614,
128
+ "step": 4000
129
+ },
130
+ {
131
+ "epoch": 7.31,
132
+ "learning_rate": 7.99771167048055e-05,
133
+ "loss": 3.2195,
134
+ "step": 4500
135
+ },
136
+ {
137
+ "epoch": 7.31,
138
+ "eval_loss": 3.5231425762176514,
139
+ "eval_runtime": 8.1433,
140
+ "eval_samples_per_second": 25.42,
141
+ "eval_steps_per_second": 3.193,
142
+ "eval_wer": 1.0,
143
+ "step": 4500
144
+ },
145
+ {
146
+ "epoch": 8.12,
147
+ "learning_rate": 7.711670480549199e-05,
148
+ "loss": 3.3582,
149
+ "step": 5000
150
+ },
151
+ {
152
+ "epoch": 8.12,
153
+ "eval_loss": 3.749286413192749,
154
+ "eval_runtime": 8.0117,
155
+ "eval_samples_per_second": 25.837,
156
+ "eval_steps_per_second": 3.245,
157
+ "eval_wer": 1.0249433106575965,
158
+ "step": 5000
159
+ },
160
+ {
161
+ "epoch": 8.93,
162
+ "learning_rate": 7.42562929061785e-05,
163
+ "loss": 3.3233,
164
+ "step": 5500
165
+ },
166
+ {
167
+ "epoch": 8.93,
168
+ "eval_loss": 3.3524105548858643,
169
+ "eval_runtime": 8.2055,
170
+ "eval_samples_per_second": 25.227,
171
+ "eval_steps_per_second": 3.169,
172
+ "eval_wer": 1.0,
173
+ "step": 5500
174
+ },
175
+ {
176
+ "epoch": 9.74,
177
+ "learning_rate": 7.1395881006865e-05,
178
+ "loss": 3.1679,
179
+ "step": 6000
180
+ },
181
+ {
182
+ "epoch": 9.74,
183
+ "eval_loss": 3.186889410018921,
184
+ "eval_runtime": 8.0182,
185
+ "eval_samples_per_second": 25.816,
186
+ "eval_steps_per_second": 3.243,
187
+ "eval_wer": 1.3877551020408163,
188
+ "step": 6000
189
+ },
190
+ {
191
+ "epoch": 10.55,
192
+ "learning_rate": 6.853546910755149e-05,
193
+ "loss": 3.1393,
194
+ "step": 6500
195
+ },
196
+ {
197
+ "epoch": 10.55,
198
+ "eval_loss": 3.0084877014160156,
199
+ "eval_runtime": 8.0014,
200
+ "eval_samples_per_second": 25.871,
201
+ "eval_steps_per_second": 3.249,
202
+ "eval_wer": 1.2947845804988662,
203
+ "step": 6500
204
+ },
205
+ {
206
+ "epoch": 11.36,
207
+ "learning_rate": 6.5675057208238e-05,
208
+ "loss": 3.1699,
209
+ "step": 7000
210
+ },
211
+ {
212
+ "epoch": 11.36,
213
+ "eval_loss": 2.997206449508667,
214
+ "eval_runtime": 8.1145,
215
+ "eval_samples_per_second": 25.51,
216
+ "eval_steps_per_second": 3.204,
217
+ "eval_wer": 1.1337868480725624,
218
+ "step": 7000
219
+ },
220
+ {
221
+ "epoch": 12.18,
222
+ "learning_rate": 6.281464530892449e-05,
223
+ "loss": 3.3382,
224
+ "step": 7500
225
+ },
226
+ {
227
+ "epoch": 12.18,
228
+ "eval_loss": 2.9686625003814697,
229
+ "eval_runtime": 8.0023,
230
+ "eval_samples_per_second": 25.868,
231
+ "eval_steps_per_second": 3.249,
232
+ "eval_wer": 1.3877551020408163,
233
+ "step": 7500
234
+ },
235
+ {
236
+ "epoch": 12.99,
237
+ "learning_rate": 5.9954233409610984e-05,
238
+ "loss": 3.0454,
239
+ "step": 8000
240
+ },
241
+ {
242
+ "epoch": 12.99,
243
+ "eval_loss": 2.968928337097168,
244
+ "eval_runtime": 8.2051,
245
+ "eval_samples_per_second": 25.228,
246
+ "eval_steps_per_second": 3.169,
247
+ "eval_wer": 1.383219954648526,
248
+ "step": 8000
249
+ },
250
+ {
251
+ "epoch": 13.8,
252
+ "learning_rate": 5.709382151029748e-05,
253
+ "loss": 3.0609,
254
+ "step": 8500
255
+ },
256
+ {
257
+ "epoch": 13.8,
258
+ "eval_loss": 2.902815580368042,
259
+ "eval_runtime": 8.0338,
260
+ "eval_samples_per_second": 25.766,
261
+ "eval_steps_per_second": 3.236,
262
+ "eval_wer": 1.3900226757369614,
263
+ "step": 8500
264
+ },
265
+ {
266
+ "epoch": 14.61,
267
+ "learning_rate": 5.423340961098399e-05,
268
+ "loss": 3.0224,
269
+ "step": 9000
270
+ },
271
+ {
272
+ "epoch": 14.61,
273
+ "eval_loss": 2.9064831733703613,
274
+ "eval_runtime": 8.2142,
275
+ "eval_samples_per_second": 25.2,
276
+ "eval_steps_per_second": 3.165,
277
+ "eval_wer": 1.3877551020408163,
278
+ "step": 9000
279
+ },
280
+ {
281
+ "epoch": 15.42,
282
+ "learning_rate": 5.137299771167048e-05,
283
+ "loss": 3.0156,
284
+ "step": 9500
285
+ },
286
+ {
287
+ "epoch": 15.42,
288
+ "eval_loss": 2.8854894638061523,
289
+ "eval_runtime": 7.9943,
290
+ "eval_samples_per_second": 25.893,
291
+ "eval_steps_per_second": 3.252,
292
+ "eval_wer": 1.3900226757369614,
293
+ "step": 9500
294
+ },
295
+ {
296
+ "epoch": 16.23,
297
+ "learning_rate": 4.851258581235698e-05,
298
+ "loss": 3.0317,
299
+ "step": 10000
300
+ },
301
+ {
302
+ "epoch": 16.23,
303
+ "eval_loss": 2.8956820964813232,
304
+ "eval_runtime": 8.1264,
305
+ "eval_samples_per_second": 25.473,
306
+ "eval_steps_per_second": 3.199,
307
+ "eval_wer": 1.3900226757369614,
308
+ "step": 10000
309
+ },
310
+ {
311
+ "epoch": 17.05,
312
+ "learning_rate": 4.565217391304348e-05,
313
+ "loss": 3.0184,
314
+ "step": 10500
315
+ },
316
+ {
317
+ "epoch": 17.05,
318
+ "eval_loss": 2.88946270942688,
319
+ "eval_runtime": 8.0387,
320
+ "eval_samples_per_second": 25.751,
321
+ "eval_steps_per_second": 3.234,
322
+ "eval_wer": 1.3900226757369614,
323
+ "step": 10500
324
+ },
325
+ {
326
+ "epoch": 17.86,
327
+ "learning_rate": 4.279176201372998e-05,
328
+ "loss": 3.0852,
329
+ "step": 11000
330
+ },
331
+ {
332
+ "epoch": 17.86,
333
+ "eval_loss": 2.8936383724212646,
334
+ "eval_runtime": 8.1682,
335
+ "eval_samples_per_second": 25.342,
336
+ "eval_steps_per_second": 3.183,
337
+ "eval_wer": 1.3900226757369614,
338
+ "step": 11000
339
+ },
340
+ {
341
+ "epoch": 18.67,
342
+ "learning_rate": 3.993135011441648e-05,
343
+ "loss": 3.0017,
344
+ "step": 11500
345
+ },
346
+ {
347
+ "epoch": 18.67,
348
+ "eval_loss": 2.8695127964019775,
349
+ "eval_runtime": 8.169,
350
+ "eval_samples_per_second": 25.34,
351
+ "eval_steps_per_second": 3.183,
352
+ "eval_wer": 1.3900226757369614,
353
+ "step": 11500
354
+ },
355
+ {
356
+ "epoch": 19.48,
357
+ "learning_rate": 3.707093821510298e-05,
358
+ "loss": 2.9337,
359
+ "step": 12000
360
+ },
361
+ {
362
+ "epoch": 19.48,
363
+ "eval_loss": 2.8768134117126465,
364
+ "eval_runtime": 8.1024,
365
+ "eval_samples_per_second": 25.548,
366
+ "eval_steps_per_second": 3.209,
367
+ "eval_wer": 1.3900226757369614,
368
+ "step": 12000
369
+ },
370
+ {
371
+ "epoch": 20.29,
372
+ "learning_rate": 3.421052631578947e-05,
373
+ "loss": 3.0017,
374
+ "step": 12500
375
+ },
376
+ {
377
+ "epoch": 20.29,
378
+ "eval_loss": 2.8580081462860107,
379
+ "eval_runtime": 8.1811,
380
+ "eval_samples_per_second": 25.302,
381
+ "eval_steps_per_second": 3.178,
382
+ "eval_wer": 1.3900226757369614,
383
+ "step": 12500
384
+ },
385
+ {
386
+ "epoch": 21.1,
387
+ "learning_rate": 3.135011441647597e-05,
388
+ "loss": 2.9472,
389
+ "step": 13000
390
+ },
391
+ {
392
+ "epoch": 21.1,
393
+ "eval_loss": 2.839965343475342,
394
+ "eval_runtime": 8.2839,
395
+ "eval_samples_per_second": 24.988,
396
+ "eval_steps_per_second": 3.139,
397
+ "eval_wer": 1.3900226757369614,
398
+ "step": 13000
399
+ },
400
+ {
401
+ "epoch": 21.92,
402
+ "learning_rate": 2.8489702517162476e-05,
403
+ "loss": 3.0214,
404
+ "step": 13500
405
+ },
406
+ {
407
+ "epoch": 21.92,
408
+ "eval_loss": 2.856123924255371,
409
+ "eval_runtime": 8.1176,
410
+ "eval_samples_per_second": 25.5,
411
+ "eval_steps_per_second": 3.203,
412
+ "eval_wer": 1.3900226757369614,
413
+ "step": 13500
414
+ },
415
+ {
416
+ "epoch": 22.73,
417
+ "learning_rate": 2.562929061784897e-05,
418
+ "loss": 2.9336,
419
+ "step": 14000
420
+ },
421
+ {
422
+ "epoch": 22.73,
423
+ "eval_loss": 2.879206895828247,
424
+ "eval_runtime": 8.1391,
425
+ "eval_samples_per_second": 25.433,
426
+ "eval_steps_per_second": 3.194,
427
+ "eval_wer": 1.3900226757369614,
428
+ "step": 14000
429
+ },
430
+ {
431
+ "epoch": 23.54,
432
+ "learning_rate": 2.276887871853547e-05,
433
+ "loss": 3.0134,
434
+ "step": 14500
435
+ },
436
+ {
437
+ "epoch": 23.54,
438
+ "eval_loss": 2.8472487926483154,
439
+ "eval_runtime": 8.0887,
440
+ "eval_samples_per_second": 25.591,
441
+ "eval_steps_per_second": 3.214,
442
+ "eval_wer": 1.3900226757369614,
443
+ "step": 14500
444
+ },
445
+ {
446
+ "epoch": 24.35,
447
+ "learning_rate": 1.990846681922197e-05,
448
+ "loss": 2.9433,
449
+ "step": 15000
450
+ },
451
+ {
452
+ "epoch": 24.35,
453
+ "eval_loss": 2.8818936347961426,
454
+ "eval_runtime": 8.193,
455
+ "eval_samples_per_second": 25.266,
456
+ "eval_steps_per_second": 3.173,
457
+ "eval_wer": 1.3900226757369614,
458
+ "step": 15000
459
+ },
460
+ {
461
+ "epoch": 25.16,
462
+ "learning_rate": 1.7048054919908468e-05,
463
+ "loss": 2.8536,
464
+ "step": 15500
465
+ },
466
+ {
467
+ "epoch": 25.16,
468
+ "eval_loss": 2.837463140487671,
469
+ "eval_runtime": 8.1663,
470
+ "eval_samples_per_second": 25.348,
471
+ "eval_steps_per_second": 3.184,
472
+ "eval_wer": 1.3900226757369614,
473
+ "step": 15500
474
+ },
475
+ {
476
+ "epoch": 25.97,
477
+ "learning_rate": 1.4187643020594965e-05,
478
+ "loss": 2.8742,
479
+ "step": 16000
480
+ },
481
+ {
482
+ "epoch": 25.97,
483
+ "eval_loss": 2.857389450073242,
484
+ "eval_runtime": 8.194,
485
+ "eval_samples_per_second": 25.262,
486
+ "eval_steps_per_second": 3.173,
487
+ "eval_wer": 1.3900226757369614,
488
+ "step": 16000
489
+ },
490
+ {
491
+ "epoch": 26.79,
492
+ "learning_rate": 1.1327231121281464e-05,
493
+ "loss": 2.8298,
494
+ "step": 16500
495
+ },
496
+ {
497
+ "epoch": 26.79,
498
+ "eval_loss": 2.982081651687622,
499
+ "eval_runtime": 8.1935,
500
+ "eval_samples_per_second": 25.264,
501
+ "eval_steps_per_second": 3.173,
502
+ "eval_wer": 1.3900226757369614,
503
+ "step": 16500
504
+ },
505
+ {
506
+ "epoch": 27.6,
507
+ "learning_rate": 8.466819221967964e-06,
508
+ "loss": 2.7439,
509
+ "step": 17000
510
+ },
511
+ {
512
+ "epoch": 27.6,
513
+ "eval_loss": 3.2741587162017822,
514
+ "eval_runtime": 8.2267,
515
+ "eval_samples_per_second": 25.162,
516
+ "eval_steps_per_second": 3.16,
517
+ "eval_wer": 1.3900226757369614,
518
+ "step": 17000
519
+ },
520
+ {
521
+ "epoch": 28.41,
522
+ "learning_rate": 5.606407322654463e-06,
523
+ "loss": 2.7008,
524
+ "step": 17500
525
+ },
526
+ {
527
+ "epoch": 28.41,
528
+ "eval_loss": 3.365966320037842,
529
+ "eval_runtime": 8.203,
530
+ "eval_samples_per_second": 25.235,
531
+ "eval_steps_per_second": 3.17,
532
+ "eval_wer": 1.3900226757369614,
533
+ "step": 17500
534
+ }
535
+ ],
536
+ "max_steps": 18480,
537
+ "num_train_epochs": 30,
538
+ "total_flos": 3.9780154122268984e+18,
539
+ "trial_name": null,
540
+ "trial_params": null
541
+ }
checkpoint-17500/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:151803200d49964e6d9f1364dede5203d89e69881055e526220d7cecd61ffdfb
3
+ size 3375
config.json ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "yip-i/uaspeech-pretrained",
3
+ "activation_dropout": 0.0,
4
+ "adapter_kernel_size": 3,
5
+ "adapter_stride": 2,
6
+ "add_adapter": false,
7
+ "apply_spec_augment": true,
8
+ "architectures": [
9
+ "Wav2Vec2ForCTC"
10
+ ],
11
+ "attention_dropout": 0.0,
12
+ "bos_token_id": 1,
13
+ "classifier_proj_size": 256,
14
+ "codevector_dim": 256,
15
+ "contrastive_logits_temperature": 0.1,
16
+ "conv_bias": true,
17
+ "conv_dim": [
18
+ 512,
19
+ 512,
20
+ 512,
21
+ 512,
22
+ 512,
23
+ 512,
24
+ 512
25
+ ],
26
+ "conv_kernel": [
27
+ 10,
28
+ 3,
29
+ 3,
30
+ 3,
31
+ 3,
32
+ 2,
33
+ 2
34
+ ],
35
+ "conv_stride": [
36
+ 5,
37
+ 2,
38
+ 2,
39
+ 2,
40
+ 2,
41
+ 2,
42
+ 2
43
+ ],
44
+ "ctc_loss_reduction": "mean",
45
+ "ctc_zero_infinity": false,
46
+ "diversity_loss_weight": 0.1,
47
+ "do_stable_layer_norm": true,
48
+ "eos_token_id": 2,
49
+ "feat_extract_activation": "gelu",
50
+ "feat_extract_dropout": 0.0,
51
+ "feat_extract_norm": "layer",
52
+ "feat_proj_dropout": 0.0,
53
+ "feat_quantizer_dropout": 0.0,
54
+ "final_dropout": 0.0,
55
+ "hidden_act": "gelu",
56
+ "hidden_dropout": 0.0,
57
+ "hidden_dropout_prob": 0.0,
58
+ "hidden_size": 768,
59
+ "initializer_range": 0.02,
60
+ "intermediate_size": 3072,
61
+ "layer_norm_eps": 1e-05,
62
+ "layerdrop": 0.0,
63
+ "mask_feature_length": 10,
64
+ "mask_feature_min_masks": 0,
65
+ "mask_feature_prob": 0.0,
66
+ "mask_time_length": 10,
67
+ "mask_time_min_masks": 2,
68
+ "mask_time_prob": 0.65,
69
+ "model_type": "wav2vec2",
70
+ "num_adapter_layers": 3,
71
+ "num_attention_heads": 12,
72
+ "num_codevector_groups": 2,
73
+ "num_codevectors_per_group": 320,
74
+ "num_conv_pos_embedding_groups": 16,
75
+ "num_conv_pos_embeddings": 128,
76
+ "num_feat_extract_layers": 7,
77
+ "num_hidden_layers": 12,
78
+ "num_negatives": 100,
79
+ "output_hidden_size": 768,
80
+ "pad_token_id": 29,
81
+ "proj_codevector_dim": 256,
82
+ "tdnn_dilation": [
83
+ 1,
84
+ 2,
85
+ 3,
86
+ 1,
87
+ 1
88
+ ],
89
+ "tdnn_dim": [
90
+ 512,
91
+ 512,
92
+ 512,
93
+ 512,
94
+ 1500
95
+ ],
96
+ "tdnn_kernel": [
97
+ 5,
98
+ 3,
99
+ 3,
100
+ 1,
101
+ 1
102
+ ],
103
+ "torch_dtype": "float32",
104
+ "transformers_version": "4.23.1",
105
+ "use_weighted_layer_sum": false,
106
+ "vocab_size": 32,
107
+ "xvector_output_dim": 512
108
+ }
optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:16c9cd721072f1285d9a09541b16d36154b0a65769ba2304ebd3cf1a47a3ccd9
3
+ size 721685265
preprocessor_config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_normalize": true,
3
+ "feature_extractor_type": "Wav2Vec2FeatureExtractor",
4
+ "feature_size": 1,
5
+ "padding_side": "right",
6
+ "padding_value": 0.0,
7
+ "return_attention_mask": false,
8
+ "sampling_rate": 16000
9
+ }
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c3b8ce0b861b34249acd899833cc5d17966f7c3820981c93099f3388451d2d17
3
+ size 377702321
rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:49daf610bd0be93ec34116c5dabb671af4b737094b6d5999d22a7ff4458cd59e
3
+ size 14503
runs/Nov14_01-43-39_70156f070b9c/1668390306.856677/events.out.tfevents.1668390306.70156f070b9c.85.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ca0942bb58cf7715bde732b0a6e3d088ea04147768ab715db2dee9de56e0a1f
3
+ size 5471
runs/Nov14_01-43-39_70156f070b9c/events.out.tfevents.1668390306.70156f070b9c.85.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:62b5c681d69c7373b8ec47ffc5e62b6cbd4ccfd518965e2f66ceae9414986bf6
3
+ size 22613
scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b5d4a13226e6eeef330f9ea7271fdddd531065e7435f741b314c8ee6f8e1c979
3
+ size 623
trainer_state.json ADDED
@@ -0,0 +1,556 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 29.22077922077922,
5
+ "global_step": 18000,
6
+ "is_hyper_param_search": false,
7
+ "is_local_process_zero": true,
8
+ "is_world_process_zero": true,
9
+ "log_history": [
10
+ {
11
+ "epoch": 0.81,
12
+ "learning_rate": 5e-05,
13
+ "loss": 8.6952,
14
+ "step": 500
15
+ },
16
+ {
17
+ "epoch": 0.81,
18
+ "eval_loss": 3.0705761909484863,
19
+ "eval_runtime": 8.0191,
20
+ "eval_samples_per_second": 25.813,
21
+ "eval_steps_per_second": 3.242,
22
+ "eval_wer": 1.1315192743764173,
23
+ "step": 500
24
+ },
25
+ {
26
+ "epoch": 1.62,
27
+ "learning_rate": 0.0001,
28
+ "loss": 3.2723,
29
+ "step": 1000
30
+ },
31
+ {
32
+ "epoch": 1.62,
33
+ "eval_loss": 2.9638686180114746,
34
+ "eval_runtime": 8.0684,
35
+ "eval_samples_per_second": 25.656,
36
+ "eval_steps_per_second": 3.222,
37
+ "eval_wer": 1.0,
38
+ "step": 1000
39
+ },
40
+ {
41
+ "epoch": 2.44,
42
+ "learning_rate": 9.713958810068651e-05,
43
+ "loss": 3.2191,
44
+ "step": 1500
45
+ },
46
+ {
47
+ "epoch": 2.44,
48
+ "eval_loss": 3.0145716667175293,
49
+ "eval_runtime": 8.0711,
50
+ "eval_samples_per_second": 25.647,
51
+ "eval_steps_per_second": 3.221,
52
+ "eval_wer": 1.0,
53
+ "step": 1500
54
+ },
55
+ {
56
+ "epoch": 3.25,
57
+ "learning_rate": 9.4279176201373e-05,
58
+ "loss": 3.0698,
59
+ "step": 2000
60
+ },
61
+ {
62
+ "epoch": 3.25,
63
+ "eval_loss": 2.9972307682037354,
64
+ "eval_runtime": 8.0326,
65
+ "eval_samples_per_second": 25.77,
66
+ "eval_steps_per_second": 3.237,
67
+ "eval_wer": 1.3356009070294785,
68
+ "step": 2000
69
+ },
70
+ {
71
+ "epoch": 4.06,
72
+ "learning_rate": 9.14187643020595e-05,
73
+ "loss": 3.1918,
74
+ "step": 2500
75
+ },
76
+ {
77
+ "epoch": 4.06,
78
+ "eval_loss": 2.9555039405822754,
79
+ "eval_runtime": 8.0447,
80
+ "eval_samples_per_second": 25.731,
81
+ "eval_steps_per_second": 3.232,
82
+ "eval_wer": 1.0,
83
+ "step": 2500
84
+ },
85
+ {
86
+ "epoch": 4.87,
87
+ "learning_rate": 8.8558352402746e-05,
88
+ "loss": 3.0932,
89
+ "step": 3000
90
+ },
91
+ {
92
+ "epoch": 4.87,
93
+ "eval_loss": 2.9776482582092285,
94
+ "eval_runtime": 8.053,
95
+ "eval_samples_per_second": 25.705,
96
+ "eval_steps_per_second": 3.229,
97
+ "eval_wer": 1.0,
98
+ "step": 3000
99
+ },
100
+ {
101
+ "epoch": 5.68,
102
+ "learning_rate": 8.569794050343249e-05,
103
+ "loss": 3.2271,
104
+ "step": 3500
105
+ },
106
+ {
107
+ "epoch": 5.68,
108
+ "eval_loss": 3.055955410003662,
109
+ "eval_runtime": 8.1309,
110
+ "eval_samples_per_second": 25.458,
111
+ "eval_steps_per_second": 3.198,
112
+ "eval_wer": 1.3378684807256236,
113
+ "step": 3500
114
+ },
115
+ {
116
+ "epoch": 6.49,
117
+ "learning_rate": 8.283752860411899e-05,
118
+ "loss": 3.2925,
119
+ "step": 4000
120
+ },
121
+ {
122
+ "epoch": 6.49,
123
+ "eval_loss": 2.9724416732788086,
124
+ "eval_runtime": 8.0677,
125
+ "eval_samples_per_second": 25.658,
126
+ "eval_steps_per_second": 3.223,
127
+ "eval_wer": 1.3900226757369614,
128
+ "step": 4000
129
+ },
130
+ {
131
+ "epoch": 7.31,
132
+ "learning_rate": 7.99771167048055e-05,
133
+ "loss": 3.2195,
134
+ "step": 4500
135
+ },
136
+ {
137
+ "epoch": 7.31,
138
+ "eval_loss": 3.5231425762176514,
139
+ "eval_runtime": 8.1433,
140
+ "eval_samples_per_second": 25.42,
141
+ "eval_steps_per_second": 3.193,
142
+ "eval_wer": 1.0,
143
+ "step": 4500
144
+ },
145
+ {
146
+ "epoch": 8.12,
147
+ "learning_rate": 7.711670480549199e-05,
148
+ "loss": 3.3582,
149
+ "step": 5000
150
+ },
151
+ {
152
+ "epoch": 8.12,
153
+ "eval_loss": 3.749286413192749,
154
+ "eval_runtime": 8.0117,
155
+ "eval_samples_per_second": 25.837,
156
+ "eval_steps_per_second": 3.245,
157
+ "eval_wer": 1.0249433106575965,
158
+ "step": 5000
159
+ },
160
+ {
161
+ "epoch": 8.93,
162
+ "learning_rate": 7.42562929061785e-05,
163
+ "loss": 3.3233,
164
+ "step": 5500
165
+ },
166
+ {
167
+ "epoch": 8.93,
168
+ "eval_loss": 3.3524105548858643,
169
+ "eval_runtime": 8.2055,
170
+ "eval_samples_per_second": 25.227,
171
+ "eval_steps_per_second": 3.169,
172
+ "eval_wer": 1.0,
173
+ "step": 5500
174
+ },
175
+ {
176
+ "epoch": 9.74,
177
+ "learning_rate": 7.1395881006865e-05,
178
+ "loss": 3.1679,
179
+ "step": 6000
180
+ },
181
+ {
182
+ "epoch": 9.74,
183
+ "eval_loss": 3.186889410018921,
184
+ "eval_runtime": 8.0182,
185
+ "eval_samples_per_second": 25.816,
186
+ "eval_steps_per_second": 3.243,
187
+ "eval_wer": 1.3877551020408163,
188
+ "step": 6000
189
+ },
190
+ {
191
+ "epoch": 10.55,
192
+ "learning_rate": 6.853546910755149e-05,
193
+ "loss": 3.1393,
194
+ "step": 6500
195
+ },
196
+ {
197
+ "epoch": 10.55,
198
+ "eval_loss": 3.0084877014160156,
199
+ "eval_runtime": 8.0014,
200
+ "eval_samples_per_second": 25.871,
201
+ "eval_steps_per_second": 3.249,
202
+ "eval_wer": 1.2947845804988662,
203
+ "step": 6500
204
+ },
205
+ {
206
+ "epoch": 11.36,
207
+ "learning_rate": 6.5675057208238e-05,
208
+ "loss": 3.1699,
209
+ "step": 7000
210
+ },
211
+ {
212
+ "epoch": 11.36,
213
+ "eval_loss": 2.997206449508667,
214
+ "eval_runtime": 8.1145,
215
+ "eval_samples_per_second": 25.51,
216
+ "eval_steps_per_second": 3.204,
217
+ "eval_wer": 1.1337868480725624,
218
+ "step": 7000
219
+ },
220
+ {
221
+ "epoch": 12.18,
222
+ "learning_rate": 6.281464530892449e-05,
223
+ "loss": 3.3382,
224
+ "step": 7500
225
+ },
226
+ {
227
+ "epoch": 12.18,
228
+ "eval_loss": 2.9686625003814697,
229
+ "eval_runtime": 8.0023,
230
+ "eval_samples_per_second": 25.868,
231
+ "eval_steps_per_second": 3.249,
232
+ "eval_wer": 1.3877551020408163,
233
+ "step": 7500
234
+ },
235
+ {
236
+ "epoch": 12.99,
237
+ "learning_rate": 5.9954233409610984e-05,
238
+ "loss": 3.0454,
239
+ "step": 8000
240
+ },
241
+ {
242
+ "epoch": 12.99,
243
+ "eval_loss": 2.968928337097168,
244
+ "eval_runtime": 8.2051,
245
+ "eval_samples_per_second": 25.228,
246
+ "eval_steps_per_second": 3.169,
247
+ "eval_wer": 1.383219954648526,
248
+ "step": 8000
249
+ },
250
+ {
251
+ "epoch": 13.8,
252
+ "learning_rate": 5.709382151029748e-05,
253
+ "loss": 3.0609,
254
+ "step": 8500
255
+ },
256
+ {
257
+ "epoch": 13.8,
258
+ "eval_loss": 2.902815580368042,
259
+ "eval_runtime": 8.0338,
260
+ "eval_samples_per_second": 25.766,
261
+ "eval_steps_per_second": 3.236,
262
+ "eval_wer": 1.3900226757369614,
263
+ "step": 8500
264
+ },
265
+ {
266
+ "epoch": 14.61,
267
+ "learning_rate": 5.423340961098399e-05,
268
+ "loss": 3.0224,
269
+ "step": 9000
270
+ },
271
+ {
272
+ "epoch": 14.61,
273
+ "eval_loss": 2.9064831733703613,
274
+ "eval_runtime": 8.2142,
275
+ "eval_samples_per_second": 25.2,
276
+ "eval_steps_per_second": 3.165,
277
+ "eval_wer": 1.3877551020408163,
278
+ "step": 9000
279
+ },
280
+ {
281
+ "epoch": 15.42,
282
+ "learning_rate": 5.137299771167048e-05,
283
+ "loss": 3.0156,
284
+ "step": 9500
285
+ },
286
+ {
287
+ "epoch": 15.42,
288
+ "eval_loss": 2.8854894638061523,
289
+ "eval_runtime": 7.9943,
290
+ "eval_samples_per_second": 25.893,
291
+ "eval_steps_per_second": 3.252,
292
+ "eval_wer": 1.3900226757369614,
293
+ "step": 9500
294
+ },
295
+ {
296
+ "epoch": 16.23,
297
+ "learning_rate": 4.851258581235698e-05,
298
+ "loss": 3.0317,
299
+ "step": 10000
300
+ },
301
+ {
302
+ "epoch": 16.23,
303
+ "eval_loss": 2.8956820964813232,
304
+ "eval_runtime": 8.1264,
305
+ "eval_samples_per_second": 25.473,
306
+ "eval_steps_per_second": 3.199,
307
+ "eval_wer": 1.3900226757369614,
308
+ "step": 10000
309
+ },
310
+ {
311
+ "epoch": 17.05,
312
+ "learning_rate": 4.565217391304348e-05,
313
+ "loss": 3.0184,
314
+ "step": 10500
315
+ },
316
+ {
317
+ "epoch": 17.05,
318
+ "eval_loss": 2.88946270942688,
319
+ "eval_runtime": 8.0387,
320
+ "eval_samples_per_second": 25.751,
321
+ "eval_steps_per_second": 3.234,
322
+ "eval_wer": 1.3900226757369614,
323
+ "step": 10500
324
+ },
325
+ {
326
+ "epoch": 17.86,
327
+ "learning_rate": 4.279176201372998e-05,
328
+ "loss": 3.0852,
329
+ "step": 11000
330
+ },
331
+ {
332
+ "epoch": 17.86,
333
+ "eval_loss": 2.8936383724212646,
334
+ "eval_runtime": 8.1682,
335
+ "eval_samples_per_second": 25.342,
336
+ "eval_steps_per_second": 3.183,
337
+ "eval_wer": 1.3900226757369614,
338
+ "step": 11000
339
+ },
340
+ {
341
+ "epoch": 18.67,
342
+ "learning_rate": 3.993135011441648e-05,
343
+ "loss": 3.0017,
344
+ "step": 11500
345
+ },
346
+ {
347
+ "epoch": 18.67,
348
+ "eval_loss": 2.8695127964019775,
349
+ "eval_runtime": 8.169,
350
+ "eval_samples_per_second": 25.34,
351
+ "eval_steps_per_second": 3.183,
352
+ "eval_wer": 1.3900226757369614,
353
+ "step": 11500
354
+ },
355
+ {
356
+ "epoch": 19.48,
357
+ "learning_rate": 3.707093821510298e-05,
358
+ "loss": 2.9337,
359
+ "step": 12000
360
+ },
361
+ {
362
+ "epoch": 19.48,
363
+ "eval_loss": 2.8768134117126465,
364
+ "eval_runtime": 8.1024,
365
+ "eval_samples_per_second": 25.548,
366
+ "eval_steps_per_second": 3.209,
367
+ "eval_wer": 1.3900226757369614,
368
+ "step": 12000
369
+ },
370
+ {
371
+ "epoch": 20.29,
372
+ "learning_rate": 3.421052631578947e-05,
373
+ "loss": 3.0017,
374
+ "step": 12500
375
+ },
376
+ {
377
+ "epoch": 20.29,
378
+ "eval_loss": 2.8580081462860107,
379
+ "eval_runtime": 8.1811,
380
+ "eval_samples_per_second": 25.302,
381
+ "eval_steps_per_second": 3.178,
382
+ "eval_wer": 1.3900226757369614,
383
+ "step": 12500
384
+ },
385
+ {
386
+ "epoch": 21.1,
387
+ "learning_rate": 3.135011441647597e-05,
388
+ "loss": 2.9472,
389
+ "step": 13000
390
+ },
391
+ {
392
+ "epoch": 21.1,
393
+ "eval_loss": 2.839965343475342,
394
+ "eval_runtime": 8.2839,
395
+ "eval_samples_per_second": 24.988,
396
+ "eval_steps_per_second": 3.139,
397
+ "eval_wer": 1.3900226757369614,
398
+ "step": 13000
399
+ },
400
+ {
401
+ "epoch": 21.92,
402
+ "learning_rate": 2.8489702517162476e-05,
403
+ "loss": 3.0214,
404
+ "step": 13500
405
+ },
406
+ {
407
+ "epoch": 21.92,
408
+ "eval_loss": 2.856123924255371,
409
+ "eval_runtime": 8.1176,
410
+ "eval_samples_per_second": 25.5,
411
+ "eval_steps_per_second": 3.203,
412
+ "eval_wer": 1.3900226757369614,
413
+ "step": 13500
414
+ },
415
+ {
416
+ "epoch": 22.73,
417
+ "learning_rate": 2.562929061784897e-05,
418
+ "loss": 2.9336,
419
+ "step": 14000
420
+ },
421
+ {
422
+ "epoch": 22.73,
423
+ "eval_loss": 2.879206895828247,
424
+ "eval_runtime": 8.1391,
425
+ "eval_samples_per_second": 25.433,
426
+ "eval_steps_per_second": 3.194,
427
+ "eval_wer": 1.3900226757369614,
428
+ "step": 14000
429
+ },
430
+ {
431
+ "epoch": 23.54,
432
+ "learning_rate": 2.276887871853547e-05,
433
+ "loss": 3.0134,
434
+ "step": 14500
435
+ },
436
+ {
437
+ "epoch": 23.54,
438
+ "eval_loss": 2.8472487926483154,
439
+ "eval_runtime": 8.0887,
440
+ "eval_samples_per_second": 25.591,
441
+ "eval_steps_per_second": 3.214,
442
+ "eval_wer": 1.3900226757369614,
443
+ "step": 14500
444
+ },
445
+ {
446
+ "epoch": 24.35,
447
+ "learning_rate": 1.990846681922197e-05,
448
+ "loss": 2.9433,
449
+ "step": 15000
450
+ },
451
+ {
452
+ "epoch": 24.35,
453
+ "eval_loss": 2.8818936347961426,
454
+ "eval_runtime": 8.193,
455
+ "eval_samples_per_second": 25.266,
456
+ "eval_steps_per_second": 3.173,
457
+ "eval_wer": 1.3900226757369614,
458
+ "step": 15000
459
+ },
460
+ {
461
+ "epoch": 25.16,
462
+ "learning_rate": 1.7048054919908468e-05,
463
+ "loss": 2.8536,
464
+ "step": 15500
465
+ },
466
+ {
467
+ "epoch": 25.16,
468
+ "eval_loss": 2.837463140487671,
469
+ "eval_runtime": 8.1663,
470
+ "eval_samples_per_second": 25.348,
471
+ "eval_steps_per_second": 3.184,
472
+ "eval_wer": 1.3900226757369614,
473
+ "step": 15500
474
+ },
475
+ {
476
+ "epoch": 25.97,
477
+ "learning_rate": 1.4187643020594965e-05,
478
+ "loss": 2.8742,
479
+ "step": 16000
480
+ },
481
+ {
482
+ "epoch": 25.97,
483
+ "eval_loss": 2.857389450073242,
484
+ "eval_runtime": 8.194,
485
+ "eval_samples_per_second": 25.262,
486
+ "eval_steps_per_second": 3.173,
487
+ "eval_wer": 1.3900226757369614,
488
+ "step": 16000
489
+ },
490
+ {
491
+ "epoch": 26.79,
492
+ "learning_rate": 1.1327231121281464e-05,
493
+ "loss": 2.8298,
494
+ "step": 16500
495
+ },
496
+ {
497
+ "epoch": 26.79,
498
+ "eval_loss": 2.982081651687622,
499
+ "eval_runtime": 8.1935,
500
+ "eval_samples_per_second": 25.264,
501
+ "eval_steps_per_second": 3.173,
502
+ "eval_wer": 1.3900226757369614,
503
+ "step": 16500
504
+ },
505
+ {
506
+ "epoch": 27.6,
507
+ "learning_rate": 8.466819221967964e-06,
508
+ "loss": 2.7439,
509
+ "step": 17000
510
+ },
511
+ {
512
+ "epoch": 27.6,
513
+ "eval_loss": 3.2741587162017822,
514
+ "eval_runtime": 8.2267,
515
+ "eval_samples_per_second": 25.162,
516
+ "eval_steps_per_second": 3.16,
517
+ "eval_wer": 1.3900226757369614,
518
+ "step": 17000
519
+ },
520
+ {
521
+ "epoch": 28.41,
522
+ "learning_rate": 5.606407322654463e-06,
523
+ "loss": 2.7008,
524
+ "step": 17500
525
+ },
526
+ {
527
+ "epoch": 28.41,
528
+ "eval_loss": 3.365966320037842,
529
+ "eval_runtime": 8.203,
530
+ "eval_samples_per_second": 25.235,
531
+ "eval_steps_per_second": 3.17,
532
+ "eval_wer": 1.3900226757369614,
533
+ "step": 17500
534
+ },
535
+ {
536
+ "epoch": 29.22,
537
+ "learning_rate": 2.745995423340961e-06,
538
+ "loss": 2.7087,
539
+ "step": 18000
540
+ },
541
+ {
542
+ "epoch": 29.22,
543
+ "eval_loss": 3.812533140182495,
544
+ "eval_runtime": 8.2382,
545
+ "eval_samples_per_second": 25.127,
546
+ "eval_steps_per_second": 3.156,
547
+ "eval_wer": 1.3900226757369614,
548
+ "step": 18000
549
+ }
550
+ ],
551
+ "max_steps": 18480,
552
+ "num_train_epochs": 30,
553
+ "total_flos": 4.093361821503758e+18,
554
+ "trial_name": null,
555
+ "trial_params": null
556
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:151803200d49964e6d9f1364dede5203d89e69881055e526220d7cecd61ffdfb
3
+ size 3375