Baselhany commited on
Commit
58de744
·
verified ·
1 Parent(s): 6cdb178

Model save

Browse files
README.md ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: transformers
3
+ license: apache-2.0
4
+ base_model: YoussefAshmawy/Graduation_Project_Whisper_base
5
+ tags:
6
+ - generated_from_trainer
7
+ metrics:
8
+ - wer
9
+ model-index:
10
+ - name: Graduation_Project_distillation_Whisper_base2
11
+ results: []
12
+ ---
13
+
14
+ <!-- This model card has been generated automatically according to the information the Trainer had access to. You
15
+ should probably proofread and complete it, then remove this comment. -->
16
+
17
+ # Graduation_Project_distillation_Whisper_base2
18
+
19
+ This model is a fine-tuned version of [YoussefAshmawy/Graduation_Project_Whisper_base](https://huggingface.co/YoussefAshmawy/Graduation_Project_Whisper_base) on an unknown dataset.
20
+ It achieves the following results on the evaluation set:
21
+ - Loss: 0.1269
22
+ - Wer: 0.4067
23
+
24
+ ## Model description
25
+
26
+ More information needed
27
+
28
+ ## Intended uses & limitations
29
+
30
+ More information needed
31
+
32
+ ## Training and evaluation data
33
+
34
+ More information needed
35
+
36
+ ## Training procedure
37
+
38
+ ### Training hyperparameters
39
+
40
+ The following hyperparameters were used during training:
41
+ - learning_rate: 0.0001
42
+ - train_batch_size: 8
43
+ - eval_batch_size: 8
44
+ - seed: 42
45
+ - gradient_accumulation_steps: 4
46
+ - total_train_batch_size: 32
47
+ - optimizer: Use OptimizerNames.ADAMW_TORCH with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
48
+ - lr_scheduler_type: linear
49
+ - lr_scheduler_warmup_steps: 500
50
+ - num_epochs: 15
51
+ - mixed_precision_training: Native AMP
52
+
53
+ ### Training results
54
+
55
+ | Training Loss | Epoch | Step | Validation Loss | Wer |
56
+ |:-------------:|:-------:|:----:|:---------------:|:------:|
57
+ | 12.6571 | 1.0 | 469 | 0.0985 | 0.5863 |
58
+ | 3.596 | 2.0 | 938 | 0.0901 | 0.4695 |
59
+ | 1.7624 | 3.0 | 1407 | 0.1011 | 0.4507 |
60
+ | 1.1052 | 4.0 | 1876 | 0.1067 | 0.4320 |
61
+ | 0.7406 | 5.0 | 2345 | 0.1194 | 0.4031 |
62
+ | 0.4646 | 6.0 | 2814 | 0.1182 | 0.4420 |
63
+ | 0.3698 | 7.0 | 3283 | 0.1201 | 0.4309 |
64
+ | 0.2819 | 8.0 | 3752 | 0.1177 | 0.3808 |
65
+ | 0.2221 | 9.0 | 4221 | 0.1144 | 0.3605 |
66
+ | 0.1999 | 10.0 | 4690 | 0.1112 | 0.3871 |
67
+ | 0.1506 | 11.0 | 5159 | 0.1104 | 0.3878 |
68
+ | 0.1271 | 12.0 | 5628 | 0.1095 | 0.3966 |
69
+ | 0.1034 | 13.0 | 6097 | 0.1102 | 0.3931 |
70
+ | 0.0816 | 14.0 | 6566 | 0.1113 | 0.4061 |
71
+ | 0.0647 | 14.9685 | 7020 | 0.1114 | 0.3778 |
72
+
73
+
74
+ ### Framework versions
75
+
76
+ - Transformers 4.51.3
77
+ - Pytorch 2.6.0+cu124
78
+ - Datasets 3.6.0
79
+ - Tokenizers 0.21.1
generation_config.json ADDED
@@ -0,0 +1,228 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_attn_implementation_autoset": true,
3
+ "_name_or_path": "YoussefAshmawy/Graduation_Project_Whisper_base",
4
+ "activation_dropout": 0.0,
5
+ "activation_function": "gelu",
6
+ "add_cross_attention": false,
7
+ "alignment_heads": [
8
+ [
9
+ 3,
10
+ 1
11
+ ],
12
+ [
13
+ 4,
14
+ 2
15
+ ],
16
+ [
17
+ 4,
18
+ 3
19
+ ],
20
+ [
21
+ 4,
22
+ 7
23
+ ],
24
+ [
25
+ 5,
26
+ 1
27
+ ],
28
+ [
29
+ 5,
30
+ 2
31
+ ],
32
+ [
33
+ 5,
34
+ 4
35
+ ],
36
+ [
37
+ 5,
38
+ 6
39
+ ]
40
+ ],
41
+ "apply_spec_augment": false,
42
+ "architectures": [
43
+ "WhisperForConditionalGeneration"
44
+ ],
45
+ "attention_dropout": 0.0,
46
+ "bos_token_id": 50257,
47
+ "chunk_size_feed_forward": 0,
48
+ "classifier_proj_size": 256,
49
+ "cross_attention_hidden_size": null,
50
+ "d_model": 512,
51
+ "decoder_attention_heads": 8,
52
+ "decoder_ffn_dim": 2048,
53
+ "decoder_layerdrop": 0.0,
54
+ "decoder_layers": 2,
55
+ "decoder_start_token_id": 50258,
56
+ "dropout": 0.0,
57
+ "encoder_attention_heads": 8,
58
+ "encoder_ffn_dim": 2048,
59
+ "encoder_layerdrop": 0.0,
60
+ "encoder_layers": 6,
61
+ "eos_token_id": 50257,
62
+ "finetuning_task": null,
63
+ "id2label": {
64
+ "0": "LABEL_0",
65
+ "1": "LABEL_1"
66
+ },
67
+ "init_std": 0.02,
68
+ "input_ids": [
69
+ [
70
+ 1,
71
+ 50272
72
+ ],
73
+ [
74
+ 2,
75
+ 50359
76
+ ],
77
+ [
78
+ 3,
79
+ 50363
80
+ ]
81
+ ],
82
+ "is_decoder": false,
83
+ "is_encoder_decoder": true,
84
+ "is_multilingual": true,
85
+ "label2id": {
86
+ "LABEL_0": 0,
87
+ "LABEL_1": 1
88
+ },
89
+ "lang_to_id": {
90
+ "<|af|>": 50327,
91
+ "<|am|>": 50334,
92
+ "<|ar|>": 50272,
93
+ "<|as|>": 50350,
94
+ "<|az|>": 50304,
95
+ "<|ba|>": 50355,
96
+ "<|be|>": 50330,
97
+ "<|bg|>": 50292,
98
+ "<|bn|>": 50302,
99
+ "<|bo|>": 50347,
100
+ "<|br|>": 50309,
101
+ "<|bs|>": 50315,
102
+ "<|ca|>": 50270,
103
+ "<|cs|>": 50283,
104
+ "<|cy|>": 50297,
105
+ "<|da|>": 50285,
106
+ "<|de|>": 50261,
107
+ "<|el|>": 50281,
108
+ "<|en|>": 50259,
109
+ "<|es|>": 50262,
110
+ "<|et|>": 50307,
111
+ "<|eu|>": 50310,
112
+ "<|fa|>": 50300,
113
+ "<|fi|>": 50277,
114
+ "<|fo|>": 50338,
115
+ "<|fr|>": 50265,
116
+ "<|gl|>": 50319,
117
+ "<|gu|>": 50333,
118
+ "<|haw|>": 50352,
119
+ "<|ha|>": 50354,
120
+ "<|he|>": 50279,
121
+ "<|hi|>": 50276,
122
+ "<|hr|>": 50291,
123
+ "<|ht|>": 50339,
124
+ "<|hu|>": 50286,
125
+ "<|hy|>": 50312,
126
+ "<|id|>": 50275,
127
+ "<|is|>": 50311,
128
+ "<|it|>": 50274,
129
+ "<|ja|>": 50266,
130
+ "<|jw|>": 50356,
131
+ "<|ka|>": 50329,
132
+ "<|kk|>": 50316,
133
+ "<|km|>": 50323,
134
+ "<|kn|>": 50306,
135
+ "<|ko|>": 50264,
136
+ "<|la|>": 50294,
137
+ "<|lb|>": 50345,
138
+ "<|ln|>": 50353,
139
+ "<|lo|>": 50336,
140
+ "<|lt|>": 50293,
141
+ "<|lv|>": 50301,
142
+ "<|mg|>": 50349,
143
+ "<|mi|>": 50295,
144
+ "<|mk|>": 50308,
145
+ "<|ml|>": 50296,
146
+ "<|mn|>": 50314,
147
+ "<|mr|>": 50320,
148
+ "<|ms|>": 50282,
149
+ "<|mt|>": 50343,
150
+ "<|my|>": 50346,
151
+ "<|ne|>": 50313,
152
+ "<|nl|>": 50271,
153
+ "<|nn|>": 50342,
154
+ "<|no|>": 50288,
155
+ "<|oc|>": 50328,
156
+ "<|pa|>": 50321,
157
+ "<|pl|>": 50269,
158
+ "<|ps|>": 50340,
159
+ "<|pt|>": 50267,
160
+ "<|ro|>": 50284,
161
+ "<|ru|>": 50263,
162
+ "<|sa|>": 50344,
163
+ "<|sd|>": 50332,
164
+ "<|si|>": 50322,
165
+ "<|sk|>": 50298,
166
+ "<|sl|>": 50305,
167
+ "<|sn|>": 50324,
168
+ "<|so|>": 50326,
169
+ "<|sq|>": 50317,
170
+ "<|sr|>": 50303,
171
+ "<|su|>": 50357,
172
+ "<|sv|>": 50273,
173
+ "<|sw|>": 50318,
174
+ "<|ta|>": 50287,
175
+ "<|te|>": 50299,
176
+ "<|tg|>": 50331,
177
+ "<|th|>": 50289,
178
+ "<|tk|>": 50341,
179
+ "<|tl|>": 50348,
180
+ "<|tr|>": 50268,
181
+ "<|tt|>": 50351,
182
+ "<|uk|>": 50280,
183
+ "<|ur|>": 50290,
184
+ "<|uz|>": 50337,
185
+ "<|vi|>": 50278,
186
+ "<|yi|>": 50335,
187
+ "<|yo|>": 50325,
188
+ "<|zh|>": 50260
189
+ },
190
+ "mask_feature_length": 10,
191
+ "mask_feature_min_masks": 0,
192
+ "mask_feature_prob": 0.0,
193
+ "mask_time_length": 10,
194
+ "mask_time_min_masks": 2,
195
+ "mask_time_prob": 0.05,
196
+ "max_initial_timestamp_index": 50,
197
+ "max_length": 448,
198
+ "max_source_positions": 1500,
199
+ "max_target_positions": 448,
200
+ "median_filter_width": 7,
201
+ "model_type": "whisper",
202
+ "no_timestamps_token_id": 50363,
203
+ "num_hidden_layers": 6,
204
+ "num_mel_bins": 80,
205
+ "pad_token_id": 50257,
206
+ "prefix": null,
207
+ "prev_sot_token_id": 50361,
208
+ "problem_type": null,
209
+ "pruned_heads": {},
210
+ "return_dict": true,
211
+ "return_timestamps": false,
212
+ "scale_embedding": false,
213
+ "sep_token_id": null,
214
+ "task_specific_params": null,
215
+ "task_to_id": {
216
+ "transcribe": 50359,
217
+ "translate": 50358
218
+ },
219
+ "tf_legacy_loss": false,
220
+ "tie_encoder_decoder": false,
221
+ "tie_word_embeddings": true,
222
+ "tokenizer_class": null,
223
+ "torchscript": false,
224
+ "transformers_version": "4.51.3",
225
+ "use_bfloat16": false,
226
+ "use_weighted_layer_sum": false,
227
+ "vocab_size": 51865
228
+ }
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:de5ffdae4dc3ec5d2330a531a00b56aba3e782dc7ff3adf4e6ce19d37d1448de
3
  size 223144592
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0a68b3521e9bd50271e376efca2d88b8dac8811b5ddf1f1e34f1d40768a442e5
3
  size 223144592
runs/Jun17_11-33-36_835254aa27d9/events.out.tfevents.1750189031.835254aa27d9.19.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5b944bfdab7ff961d27e9578c22f64735342cc7fb7d83a535aaf1d6403322e2c
3
+ size 406