yoad commited on
Commit
c044e28
·
verified ·
1 Parent(s): 7efbbf9

Upload folder using huggingface_hub

Browse files
config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_attn_implementation_autoset": true,
3
+ "_name_or_path": "/mlspeech/data/yoadsnapir/models/heb_tiny_exp_3",
4
+ "activation_dropout": 0.0,
5
+ "activation_function": "gelu",
6
+ "apply_spec_augment": false,
7
+ "architectures": [
8
+ "WhisperForConditionalGeneration"
9
+ ],
10
+ "attention_dropout": 0.0,
11
+ "begin_suppress_tokens": null,
12
+ "bos_token_id": 50257,
13
+ "classifier_proj_size": 256,
14
+ "d_model": 384,
15
+ "decoder_attention_heads": 6,
16
+ "decoder_ffn_dim": 1536,
17
+ "decoder_layerdrop": 0.0,
18
+ "decoder_layers": 4,
19
+ "decoder_start_token_id": 50258,
20
+ "dropout": 0.0,
21
+ "encoder_attention_heads": 6,
22
+ "encoder_ffn_dim": 1536,
23
+ "encoder_layerdrop": 0.0,
24
+ "encoder_layers": 4,
25
+ "eos_token_id": 50257,
26
+ "forced_decoder_ids": null,
27
+ "init_std": 0.02,
28
+ "is_encoder_decoder": true,
29
+ "mask_feature_length": 10,
30
+ "mask_feature_min_masks": 0,
31
+ "mask_feature_prob": 0.0,
32
+ "mask_time_length": 10,
33
+ "mask_time_min_masks": 2,
34
+ "mask_time_prob": 0.05,
35
+ "max_length": null,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_hidden_layers": 4,
41
+ "num_mel_bins": 80,
42
+ "pad_token_id": 50257,
43
+ "scale_embedding": false,
44
+ "torch_dtype": "float32",
45
+ "transformers_version": "4.48.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51865
49
+ }
generation_config.json ADDED
@@ -0,0 +1,160 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 2,
5
+ 2
6
+ ],
7
+ [
8
+ 3,
9
+ 0
10
+ ],
11
+ [
12
+ 3,
13
+ 2
14
+ ],
15
+ [
16
+ 3,
17
+ 3
18
+ ],
19
+ [
20
+ 3,
21
+ 4
22
+ ],
23
+ [
24
+ 3,
25
+ 5
26
+ ]
27
+ ],
28
+ "attn_implementation": "sdpa",
29
+ "begin_suppress_tokens": [
30
+ 220,
31
+ 50257
32
+ ],
33
+ "bos_token_id": 50257,
34
+ "decoder_start_token_id": 50258,
35
+ "eos_token_id": 50257,
36
+ "forced_decoder_ids": [
37
+ [
38
+ 1,
39
+ null
40
+ ],
41
+ [
42
+ 2,
43
+ 50359
44
+ ]
45
+ ],
46
+ "is_multilingual": true,
47
+ "lang_to_id": {
48
+ "<|af|>": 50327,
49
+ "<|am|>": 50334,
50
+ "<|ar|>": 50272,
51
+ "<|as|>": 50350,
52
+ "<|az|>": 50304,
53
+ "<|ba|>": 50355,
54
+ "<|be|>": 50330,
55
+ "<|bg|>": 50292,
56
+ "<|bn|>": 50302,
57
+ "<|bo|>": 50347,
58
+ "<|br|>": 50309,
59
+ "<|bs|>": 50315,
60
+ "<|ca|>": 50270,
61
+ "<|cs|>": 50283,
62
+ "<|cy|>": 50297,
63
+ "<|da|>": 50285,
64
+ "<|de|>": 50261,
65
+ "<|el|>": 50281,
66
+ "<|en|>": 50259,
67
+ "<|es|>": 50262,
68
+ "<|et|>": 50307,
69
+ "<|eu|>": 50310,
70
+ "<|fa|>": 50300,
71
+ "<|fi|>": 50277,
72
+ "<|fo|>": 50338,
73
+ "<|fr|>": 50265,
74
+ "<|gl|>": 50319,
75
+ "<|gu|>": 50333,
76
+ "<|haw|>": 50352,
77
+ "<|ha|>": 50354,
78
+ "<|he|>": 50279,
79
+ "<|hi|>": 50276,
80
+ "<|hr|>": 50291,
81
+ "<|ht|>": 50339,
82
+ "<|hu|>": 50286,
83
+ "<|hy|>": 50312,
84
+ "<|id|>": 50275,
85
+ "<|is|>": 50311,
86
+ "<|it|>": 50274,
87
+ "<|ja|>": 50266,
88
+ "<|jw|>": 50356,
89
+ "<|ka|>": 50329,
90
+ "<|kk|>": 50316,
91
+ "<|km|>": 50323,
92
+ "<|kn|>": 50306,
93
+ "<|ko|>": 50264,
94
+ "<|la|>": 50294,
95
+ "<|lb|>": 50345,
96
+ "<|ln|>": 50353,
97
+ "<|lo|>": 50336,
98
+ "<|lt|>": 50293,
99
+ "<|lv|>": 50301,
100
+ "<|mg|>": 50349,
101
+ "<|mi|>": 50295,
102
+ "<|mk|>": 50308,
103
+ "<|ml|>": 50296,
104
+ "<|mn|>": 50314,
105
+ "<|mr|>": 50320,
106
+ "<|ms|>": 50282,
107
+ "<|mt|>": 50343,
108
+ "<|my|>": 50346,
109
+ "<|ne|>": 50313,
110
+ "<|nl|>": 50271,
111
+ "<|nn|>": 50342,
112
+ "<|no|>": 50288,
113
+ "<|oc|>": 50328,
114
+ "<|pa|>": 50321,
115
+ "<|pl|>": 50269,
116
+ "<|ps|>": 50340,
117
+ "<|pt|>": 50267,
118
+ "<|ro|>": 50284,
119
+ "<|ru|>": 50263,
120
+ "<|sa|>": 50344,
121
+ "<|sd|>": 50332,
122
+ "<|si|>": 50322,
123
+ "<|sk|>": 50298,
124
+ "<|sl|>": 50305,
125
+ "<|sn|>": 50324,
126
+ "<|so|>": 50326,
127
+ "<|sq|>": 50317,
128
+ "<|sr|>": 50303,
129
+ "<|su|>": 50357,
130
+ "<|sv|>": 50273,
131
+ "<|sw|>": 50318,
132
+ "<|ta|>": 50287,
133
+ "<|te|>": 50299,
134
+ "<|tg|>": 50331,
135
+ "<|th|>": 50289,
136
+ "<|tk|>": 50341,
137
+ "<|tl|>": 50348,
138
+ "<|tr|>": 50268,
139
+ "<|tt|>": 50351,
140
+ "<|uk|>": 50280,
141
+ "<|ur|>": 50290,
142
+ "<|uz|>": 50337,
143
+ "<|vi|>": 50278,
144
+ "<|yi|>": 50335,
145
+ "<|yo|>": 50325,
146
+ "<|zh|>": 50260
147
+ },
148
+ "max_initial_timestamp_index": 50,
149
+ "max_length": 448,
150
+ "no_timestamps_token_id": 50363,
151
+ "pad_token_id": 50257,
152
+ "prev_sot_token_id": 50361,
153
+ "return_timestamps": false,
154
+ "suppress_tokens": [],
155
+ "task_to_id": {
156
+ "transcribe": 50359,
157
+ "translate": 50358
158
+ },
159
+ "transformers_version": "4.48.1"
160
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f79730aa47265b7fc71a0549c0678b41e824331d7501c5545cac7ad96ca0d921
3
+ size 151061672
preprocessor_config.json ADDED
The diff for this file is too large to render. See raw diff
 
special_tokens_map.json ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|endoftext|>",
4
+ "<|startoftranscript|>",
5
+ "<|en|>",
6
+ "<|zh|>",
7
+ "<|de|>",
8
+ "<|es|>",
9
+ "<|ru|>",
10
+ "<|ko|>",
11
+ "<|fr|>",
12
+ "<|ja|>",
13
+ "<|pt|>",
14
+ "<|tr|>",
15
+ "<|pl|>",
16
+ "<|ca|>",
17
+ "<|nl|>",
18
+ "<|ar|>",
19
+ "<|sv|>",
20
+ "<|it|>",
21
+ "<|id|>",
22
+ "<|hi|>",
23
+ "<|fi|>",
24
+ "<|vi|>",
25
+ "<|he|>",
26
+ "<|uk|>",
27
+ "<|el|>",
28
+ "<|ms|>",
29
+ "<|cs|>",
30
+ "<|ro|>",
31
+ "<|da|>",
32
+ "<|hu|>",
33
+ "<|ta|>",
34
+ "<|no|>",
35
+ "<|th|>",
36
+ "<|ur|>",
37
+ "<|hr|>",
38
+ "<|bg|>",
39
+ "<|lt|>",
40
+ "<|la|>",
41
+ "<|mi|>",
42
+ "<|ml|>",
43
+ "<|cy|>",
44
+ "<|sk|>",
45
+ "<|te|>",
46
+ "<|fa|>",
47
+ "<|lv|>",
48
+ "<|bn|>",
49
+ "<|sr|>",
50
+ "<|az|>",
51
+ "<|sl|>",
52
+ "<|kn|>",
53
+ "<|et|>",
54
+ "<|mk|>",
55
+ "<|br|>",
56
+ "<|eu|>",
57
+ "<|is|>",
58
+ "<|hy|>",
59
+ "<|ne|>",
60
+ "<|mn|>",
61
+ "<|bs|>",
62
+ "<|kk|>",
63
+ "<|sq|>",
64
+ "<|sw|>",
65
+ "<|gl|>",
66
+ "<|mr|>",
67
+ "<|pa|>",
68
+ "<|si|>",
69
+ "<|km|>",
70
+ "<|sn|>",
71
+ "<|yo|>",
72
+ "<|so|>",
73
+ "<|af|>",
74
+ "<|oc|>",
75
+ "<|ka|>",
76
+ "<|be|>",
77
+ "<|tg|>",
78
+ "<|sd|>",
79
+ "<|gu|>",
80
+ "<|am|>",
81
+ "<|yi|>",
82
+ "<|lo|>",
83
+ "<|uz|>",
84
+ "<|fo|>",
85
+ "<|ht|>",
86
+ "<|ps|>",
87
+ "<|tk|>",
88
+ "<|nn|>",
89
+ "<|mt|>",
90
+ "<|sa|>",
91
+ "<|lb|>",
92
+ "<|my|>",
93
+ "<|bo|>",
94
+ "<|tl|>",
95
+ "<|mg|>",
96
+ "<|as|>",
97
+ "<|tt|>",
98
+ "<|haw|>",
99
+ "<|ln|>",
100
+ "<|ha|>",
101
+ "<|ba|>",
102
+ "<|jw|>",
103
+ "<|su|>",
104
+ "<|translate|>",
105
+ "<|transcribe|>",
106
+ "<|startoflm|>",
107
+ "<|startofprev|>",
108
+ "<|nocaptions|>",
109
+ "<|notimestamps|>"
110
+ ],
111
+ "bos_token": {
112
+ "content": "<|endoftext|>",
113
+ "lstrip": false,
114
+ "normalized": false,
115
+ "rstrip": false,
116
+ "single_word": false
117
+ },
118
+ "eos_token": {
119
+ "content": "<|endoftext|>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false
124
+ },
125
+ "pad_token": {
126
+ "content": "<|endoftext|>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false
131
+ },
132
+ "unk_token": {
133
+ "content": "<|endoftext|>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false
138
+ }
139
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
The diff for this file is too large to render. See raw diff
 
vocab.json ADDED
The diff for this file is too large to render. See raw diff