orgit commited on
Commit
ba99c86
·
0 Parent(s):

Initial commit

Browse files
Files changed (36) hide show
  1. .gitattributes +35 -0
  2. README.md +7 -0
  3. medium/AudioEncoder.mlmodelc/analytics/coremldata.bin +3 -0
  4. medium/AudioEncoder.mlmodelc/coremldata.bin +3 -0
  5. medium/AudioEncoder.mlmodelc/metadata.json +73 -0
  6. medium/AudioEncoder.mlmodelc/model.mil +0 -0
  7. medium/AudioEncoder.mlmodelc/weights/weight.bin +3 -0
  8. medium/MelSpectrogram.mlmodelc/analytics/coremldata.bin +3 -0
  9. medium/MelSpectrogram.mlmodelc/coremldata.bin +3 -0
  10. medium/MelSpectrogram.mlmodelc/metadata.json +75 -0
  11. medium/MelSpectrogram.mlmodelc/model.mil +66 -0
  12. medium/MelSpectrogram.mlmodelc/weights/weight.bin +3 -0
  13. medium/TextDecoder.mlmodelc/analytics/coremldata.bin +3 -0
  14. medium/TextDecoder.mlmodelc/coremldata.bin +3 -0
  15. medium/TextDecoder.mlmodelc/metadata.json +171 -0
  16. medium/TextDecoder.mlmodelc/model.mil +0 -0
  17. medium/TextDecoder.mlmodelc/weights/weight.bin +3 -0
  18. medium/config.json +49 -0
  19. medium/generation_config.json +161 -0
  20. small/AudioEncoder.mlmodelc/analytics/coremldata.bin +3 -0
  21. small/AudioEncoder.mlmodelc/coremldata.bin +3 -0
  22. small/AudioEncoder.mlmodelc/metadata.json +73 -0
  23. small/AudioEncoder.mlmodelc/model.mil +0 -0
  24. small/AudioEncoder.mlmodelc/weights/weight.bin +3 -0
  25. small/MelSpectrogram.mlmodelc/analytics/coremldata.bin +3 -0
  26. small/MelSpectrogram.mlmodelc/coremldata.bin +3 -0
  27. small/MelSpectrogram.mlmodelc/metadata.json +75 -0
  28. small/MelSpectrogram.mlmodelc/model.mil +66 -0
  29. small/MelSpectrogram.mlmodelc/weights/weight.bin +3 -0
  30. small/TextDecoder.mlmodelc/analytics/coremldata.bin +3 -0
  31. small/TextDecoder.mlmodelc/coremldata.bin +3 -0
  32. small/TextDecoder.mlmodelc/metadata.json +171 -0
  33. small/TextDecoder.mlmodelc/model.mil +0 -0
  34. small/TextDecoder.mlmodelc/weights/weight.bin +3 -0
  35. small/config.json +49 -0
  36. small/generation_config.json +161 -0
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ ---
4
+
5
+ # Tape
6
+
7
+ Coming soon.
medium/AudioEncoder.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1a8ab7aec23b627541276af4d0bb0e4000050282e2bcb684b64584fba27a9a4
3
+ size 243
medium/AudioEncoder.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:90b5e31a285d2c0af400e27a7e38838beaec50cf44da2d7534fbdec70bfcc686
3
+ size 409
medium/AudioEncoder.mlmodelc/metadata.json ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Mixed (Float16, Palettized (8 bits), Sparse)",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 1280 × 1 × 1500)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 1280, 1, 1500]",
13
+ "name" : "encoder_output_embeds",
14
+ "type" : "MultiArray"
15
+ }
16
+ ],
17
+ "modelParameters" : [
18
+
19
+ ],
20
+ "specificationVersion" : 7,
21
+ "mlProgramOperationTypeHistogram" : {
22
+ "Concat" : 672,
23
+ "Ios16.mul" : 3840,
24
+ "Ios16.layerNorm" : 65,
25
+ "SliceByIndex" : 5760,
26
+ "Ios16.constexprLutToDense" : 194,
27
+ "Transpose" : 32,
28
+ "Ios16.einsum" : 7680,
29
+ "Ios16.conv" : 388,
30
+ "Ios16.add" : 259,
31
+ "Ios16.constexprSparseToDense" : 192,
32
+ "Ios16.softmax" : 3840,
33
+ "Ios16.gelu" : 34,
34
+ "Ios16.batchNorm" : 65
35
+ },
36
+ "computePrecision" : "Mixed (Float16, Int32)",
37
+ "isUpdatable" : "0",
38
+ "stateSchema" : [
39
+
40
+ ],
41
+ "availability" : {
42
+ "macOS" : "13.0",
43
+ "tvOS" : "16.0",
44
+ "visionOS" : "1.0",
45
+ "watchOS" : "9.0",
46
+ "iOS" : "16.0",
47
+ "macCatalyst" : "16.0"
48
+ },
49
+ "modelType" : {
50
+ "name" : "MLModelType_mlProgram"
51
+ },
52
+ "userDefinedMetadata" : {
53
+ "com.github.apple.coremltools.conversion_date" : "2026-03-05",
54
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
55
+ "com.github.apple.coremltools.version" : "9.0",
56
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
57
+ },
58
+ "inputSchema" : [
59
+ {
60
+ "hasShapeFlexibility" : "0",
61
+ "isOptional" : "0",
62
+ "dataType" : "Float16",
63
+ "formattedType" : "MultiArray (Float16 1 × 128 × 1 × 3000)",
64
+ "shortDescription" : "",
65
+ "shape" : "[1, 128, 1, 3000]",
66
+ "name" : "melspectrogram_features",
67
+ "type" : "MultiArray"
68
+ }
69
+ ],
70
+ "generatedClassName" : "AudioEncoder_mixedBitPalettized_8_bit",
71
+ "method" : "predict"
72
+ }
73
+ ]
medium/AudioEncoder.mlmodelc/model.mil ADDED
The diff for this file is too large to render. See raw diff
 
medium/AudioEncoder.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e260848b44cd73c7ce878c55504d678bd12e602cbeaa39c2596857e4941ec4b
3
+ size 739291328
medium/MelSpectrogram.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2592a0e27cd8a703fe7c7d2a6a0fca872efebfbb8e2f0640320a89c5a553573d
3
+ size 243
medium/MelSpectrogram.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9f049e1a21b69f3617487af4198470abb6eea4a98eb8dbcaf57c9eaca640ff91
3
+ size 390
medium/MelSpectrogram.mlmodelc/metadata.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Float16",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 128 × 1 × 3000)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 128, 1, 3000]",
13
+ "name" : "melspectrogram_features",
14
+ "type" : "MultiArray"
15
+ }
16
+ ],
17
+ "modelParameters" : [
18
+
19
+ ],
20
+ "specificationVersion" : 7,
21
+ "mlProgramOperationTypeHistogram" : {
22
+ "Ios16.reshape" : 2,
23
+ "Ios16.mul" : 2,
24
+ "SliceByIndex" : 1,
25
+ "Ios16.sub" : 1,
26
+ "Ios16.log" : 1,
27
+ "Ios16.square" : 2,
28
+ "Ios16.add" : 3,
29
+ "Squeeze" : 2,
30
+ "Ios16.matmul" : 1,
31
+ "Ios16.conv" : 2,
32
+ "Ios16.maximum" : 1,
33
+ "ExpandDims" : 4,
34
+ "Ios16.reduceMax" : 1,
35
+ "Identity" : 1,
36
+ "Pad" : 1
37
+ },
38
+ "computePrecision" : "Mixed (Float16, Int32)",
39
+ "isUpdatable" : "0",
40
+ "stateSchema" : [
41
+
42
+ ],
43
+ "availability" : {
44
+ "macOS" : "13.0",
45
+ "tvOS" : "16.0",
46
+ "visionOS" : "1.0",
47
+ "watchOS" : "9.0",
48
+ "iOS" : "16.0",
49
+ "macCatalyst" : "16.0"
50
+ },
51
+ "modelType" : {
52
+ "name" : "MLModelType_mlProgram"
53
+ },
54
+ "userDefinedMetadata" : {
55
+ "com.github.apple.coremltools.conversion_date" : "2026-03-04",
56
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
57
+ "com.github.apple.coremltools.version" : "9.0",
58
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
59
+ },
60
+ "inputSchema" : [
61
+ {
62
+ "hasShapeFlexibility" : "0",
63
+ "isOptional" : "0",
64
+ "dataType" : "Float16",
65
+ "formattedType" : "MultiArray (Float16 480000)",
66
+ "shortDescription" : "",
67
+ "shape" : "[480000]",
68
+ "name" : "audio",
69
+ "type" : "MultiArray"
70
+ }
71
+ ],
72
+ "generatedClassName" : "MelSpectrogram",
73
+ "method" : "predict"
74
+ }
75
+ ]
medium/MelSpectrogram.mlmodelc/model.mil ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ program(1.0)
2
+ [buildInfo = dict<tensor<string, []>, tensor<string, []>>({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-torch", "2.5.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
3
+ {
4
+ func main<ios16>(tensor<fp16, [480000]> audio) {
5
+ tensor<int32, [3]> var_10 = const()[name = tensor<string, []>("op_10"), val = tensor<int32, [3]>([1, 1, 480000])];
6
+ tensor<fp16, [1, 1, 480000]> input_1_cast_fp16 = reshape(shape = var_10, x = audio)[name = tensor<string, []>("input_1_cast_fp16")];
7
+ tensor<int32, [6]> input_3_pad_0 = const()[name = tensor<string, []>("input_3_pad_0"), val = tensor<int32, [6]>([0, 0, 0, 0, 200, 200])];
8
+ tensor<string, []> input_3_mode_0 = const()[name = tensor<string, []>("input_3_mode_0"), val = tensor<string, []>("reflect")];
9
+ tensor<fp16, []> const_1_to_fp16 = const()[name = tensor<string, []>("const_1_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
10
+ tensor<fp16, [1, 1, 480400]> input_3_cast_fp16 = pad(constant_val = const_1_to_fp16, mode = input_3_mode_0, pad = input_3_pad_0, x = input_1_cast_fp16)[name = tensor<string, []>("input_3_cast_fp16")];
11
+ tensor<int32, [1]> var_22 = const()[name = tensor<string, []>("op_22"), val = tensor<int32, [1]>([480400])];
12
+ tensor<fp16, [480400]> input_cast_fp16 = reshape(shape = var_22, x = input_3_cast_fp16)[name = tensor<string, []>("input_cast_fp16")];
13
+ tensor<int32, [1]> expand_dims_0_axes_0 = const()[name = tensor<string, []>("expand_dims_0_axes_0"), val = tensor<int32, [1]>([0])];
14
+ tensor<fp16, [1, 480400]> expand_dims_0_cast_fp16 = expand_dims(axes = expand_dims_0_axes_0, x = input_cast_fp16)[name = tensor<string, []>("expand_dims_0_cast_fp16")];
15
+ tensor<int32, [1]> expand_dims_3 = const()[name = tensor<string, []>("expand_dims_3"), val = tensor<int32, [1]>([160])];
16
+ tensor<int32, [1]> expand_dims_4_axes_0 = const()[name = tensor<string, []>("expand_dims_4_axes_0"), val = tensor<int32, [1]>([1])];
17
+ tensor<fp16, [1, 1, 480400]> expand_dims_4_cast_fp16 = expand_dims(axes = expand_dims_4_axes_0, x = expand_dims_0_cast_fp16)[name = tensor<string, []>("expand_dims_4_cast_fp16")];
18
+ tensor<string, []> conv_0_pad_type_0 = const()[name = tensor<string, []>("conv_0_pad_type_0"), val = tensor<string, []>("valid")];
19
+ tensor<int32, [2]> conv_0_pad_0 = const()[name = tensor<string, []>("conv_0_pad_0"), val = tensor<int32, [2]>([0, 0])];
20
+ tensor<int32, [1]> conv_0_dilations_0 = const()[name = tensor<string, []>("conv_0_dilations_0"), val = tensor<int32, [1]>([1])];
21
+ tensor<int32, []> conv_0_groups_0 = const()[name = tensor<string, []>("conv_0_groups_0"), val = tensor<int32, []>(1)];
22
+ tensor<fp16, [201, 1, 400]> expand_dims_1_to_fp16 = const()[name = tensor<string, []>("expand_dims_1_to_fp16"), val = tensor<fp16, [201, 1, 400]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(64)))];
23
+ tensor<fp16, [1, 201, 3001]> conv_0_cast_fp16 = conv(dilations = conv_0_dilations_0, groups = conv_0_groups_0, pad = conv_0_pad_0, pad_type = conv_0_pad_type_0, strides = expand_dims_3, weight = expand_dims_1_to_fp16, x = expand_dims_4_cast_fp16)[name = tensor<string, []>("conv_0_cast_fp16")];
24
+ tensor<string, []> conv_1_pad_type_0 = const()[name = tensor<string, []>("conv_1_pad_type_0"), val = tensor<string, []>("valid")];
25
+ tensor<int32, [2]> conv_1_pad_0 = const()[name = tensor<string, []>("conv_1_pad_0"), val = tensor<int32, [2]>([0, 0])];
26
+ tensor<int32, [1]> conv_1_dilations_0 = const()[name = tensor<string, []>("conv_1_dilations_0"), val = tensor<int32, [1]>([1])];
27
+ tensor<int32, []> conv_1_groups_0 = const()[name = tensor<string, []>("conv_1_groups_0"), val = tensor<int32, []>(1)];
28
+ tensor<fp16, [201, 1, 400]> expand_dims_2_to_fp16 = const()[name = tensor<string, []>("expand_dims_2_to_fp16"), val = tensor<fp16, [201, 1, 400]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(160960)))];
29
+ tensor<fp16, [1, 201, 3001]> conv_1_cast_fp16 = conv(dilations = conv_1_dilations_0, groups = conv_1_groups_0, pad = conv_1_pad_0, pad_type = conv_1_pad_type_0, strides = expand_dims_3, weight = expand_dims_2_to_fp16, x = expand_dims_4_cast_fp16)[name = tensor<string, []>("conv_1_cast_fp16")];
30
+ tensor<int32, [1]> squeeze_0_axes_0 = const()[name = tensor<string, []>("squeeze_0_axes_0"), val = tensor<int32, [1]>([0])];
31
+ tensor<fp16, [201, 3001]> squeeze_0_cast_fp16 = squeeze(axes = squeeze_0_axes_0, x = conv_0_cast_fp16)[name = tensor<string, []>("squeeze_0_cast_fp16")];
32
+ tensor<int32, [1]> squeeze_1_axes_0 = const()[name = tensor<string, []>("squeeze_1_axes_0"), val = tensor<int32, [1]>([0])];
33
+ tensor<fp16, [201, 3001]> squeeze_1_cast_fp16 = squeeze(axes = squeeze_1_axes_0, x = conv_1_cast_fp16)[name = tensor<string, []>("squeeze_1_cast_fp16")];
34
+ tensor<fp16, [201, 3001]> square_0_cast_fp16 = square(x = squeeze_0_cast_fp16)[name = tensor<string, []>("square_0_cast_fp16")];
35
+ tensor<fp16, [201, 3001]> square_1_cast_fp16 = square(x = squeeze_1_cast_fp16)[name = tensor<string, []>("square_1_cast_fp16")];
36
+ tensor<fp16, [201, 3001]> add_1_cast_fp16 = add(x = square_0_cast_fp16, y = square_1_cast_fp16)[name = tensor<string, []>("add_1_cast_fp16")];
37
+ tensor<fp16, [201, 3001]> magnitudes_1_cast_fp16 = identity(x = add_1_cast_fp16)[name = tensor<string, []>("magnitudes_1_cast_fp16")];
38
+ tensor<int32, [2]> magnitudes_begin_0 = const()[name = tensor<string, []>("magnitudes_begin_0"), val = tensor<int32, [2]>([0, 0])];
39
+ tensor<int32, [2]> magnitudes_end_0 = const()[name = tensor<string, []>("magnitudes_end_0"), val = tensor<int32, [2]>([201, 3000])];
40
+ tensor<bool, [2]> magnitudes_end_mask_0 = const()[name = tensor<string, []>("magnitudes_end_mask_0"), val = tensor<bool, [2]>([true, false])];
41
+ tensor<fp16, [201, 3000]> magnitudes_cast_fp16 = slice_by_index(begin = magnitudes_begin_0, end = magnitudes_end_0, end_mask = magnitudes_end_mask_0, x = magnitudes_1_cast_fp16)[name = tensor<string, []>("magnitudes_cast_fp16")];
42
+ tensor<bool, []> mel_spec_1_transpose_x_0 = const()[name = tensor<string, []>("mel_spec_1_transpose_x_0"), val = tensor<bool, []>(false)];
43
+ tensor<bool, []> mel_spec_1_transpose_y_0 = const()[name = tensor<string, []>("mel_spec_1_transpose_y_0"), val = tensor<bool, []>(false)];
44
+ tensor<fp16, [128, 201]> mel_filters_to_fp16 = const()[name = tensor<string, []>("mel_filters_to_fp16"), val = tensor<fp16, [128, 201]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(321856)))];
45
+ tensor<fp16, [128, 3000]> mel_spec_1_cast_fp16 = matmul(transpose_x = mel_spec_1_transpose_x_0, transpose_y = mel_spec_1_transpose_y_0, x = mel_filters_to_fp16, y = magnitudes_cast_fp16)[name = tensor<string, []>("mel_spec_1_cast_fp16")];
46
+ tensor<fp16, []> var_41_to_fp16 = const()[name = tensor<string, []>("op_41_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
47
+ tensor<fp16, [128, 3000]> mel_spec_cast_fp16 = add(x = mel_spec_1_cast_fp16, y = var_41_to_fp16)[name = tensor<string, []>("mel_spec_cast_fp16")];
48
+ tensor<fp16, []> log_0_epsilon_0_to_fp16 = const()[name = tensor<string, []>("log_0_epsilon_0_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
49
+ tensor<fp16, [128, 3000]> log_0_cast_fp16 = log(epsilon = log_0_epsilon_0_to_fp16, x = mel_spec_cast_fp16)[name = tensor<string, []>("log_0_cast_fp16")];
50
+ tensor<fp16, []> mul_0_y_0_to_fp16 = const()[name = tensor<string, []>("mul_0_y_0_to_fp16"), val = tensor<fp16, []>(0x1.bccp-2)];
51
+ tensor<fp16, [128, 3000]> mul_0_cast_fp16 = mul(x = log_0_cast_fp16, y = mul_0_y_0_to_fp16)[name = tensor<string, []>("mul_0_cast_fp16")];
52
+ tensor<bool, []> var_44_keep_dims_0 = const()[name = tensor<string, []>("op_44_keep_dims_0"), val = tensor<bool, []>(false)];
53
+ tensor<fp16, []> var_44_cast_fp16 = reduce_max(keep_dims = var_44_keep_dims_0, x = mul_0_cast_fp16)[name = tensor<string, []>("op_44_cast_fp16")];
54
+ tensor<fp16, []> var_46_to_fp16 = const()[name = tensor<string, []>("op_46_to_fp16"), val = tensor<fp16, []>(0x1p+3)];
55
+ tensor<fp16, []> var_47_cast_fp16 = sub(x = var_44_cast_fp16, y = var_46_to_fp16)[name = tensor<string, []>("op_47_cast_fp16")];
56
+ tensor<fp16, [128, 3000]> log_spec_3_cast_fp16 = maximum(x = mul_0_cast_fp16, y = var_47_cast_fp16)[name = tensor<string, []>("log_spec_3_cast_fp16")];
57
+ tensor<fp16, []> var_50_to_fp16 = const()[name = tensor<string, []>("op_50_to_fp16"), val = tensor<fp16, []>(0x1p+2)];
58
+ tensor<fp16, [128, 3000]> var_51_cast_fp16 = add(x = log_spec_3_cast_fp16, y = var_50_to_fp16)[name = tensor<string, []>("op_51_cast_fp16")];
59
+ tensor<fp16, []> _inversed_log_spec_y_0_to_fp16 = const()[name = tensor<string, []>("_inversed_log_spec_y_0_to_fp16"), val = tensor<fp16, []>(0x1p-2)];
60
+ tensor<fp16, [128, 3000]> _inversed_log_spec_cast_fp16 = mul(x = var_51_cast_fp16, y = _inversed_log_spec_y_0_to_fp16)[name = tensor<string, []>("_inversed_log_spec_cast_fp16")];
61
+ tensor<int32, [1]> var_55_axes_0 = const()[name = tensor<string, []>("op_55_axes_0"), val = tensor<int32, [1]>([0])];
62
+ tensor<fp16, [1, 128, 3000]> var_55_cast_fp16 = expand_dims(axes = var_55_axes_0, x = _inversed_log_spec_cast_fp16)[name = tensor<string, []>("op_55_cast_fp16")];
63
+ tensor<int32, [1]> var_62_axes_0 = const()[name = tensor<string, []>("op_62_axes_0"), val = tensor<int32, [1]>([2])];
64
+ tensor<fp16, [1, 128, 1, 3000]> melspectrogram_features = expand_dims(axes = var_62_axes_0, x = var_55_cast_fp16)[name = tensor<string, []>("op_62_cast_fp16")];
65
+ } -> (melspectrogram_features);
66
+ }
medium/MelSpectrogram.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:009d9fb8f6b589accfa08cebf1c712ef07c3405229ce3cfb3a57ee033c9d8a49
3
+ size 373376
medium/TextDecoder.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d00c58a5bc9e64c3f466ce5697c1df117006d67096917d36987a00d03f965e7e
3
+ size 243
medium/TextDecoder.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d2e36bbe66f1a4977d0965bce7c797fea3dc14250e96e03918f668669b55d49
3
+ size 694
medium/TextDecoder.mlmodelc/metadata.json ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Mixed (Float16, Palettized (8 bits), Sparse)",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 1 × 51866)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 1, 51866]",
13
+ "name" : "logits",
14
+ "type" : "MultiArray"
15
+ },
16
+ {
17
+ "hasShapeFlexibility" : "0",
18
+ "isOptional" : "0",
19
+ "dataType" : "Float16",
20
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)",
21
+ "shortDescription" : "",
22
+ "shape" : "[1, 5120, 1, 1]",
23
+ "name" : "key_cache_updates",
24
+ "type" : "MultiArray"
25
+ },
26
+ {
27
+ "hasShapeFlexibility" : "0",
28
+ "isOptional" : "0",
29
+ "dataType" : "Float16",
30
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)",
31
+ "shortDescription" : "",
32
+ "shape" : "[1, 5120, 1, 1]",
33
+ "name" : "value_cache_updates",
34
+ "type" : "MultiArray"
35
+ },
36
+ {
37
+ "hasShapeFlexibility" : "0",
38
+ "isOptional" : "0",
39
+ "dataType" : "Float16",
40
+ "formattedType" : "MultiArray (Float16 1 × 1500)",
41
+ "shortDescription" : "",
42
+ "shape" : "[1, 1500]",
43
+ "name" : "alignment_heads_weights",
44
+ "type" : "MultiArray"
45
+ }
46
+ ],
47
+ "modelParameters" : [
48
+
49
+ ],
50
+ "specificationVersion" : 7,
51
+ "mlProgramOperationTypeHistogram" : {
52
+ "Transpose" : 1,
53
+ "Squeeze" : 1,
54
+ "Ios16.gather" : 3,
55
+ "Ios16.softmax" : 8,
56
+ "Ios16.reduceMean" : 1,
57
+ "Split" : 2,
58
+ "Ios16.linear" : 1,
59
+ "Ios16.add" : 66,
60
+ "Concat" : 3,
61
+ "ExpandDims" : 6,
62
+ "Ios16.sub" : 1,
63
+ "Ios16.conv" : 80,
64
+ "Ios16.gelu" : 4,
65
+ "Ios16.constexprLutToDense" : 40,
66
+ "Ios16.constexprSparseToDense" : 41,
67
+ "Ios16.layerNorm" : 13,
68
+ "SliceByIndex" : 12,
69
+ "Ios16.matmul" : 16,
70
+ "Ios16.batchNorm" : 13,
71
+ "Ios16.reshape" : 32,
72
+ "Ios16.mul" : 24
73
+ },
74
+ "computePrecision" : "Mixed (Float16, Int32)",
75
+ "isUpdatable" : "0",
76
+ "stateSchema" : [
77
+
78
+ ],
79
+ "availability" : {
80
+ "macOS" : "13.0",
81
+ "tvOS" : "16.0",
82
+ "visionOS" : "1.0",
83
+ "watchOS" : "9.0",
84
+ "iOS" : "16.0",
85
+ "macCatalyst" : "16.0"
86
+ },
87
+ "modelType" : {
88
+ "name" : "MLModelType_mlProgram"
89
+ },
90
+ "userDefinedMetadata" : {
91
+ "com.github.apple.coremltools.conversion_date" : "2026-03-04",
92
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
93
+ "com.github.apple.coremltools.version" : "9.0",
94
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
95
+ },
96
+ "inputSchema" : [
97
+ {
98
+ "hasShapeFlexibility" : "0",
99
+ "isOptional" : "0",
100
+ "dataType" : "Int32",
101
+ "formattedType" : "MultiArray (Int32 1)",
102
+ "shortDescription" : "",
103
+ "shape" : "[1]",
104
+ "name" : "input_ids",
105
+ "type" : "MultiArray"
106
+ },
107
+ {
108
+ "hasShapeFlexibility" : "0",
109
+ "isOptional" : "0",
110
+ "dataType" : "Int32",
111
+ "formattedType" : "MultiArray (Int32 1)",
112
+ "shortDescription" : "",
113
+ "shape" : "[1]",
114
+ "name" : "cache_length",
115
+ "type" : "MultiArray"
116
+ },
117
+ {
118
+ "hasShapeFlexibility" : "0",
119
+ "isOptional" : "0",
120
+ "dataType" : "Float16",
121
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 448)",
122
+ "shortDescription" : "",
123
+ "shape" : "[1, 5120, 1, 448]",
124
+ "name" : "key_cache",
125
+ "type" : "MultiArray"
126
+ },
127
+ {
128
+ "hasShapeFlexibility" : "0",
129
+ "isOptional" : "0",
130
+ "dataType" : "Float16",
131
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 448)",
132
+ "shortDescription" : "",
133
+ "shape" : "[1, 5120, 1, 448]",
134
+ "name" : "value_cache",
135
+ "type" : "MultiArray"
136
+ },
137
+ {
138
+ "hasShapeFlexibility" : "0",
139
+ "isOptional" : "0",
140
+ "dataType" : "Float16",
141
+ "formattedType" : "MultiArray (Float16 1 × 448)",
142
+ "shortDescription" : "",
143
+ "shape" : "[1, 448]",
144
+ "name" : "kv_cache_update_mask",
145
+ "type" : "MultiArray"
146
+ },
147
+ {
148
+ "hasShapeFlexibility" : "0",
149
+ "isOptional" : "0",
150
+ "dataType" : "Float16",
151
+ "formattedType" : "MultiArray (Float16 1 × 1280 × 1 × 1500)",
152
+ "shortDescription" : "",
153
+ "shape" : "[1, 1280, 1, 1500]",
154
+ "name" : "encoder_output_embeds",
155
+ "type" : "MultiArray"
156
+ },
157
+ {
158
+ "hasShapeFlexibility" : "0",
159
+ "isOptional" : "0",
160
+ "dataType" : "Float16",
161
+ "formattedType" : "MultiArray (Float16 1 × 448)",
162
+ "shortDescription" : "",
163
+ "shape" : "[1, 448]",
164
+ "name" : "decoder_key_padding_mask",
165
+ "type" : "MultiArray"
166
+ }
167
+ ],
168
+ "generatedClassName" : "TextDecoder_mixedBitPalettized_8_bit",
169
+ "method" : "predict"
170
+ }
171
+ ]
medium/TextDecoder.mlmodelc/model.mil ADDED
The diff for this file is too large to render. See raw diff
 
medium/TextDecoder.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:24eb357578598468c1b847977551748582a0b3d912ffbe65adeb79790582b329
3
+ size 253995124
medium/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50256
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 4,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "encoder_attention_heads": 20,
23
+ "encoder_ffn_dim": 5120,
24
+ "encoder_layerdrop": 0.0,
25
+ "encoder_layers": 32,
26
+ "eos_token_id": 50257,
27
+ "forced_decoder_ids": null,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_hidden_layers": 32,
41
+ "num_mel_bins": 128,
42
+ "pad_token_id": 50257,
43
+ "scale_embedding": false,
44
+ "torch_dtype": "float32",
45
+ "transformers_version": "4.51.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
medium/generation_config.json ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 2,
5
+ 4
6
+ ],
7
+ [
8
+ 2,
9
+ 11
10
+ ],
11
+ [
12
+ 3,
13
+ 3
14
+ ],
15
+ [
16
+ 3,
17
+ 6
18
+ ],
19
+ [
20
+ 3,
21
+ 11
22
+ ],
23
+ [
24
+ 3,
25
+ 14
26
+ ]
27
+ ],
28
+ "attn_implementation": "sdpa",
29
+ "begin_suppress_tokens": [
30
+ 220,
31
+ 50257
32
+ ],
33
+ "bos_token_id": 50257,
34
+ "decoder_start_token_id": 50258,
35
+ "eos_token_id": 50257,
36
+ "forced_decoder_ids": [
37
+ [
38
+ 1,
39
+ null
40
+ ],
41
+ [
42
+ 2,
43
+ 50360
44
+ ]
45
+ ],
46
+ "is_multilingual": true,
47
+ "lang_to_id": {
48
+ "<|af|>": 50327,
49
+ "<|am|>": 50334,
50
+ "<|ar|>": 50272,
51
+ "<|as|>": 50350,
52
+ "<|az|>": 50304,
53
+ "<|ba|>": 50355,
54
+ "<|be|>": 50330,
55
+ "<|bg|>": 50292,
56
+ "<|bn|>": 50302,
57
+ "<|bo|>": 50347,
58
+ "<|br|>": 50309,
59
+ "<|bs|>": 50315,
60
+ "<|ca|>": 50270,
61
+ "<|cs|>": 50283,
62
+ "<|cy|>": 50297,
63
+ "<|da|>": 50285,
64
+ "<|de|>": 50261,
65
+ "<|el|>": 50281,
66
+ "<|en|>": 50259,
67
+ "<|es|>": 50262,
68
+ "<|et|>": 50307,
69
+ "<|eu|>": 50310,
70
+ "<|fa|>": 50300,
71
+ "<|fi|>": 50277,
72
+ "<|fo|>": 50338,
73
+ "<|fr|>": 50265,
74
+ "<|gl|>": 50319,
75
+ "<|gu|>": 50333,
76
+ "<|haw|>": 50352,
77
+ "<|ha|>": 50354,
78
+ "<|he|>": 50279,
79
+ "<|hi|>": 50276,
80
+ "<|hr|>": 50291,
81
+ "<|ht|>": 50339,
82
+ "<|hu|>": 50286,
83
+ "<|hy|>": 50312,
84
+ "<|id|>": 50275,
85
+ "<|is|>": 50311,
86
+ "<|it|>": 50274,
87
+ "<|ja|>": 50266,
88
+ "<|jw|>": 50356,
89
+ "<|ka|>": 50329,
90
+ "<|kk|>": 50316,
91
+ "<|km|>": 50323,
92
+ "<|kn|>": 50306,
93
+ "<|ko|>": 50264,
94
+ "<|la|>": 50294,
95
+ "<|lb|>": 50345,
96
+ "<|ln|>": 50353,
97
+ "<|lo|>": 50336,
98
+ "<|lt|>": 50293,
99
+ "<|lv|>": 50301,
100
+ "<|mg|>": 50349,
101
+ "<|mi|>": 50295,
102
+ "<|mk|>": 50308,
103
+ "<|ml|>": 50296,
104
+ "<|mn|>": 50314,
105
+ "<|mr|>": 50320,
106
+ "<|ms|>": 50282,
107
+ "<|mt|>": 50343,
108
+ "<|my|>": 50346,
109
+ "<|ne|>": 50313,
110
+ "<|nl|>": 50271,
111
+ "<|nn|>": 50342,
112
+ "<|no|>": 50288,
113
+ "<|oc|>": 50328,
114
+ "<|pa|>": 50321,
115
+ "<|pl|>": 50269,
116
+ "<|ps|>": 50340,
117
+ "<|pt|>": 50267,
118
+ "<|ro|>": 50284,
119
+ "<|ru|>": 50263,
120
+ "<|sa|>": 50344,
121
+ "<|sd|>": 50332,
122
+ "<|si|>": 50322,
123
+ "<|sk|>": 50298,
124
+ "<|sl|>": 50305,
125
+ "<|sn|>": 50324,
126
+ "<|so|>": 50326,
127
+ "<|sq|>": 50317,
128
+ "<|sr|>": 50303,
129
+ "<|su|>": 50357,
130
+ "<|sv|>": 50273,
131
+ "<|sw|>": 50318,
132
+ "<|ta|>": 50287,
133
+ "<|te|>": 50299,
134
+ "<|tg|>": 50331,
135
+ "<|th|>": 50289,
136
+ "<|tk|>": 50341,
137
+ "<|tl|>": 50348,
138
+ "<|tr|>": 50268,
139
+ "<|tt|>": 50351,
140
+ "<|uk|>": 50280,
141
+ "<|ur|>": 50290,
142
+ "<|uz|>": 50337,
143
+ "<|vi|>": 50278,
144
+ "<|yi|>": 50335,
145
+ "<|yo|>": 50325,
146
+ "<|yue|>": 50358,
147
+ "<|zh|>": 50260
148
+ },
149
+ "max_initial_timestamp_index": 50,
150
+ "max_length": 448,
151
+ "no_timestamps_token_id": 50364,
152
+ "pad_token_id": 50257,
153
+ "prev_sot_token_id": 50362,
154
+ "return_timestamps": false,
155
+ "suppress_tokens": [],
156
+ "task_to_id": {
157
+ "transcribe": 50360,
158
+ "translate": 50359
159
+ },
160
+ "transformers_version": "4.51.1"
161
+ }
small/AudioEncoder.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:008ccf465f33b6f52e7c1e5edc02b30cc06f80ac33cfc35f8ecd7168e11f186b
3
+ size 243
small/AudioEncoder.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7f784a1abe9f56ebc42a3ad94536b20f9c4e6f3a41f5705b4d81d06b7c097454
3
+ size 409
small/AudioEncoder.mlmodelc/metadata.json ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Mixed (Float16, Palettized (4 bits), Sparse)",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 1280 × 1 × 1500)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 1280, 1, 1500]",
13
+ "name" : "encoder_output_embeds",
14
+ "type" : "MultiArray"
15
+ }
16
+ ],
17
+ "modelParameters" : [
18
+
19
+ ],
20
+ "specificationVersion" : 7,
21
+ "mlProgramOperationTypeHistogram" : {
22
+ "Concat" : 672,
23
+ "Ios16.mul" : 3840,
24
+ "Ios16.layerNorm" : 65,
25
+ "SliceByIndex" : 5760,
26
+ "Ios16.constexprLutToDense" : 194,
27
+ "Transpose" : 32,
28
+ "Ios16.einsum" : 7680,
29
+ "Ios16.conv" : 388,
30
+ "Ios16.add" : 259,
31
+ "Ios16.constexprSparseToDense" : 192,
32
+ "Ios16.softmax" : 3840,
33
+ "Ios16.gelu" : 34,
34
+ "Ios16.batchNorm" : 65
35
+ },
36
+ "computePrecision" : "Mixed (Float16, Int32)",
37
+ "isUpdatable" : "0",
38
+ "stateSchema" : [
39
+
40
+ ],
41
+ "availability" : {
42
+ "macOS" : "13.0",
43
+ "tvOS" : "16.0",
44
+ "visionOS" : "1.0",
45
+ "watchOS" : "9.0",
46
+ "iOS" : "16.0",
47
+ "macCatalyst" : "16.0"
48
+ },
49
+ "modelType" : {
50
+ "name" : "MLModelType_mlProgram"
51
+ },
52
+ "userDefinedMetadata" : {
53
+ "com.github.apple.coremltools.conversion_date" : "2026-03-05",
54
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
55
+ "com.github.apple.coremltools.version" : "9.0",
56
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
57
+ },
58
+ "inputSchema" : [
59
+ {
60
+ "hasShapeFlexibility" : "0",
61
+ "isOptional" : "0",
62
+ "dataType" : "Float16",
63
+ "formattedType" : "MultiArray (Float16 1 × 128 × 1 × 3000)",
64
+ "shortDescription" : "",
65
+ "shape" : "[1, 128, 1, 3000]",
66
+ "name" : "melspectrogram_features",
67
+ "type" : "MultiArray"
68
+ }
69
+ ],
70
+ "generatedClassName" : "AudioEncoder_mixedBitPalettized_4_bit",
71
+ "method" : "predict"
72
+ }
73
+ ]
small/AudioEncoder.mlmodelc/model.mil ADDED
The diff for this file is too large to render. See raw diff
 
small/AudioEncoder.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8326c81c394fa31d6731324964f47b6843f1b8da6c050a85dbb505af0c7d73da
3
+ size 421928256
small/MelSpectrogram.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73b090b1ed51460119344c50a187f8bbbb2d318fccad4ef8bc04280471e03200
3
+ size 243
small/MelSpectrogram.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9f0a6e03066f401634997882fc0c48324e31ce40ff1d2a0527d88036af61795f
3
+ size 390
small/MelSpectrogram.mlmodelc/metadata.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Float16",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 128 × 1 × 3000)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 128, 1, 3000]",
13
+ "name" : "melspectrogram_features",
14
+ "type" : "MultiArray"
15
+ }
16
+ ],
17
+ "modelParameters" : [
18
+
19
+ ],
20
+ "specificationVersion" : 7,
21
+ "mlProgramOperationTypeHistogram" : {
22
+ "Ios16.reshape" : 2,
23
+ "Ios16.mul" : 2,
24
+ "SliceByIndex" : 1,
25
+ "Ios16.sub" : 1,
26
+ "Ios16.log" : 1,
27
+ "Ios16.square" : 2,
28
+ "Ios16.add" : 3,
29
+ "Squeeze" : 2,
30
+ "Ios16.matmul" : 1,
31
+ "Ios16.conv" : 2,
32
+ "Ios16.maximum" : 1,
33
+ "ExpandDims" : 4,
34
+ "Ios16.reduceMax" : 1,
35
+ "Identity" : 1,
36
+ "Pad" : 1
37
+ },
38
+ "computePrecision" : "Mixed (Float16, Int32)",
39
+ "isUpdatable" : "0",
40
+ "stateSchema" : [
41
+
42
+ ],
43
+ "availability" : {
44
+ "macOS" : "13.0",
45
+ "tvOS" : "16.0",
46
+ "visionOS" : "1.0",
47
+ "watchOS" : "9.0",
48
+ "iOS" : "16.0",
49
+ "macCatalyst" : "16.0"
50
+ },
51
+ "modelType" : {
52
+ "name" : "MLModelType_mlProgram"
53
+ },
54
+ "userDefinedMetadata" : {
55
+ "com.github.apple.coremltools.conversion_date" : "2026-03-05",
56
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
57
+ "com.github.apple.coremltools.version" : "9.0",
58
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
59
+ },
60
+ "inputSchema" : [
61
+ {
62
+ "hasShapeFlexibility" : "0",
63
+ "isOptional" : "0",
64
+ "dataType" : "Float16",
65
+ "formattedType" : "MultiArray (Float16 480000)",
66
+ "shortDescription" : "",
67
+ "shape" : "[480000]",
68
+ "name" : "audio",
69
+ "type" : "MultiArray"
70
+ }
71
+ ],
72
+ "generatedClassName" : "MelSpectrogram",
73
+ "method" : "predict"
74
+ }
75
+ ]
small/MelSpectrogram.mlmodelc/model.mil ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ program(1.0)
2
+ [buildInfo = dict<tensor<string, []>, tensor<string, []>>({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-torch", "2.5.0"}, {"coremltools-source-dialect", "TorchScript"}, {"coremltools-version", "9.0"}})]
3
+ {
4
+ func main<ios16>(tensor<fp16, [480000]> audio) {
5
+ tensor<int32, [3]> var_10 = const()[name = tensor<string, []>("op_10"), val = tensor<int32, [3]>([1, 1, 480000])];
6
+ tensor<fp16, [1, 1, 480000]> input_1_cast_fp16 = reshape(shape = var_10, x = audio)[name = tensor<string, []>("input_1_cast_fp16")];
7
+ tensor<int32, [6]> input_3_pad_0 = const()[name = tensor<string, []>("input_3_pad_0"), val = tensor<int32, [6]>([0, 0, 0, 0, 200, 200])];
8
+ tensor<string, []> input_3_mode_0 = const()[name = tensor<string, []>("input_3_mode_0"), val = tensor<string, []>("reflect")];
9
+ tensor<fp16, []> const_1_to_fp16 = const()[name = tensor<string, []>("const_1_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
10
+ tensor<fp16, [1, 1, 480400]> input_3_cast_fp16 = pad(constant_val = const_1_to_fp16, mode = input_3_mode_0, pad = input_3_pad_0, x = input_1_cast_fp16)[name = tensor<string, []>("input_3_cast_fp16")];
11
+ tensor<int32, [1]> var_22 = const()[name = tensor<string, []>("op_22"), val = tensor<int32, [1]>([480400])];
12
+ tensor<fp16, [480400]> input_cast_fp16 = reshape(shape = var_22, x = input_3_cast_fp16)[name = tensor<string, []>("input_cast_fp16")];
13
+ tensor<int32, [1]> expand_dims_0_axes_0 = const()[name = tensor<string, []>("expand_dims_0_axes_0"), val = tensor<int32, [1]>([0])];
14
+ tensor<fp16, [1, 480400]> expand_dims_0_cast_fp16 = expand_dims(axes = expand_dims_0_axes_0, x = input_cast_fp16)[name = tensor<string, []>("expand_dims_0_cast_fp16")];
15
+ tensor<int32, [1]> expand_dims_3 = const()[name = tensor<string, []>("expand_dims_3"), val = tensor<int32, [1]>([160])];
16
+ tensor<int32, [1]> expand_dims_4_axes_0 = const()[name = tensor<string, []>("expand_dims_4_axes_0"), val = tensor<int32, [1]>([1])];
17
+ tensor<fp16, [1, 1, 480400]> expand_dims_4_cast_fp16 = expand_dims(axes = expand_dims_4_axes_0, x = expand_dims_0_cast_fp16)[name = tensor<string, []>("expand_dims_4_cast_fp16")];
18
+ tensor<string, []> conv_0_pad_type_0 = const()[name = tensor<string, []>("conv_0_pad_type_0"), val = tensor<string, []>("valid")];
19
+ tensor<int32, [2]> conv_0_pad_0 = const()[name = tensor<string, []>("conv_0_pad_0"), val = tensor<int32, [2]>([0, 0])];
20
+ tensor<int32, [1]> conv_0_dilations_0 = const()[name = tensor<string, []>("conv_0_dilations_0"), val = tensor<int32, [1]>([1])];
21
+ tensor<int32, []> conv_0_groups_0 = const()[name = tensor<string, []>("conv_0_groups_0"), val = tensor<int32, []>(1)];
22
+ tensor<fp16, [201, 1, 400]> expand_dims_1_to_fp16 = const()[name = tensor<string, []>("expand_dims_1_to_fp16"), val = tensor<fp16, [201, 1, 400]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(64)))];
23
+ tensor<fp16, [1, 201, 3001]> conv_0_cast_fp16 = conv(dilations = conv_0_dilations_0, groups = conv_0_groups_0, pad = conv_0_pad_0, pad_type = conv_0_pad_type_0, strides = expand_dims_3, weight = expand_dims_1_to_fp16, x = expand_dims_4_cast_fp16)[name = tensor<string, []>("conv_0_cast_fp16")];
24
+ tensor<string, []> conv_1_pad_type_0 = const()[name = tensor<string, []>("conv_1_pad_type_0"), val = tensor<string, []>("valid")];
25
+ tensor<int32, [2]> conv_1_pad_0 = const()[name = tensor<string, []>("conv_1_pad_0"), val = tensor<int32, [2]>([0, 0])];
26
+ tensor<int32, [1]> conv_1_dilations_0 = const()[name = tensor<string, []>("conv_1_dilations_0"), val = tensor<int32, [1]>([1])];
27
+ tensor<int32, []> conv_1_groups_0 = const()[name = tensor<string, []>("conv_1_groups_0"), val = tensor<int32, []>(1)];
28
+ tensor<fp16, [201, 1, 400]> expand_dims_2_to_fp16 = const()[name = tensor<string, []>("expand_dims_2_to_fp16"), val = tensor<fp16, [201, 1, 400]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(160960)))];
29
+ tensor<fp16, [1, 201, 3001]> conv_1_cast_fp16 = conv(dilations = conv_1_dilations_0, groups = conv_1_groups_0, pad = conv_1_pad_0, pad_type = conv_1_pad_type_0, strides = expand_dims_3, weight = expand_dims_2_to_fp16, x = expand_dims_4_cast_fp16)[name = tensor<string, []>("conv_1_cast_fp16")];
30
+ tensor<int32, [1]> squeeze_0_axes_0 = const()[name = tensor<string, []>("squeeze_0_axes_0"), val = tensor<int32, [1]>([0])];
31
+ tensor<fp16, [201, 3001]> squeeze_0_cast_fp16 = squeeze(axes = squeeze_0_axes_0, x = conv_0_cast_fp16)[name = tensor<string, []>("squeeze_0_cast_fp16")];
32
+ tensor<int32, [1]> squeeze_1_axes_0 = const()[name = tensor<string, []>("squeeze_1_axes_0"), val = tensor<int32, [1]>([0])];
33
+ tensor<fp16, [201, 3001]> squeeze_1_cast_fp16 = squeeze(axes = squeeze_1_axes_0, x = conv_1_cast_fp16)[name = tensor<string, []>("squeeze_1_cast_fp16")];
34
+ tensor<fp16, [201, 3001]> square_0_cast_fp16 = square(x = squeeze_0_cast_fp16)[name = tensor<string, []>("square_0_cast_fp16")];
35
+ tensor<fp16, [201, 3001]> square_1_cast_fp16 = square(x = squeeze_1_cast_fp16)[name = tensor<string, []>("square_1_cast_fp16")];
36
+ tensor<fp16, [201, 3001]> add_1_cast_fp16 = add(x = square_0_cast_fp16, y = square_1_cast_fp16)[name = tensor<string, []>("add_1_cast_fp16")];
37
+ tensor<fp16, [201, 3001]> magnitudes_1_cast_fp16 = identity(x = add_1_cast_fp16)[name = tensor<string, []>("magnitudes_1_cast_fp16")];
38
+ tensor<int32, [2]> magnitudes_begin_0 = const()[name = tensor<string, []>("magnitudes_begin_0"), val = tensor<int32, [2]>([0, 0])];
39
+ tensor<int32, [2]> magnitudes_end_0 = const()[name = tensor<string, []>("magnitudes_end_0"), val = tensor<int32, [2]>([201, 3000])];
40
+ tensor<bool, [2]> magnitudes_end_mask_0 = const()[name = tensor<string, []>("magnitudes_end_mask_0"), val = tensor<bool, [2]>([true, false])];
41
+ tensor<fp16, [201, 3000]> magnitudes_cast_fp16 = slice_by_index(begin = magnitudes_begin_0, end = magnitudes_end_0, end_mask = magnitudes_end_mask_0, x = magnitudes_1_cast_fp16)[name = tensor<string, []>("magnitudes_cast_fp16")];
42
+ tensor<bool, []> mel_spec_1_transpose_x_0 = const()[name = tensor<string, []>("mel_spec_1_transpose_x_0"), val = tensor<bool, []>(false)];
43
+ tensor<bool, []> mel_spec_1_transpose_y_0 = const()[name = tensor<string, []>("mel_spec_1_transpose_y_0"), val = tensor<bool, []>(false)];
44
+ tensor<fp16, [128, 201]> mel_filters_to_fp16 = const()[name = tensor<string, []>("mel_filters_to_fp16"), val = tensor<fp16, [128, 201]>(BLOBFILE(path = tensor<string, []>("@model_path/weights/weight.bin"), offset = tensor<uint64, []>(321856)))];
45
+ tensor<fp16, [128, 3000]> mel_spec_1_cast_fp16 = matmul(transpose_x = mel_spec_1_transpose_x_0, transpose_y = mel_spec_1_transpose_y_0, x = mel_filters_to_fp16, y = magnitudes_cast_fp16)[name = tensor<string, []>("mel_spec_1_cast_fp16")];
46
+ tensor<fp16, []> var_41_to_fp16 = const()[name = tensor<string, []>("op_41_to_fp16"), val = tensor<fp16, []>(0x1p-24)];
47
+ tensor<fp16, [128, 3000]> mel_spec_cast_fp16 = add(x = mel_spec_1_cast_fp16, y = var_41_to_fp16)[name = tensor<string, []>("mel_spec_cast_fp16")];
48
+ tensor<fp16, []> log_0_epsilon_0_to_fp16 = const()[name = tensor<string, []>("log_0_epsilon_0_to_fp16"), val = tensor<fp16, []>(0x0p+0)];
49
+ tensor<fp16, [128, 3000]> log_0_cast_fp16 = log(epsilon = log_0_epsilon_0_to_fp16, x = mel_spec_cast_fp16)[name = tensor<string, []>("log_0_cast_fp16")];
50
+ tensor<fp16, []> mul_0_y_0_to_fp16 = const()[name = tensor<string, []>("mul_0_y_0_to_fp16"), val = tensor<fp16, []>(0x1.bccp-2)];
51
+ tensor<fp16, [128, 3000]> mul_0_cast_fp16 = mul(x = log_0_cast_fp16, y = mul_0_y_0_to_fp16)[name = tensor<string, []>("mul_0_cast_fp16")];
52
+ tensor<bool, []> var_44_keep_dims_0 = const()[name = tensor<string, []>("op_44_keep_dims_0"), val = tensor<bool, []>(false)];
53
+ tensor<fp16, []> var_44_cast_fp16 = reduce_max(keep_dims = var_44_keep_dims_0, x = mul_0_cast_fp16)[name = tensor<string, []>("op_44_cast_fp16")];
54
+ tensor<fp16, []> var_46_to_fp16 = const()[name = tensor<string, []>("op_46_to_fp16"), val = tensor<fp16, []>(0x1p+3)];
55
+ tensor<fp16, []> var_47_cast_fp16 = sub(x = var_44_cast_fp16, y = var_46_to_fp16)[name = tensor<string, []>("op_47_cast_fp16")];
56
+ tensor<fp16, [128, 3000]> log_spec_3_cast_fp16 = maximum(x = mul_0_cast_fp16, y = var_47_cast_fp16)[name = tensor<string, []>("log_spec_3_cast_fp16")];
57
+ tensor<fp16, []> var_50_to_fp16 = const()[name = tensor<string, []>("op_50_to_fp16"), val = tensor<fp16, []>(0x1p+2)];
58
+ tensor<fp16, [128, 3000]> var_51_cast_fp16 = add(x = log_spec_3_cast_fp16, y = var_50_to_fp16)[name = tensor<string, []>("op_51_cast_fp16")];
59
+ tensor<fp16, []> _inversed_log_spec_y_0_to_fp16 = const()[name = tensor<string, []>("_inversed_log_spec_y_0_to_fp16"), val = tensor<fp16, []>(0x1p-2)];
60
+ tensor<fp16, [128, 3000]> _inversed_log_spec_cast_fp16 = mul(x = var_51_cast_fp16, y = _inversed_log_spec_y_0_to_fp16)[name = tensor<string, []>("_inversed_log_spec_cast_fp16")];
61
+ tensor<int32, [1]> var_55_axes_0 = const()[name = tensor<string, []>("op_55_axes_0"), val = tensor<int32, [1]>([0])];
62
+ tensor<fp16, [1, 128, 3000]> var_55_cast_fp16 = expand_dims(axes = var_55_axes_0, x = _inversed_log_spec_cast_fp16)[name = tensor<string, []>("op_55_cast_fp16")];
63
+ tensor<int32, [1]> var_62_axes_0 = const()[name = tensor<string, []>("op_62_axes_0"), val = tensor<int32, [1]>([2])];
64
+ tensor<fp16, [1, 128, 1, 3000]> melspectrogram_features = expand_dims(axes = var_62_axes_0, x = var_55_cast_fp16)[name = tensor<string, []>("op_62_cast_fp16")];
65
+ } -> (melspectrogram_features);
66
+ }
small/MelSpectrogram.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:009d9fb8f6b589accfa08cebf1c712ef07c3405229ce3cfb3a57ee033c9d8a49
3
+ size 373376
small/TextDecoder.mlmodelc/analytics/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d00c58a5bc9e64c3f466ce5697c1df117006d67096917d36987a00d03f965e7e
3
+ size 243
small/TextDecoder.mlmodelc/coremldata.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d2e36bbe66f1a4977d0965bce7c797fea3dc14250e96e03918f668669b55d49
3
+ size 694
small/TextDecoder.mlmodelc/metadata.json ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "metadataOutputVersion" : "3.0",
4
+ "storagePrecision" : "Mixed (Float16, Palettized (8 bits), Sparse)",
5
+ "outputSchema" : [
6
+ {
7
+ "hasShapeFlexibility" : "0",
8
+ "isOptional" : "0",
9
+ "dataType" : "Float16",
10
+ "formattedType" : "MultiArray (Float16 1 × 1 × 51866)",
11
+ "shortDescription" : "",
12
+ "shape" : "[1, 1, 51866]",
13
+ "name" : "logits",
14
+ "type" : "MultiArray"
15
+ },
16
+ {
17
+ "hasShapeFlexibility" : "0",
18
+ "isOptional" : "0",
19
+ "dataType" : "Float16",
20
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)",
21
+ "shortDescription" : "",
22
+ "shape" : "[1, 5120, 1, 1]",
23
+ "name" : "key_cache_updates",
24
+ "type" : "MultiArray"
25
+ },
26
+ {
27
+ "hasShapeFlexibility" : "0",
28
+ "isOptional" : "0",
29
+ "dataType" : "Float16",
30
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 1)",
31
+ "shortDescription" : "",
32
+ "shape" : "[1, 5120, 1, 1]",
33
+ "name" : "value_cache_updates",
34
+ "type" : "MultiArray"
35
+ },
36
+ {
37
+ "hasShapeFlexibility" : "0",
38
+ "isOptional" : "0",
39
+ "dataType" : "Float16",
40
+ "formattedType" : "MultiArray (Float16 1 × 1500)",
41
+ "shortDescription" : "",
42
+ "shape" : "[1, 1500]",
43
+ "name" : "alignment_heads_weights",
44
+ "type" : "MultiArray"
45
+ }
46
+ ],
47
+ "modelParameters" : [
48
+
49
+ ],
50
+ "specificationVersion" : 7,
51
+ "mlProgramOperationTypeHistogram" : {
52
+ "Transpose" : 1,
53
+ "Squeeze" : 1,
54
+ "Ios16.gather" : 3,
55
+ "Ios16.softmax" : 8,
56
+ "Ios16.reduceMean" : 1,
57
+ "Split" : 2,
58
+ "Ios16.linear" : 1,
59
+ "Ios16.add" : 66,
60
+ "Concat" : 3,
61
+ "ExpandDims" : 6,
62
+ "Ios16.sub" : 1,
63
+ "Ios16.conv" : 80,
64
+ "Ios16.gelu" : 4,
65
+ "Ios16.constexprLutToDense" : 40,
66
+ "Ios16.constexprSparseToDense" : 41,
67
+ "Ios16.layerNorm" : 13,
68
+ "SliceByIndex" : 12,
69
+ "Ios16.matmul" : 16,
70
+ "Ios16.batchNorm" : 13,
71
+ "Ios16.reshape" : 32,
72
+ "Ios16.mul" : 24
73
+ },
74
+ "computePrecision" : "Mixed (Float16, Int32)",
75
+ "isUpdatable" : "0",
76
+ "stateSchema" : [
77
+
78
+ ],
79
+ "availability" : {
80
+ "macOS" : "13.0",
81
+ "tvOS" : "16.0",
82
+ "visionOS" : "1.0",
83
+ "watchOS" : "9.0",
84
+ "iOS" : "16.0",
85
+ "macCatalyst" : "16.0"
86
+ },
87
+ "modelType" : {
88
+ "name" : "MLModelType_mlProgram"
89
+ },
90
+ "userDefinedMetadata" : {
91
+ "com.github.apple.coremltools.conversion_date" : "2026-03-04",
92
+ "com.github.apple.coremltools.source" : "torch==2.5.0",
93
+ "com.github.apple.coremltools.version" : "9.0",
94
+ "com.github.apple.coremltools.source_dialect" : "TorchScript"
95
+ },
96
+ "inputSchema" : [
97
+ {
98
+ "hasShapeFlexibility" : "0",
99
+ "isOptional" : "0",
100
+ "dataType" : "Int32",
101
+ "formattedType" : "MultiArray (Int32 1)",
102
+ "shortDescription" : "",
103
+ "shape" : "[1]",
104
+ "name" : "input_ids",
105
+ "type" : "MultiArray"
106
+ },
107
+ {
108
+ "hasShapeFlexibility" : "0",
109
+ "isOptional" : "0",
110
+ "dataType" : "Int32",
111
+ "formattedType" : "MultiArray (Int32 1)",
112
+ "shortDescription" : "",
113
+ "shape" : "[1]",
114
+ "name" : "cache_length",
115
+ "type" : "MultiArray"
116
+ },
117
+ {
118
+ "hasShapeFlexibility" : "0",
119
+ "isOptional" : "0",
120
+ "dataType" : "Float16",
121
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 448)",
122
+ "shortDescription" : "",
123
+ "shape" : "[1, 5120, 1, 448]",
124
+ "name" : "key_cache",
125
+ "type" : "MultiArray"
126
+ },
127
+ {
128
+ "hasShapeFlexibility" : "0",
129
+ "isOptional" : "0",
130
+ "dataType" : "Float16",
131
+ "formattedType" : "MultiArray (Float16 1 × 5120 × 1 × 448)",
132
+ "shortDescription" : "",
133
+ "shape" : "[1, 5120, 1, 448]",
134
+ "name" : "value_cache",
135
+ "type" : "MultiArray"
136
+ },
137
+ {
138
+ "hasShapeFlexibility" : "0",
139
+ "isOptional" : "0",
140
+ "dataType" : "Float16",
141
+ "formattedType" : "MultiArray (Float16 1 × 448)",
142
+ "shortDescription" : "",
143
+ "shape" : "[1, 448]",
144
+ "name" : "kv_cache_update_mask",
145
+ "type" : "MultiArray"
146
+ },
147
+ {
148
+ "hasShapeFlexibility" : "0",
149
+ "isOptional" : "0",
150
+ "dataType" : "Float16",
151
+ "formattedType" : "MultiArray (Float16 1 × 1280 × 1 × 1500)",
152
+ "shortDescription" : "",
153
+ "shape" : "[1, 1280, 1, 1500]",
154
+ "name" : "encoder_output_embeds",
155
+ "type" : "MultiArray"
156
+ },
157
+ {
158
+ "hasShapeFlexibility" : "0",
159
+ "isOptional" : "0",
160
+ "dataType" : "Float16",
161
+ "formattedType" : "MultiArray (Float16 1 × 448)",
162
+ "shortDescription" : "",
163
+ "shape" : "[1, 448]",
164
+ "name" : "decoder_key_padding_mask",
165
+ "type" : "MultiArray"
166
+ }
167
+ ],
168
+ "generatedClassName" : "TextDecoder_mixedBitPalettized_8_bit",
169
+ "method" : "predict"
170
+ }
171
+ ]
small/TextDecoder.mlmodelc/model.mil ADDED
The diff for this file is too large to render. See raw diff
 
small/TextDecoder.mlmodelc/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:24eb357578598468c1b847977551748582a0b3d912ffbe65adeb79790582b329
3
+ size 253995124
small/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50256
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 4,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "encoder_attention_heads": 20,
23
+ "encoder_ffn_dim": 5120,
24
+ "encoder_layerdrop": 0.0,
25
+ "encoder_layers": 32,
26
+ "eos_token_id": 50257,
27
+ "forced_decoder_ids": null,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_hidden_layers": 32,
41
+ "num_mel_bins": 128,
42
+ "pad_token_id": 50257,
43
+ "scale_embedding": false,
44
+ "torch_dtype": "float32",
45
+ "transformers_version": "4.51.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
small/generation_config.json ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 2,
5
+ 4
6
+ ],
7
+ [
8
+ 2,
9
+ 11
10
+ ],
11
+ [
12
+ 3,
13
+ 3
14
+ ],
15
+ [
16
+ 3,
17
+ 6
18
+ ],
19
+ [
20
+ 3,
21
+ 11
22
+ ],
23
+ [
24
+ 3,
25
+ 14
26
+ ]
27
+ ],
28
+ "attn_implementation": "sdpa",
29
+ "begin_suppress_tokens": [
30
+ 220,
31
+ 50257
32
+ ],
33
+ "bos_token_id": 50257,
34
+ "decoder_start_token_id": 50258,
35
+ "eos_token_id": 50257,
36
+ "forced_decoder_ids": [
37
+ [
38
+ 1,
39
+ null
40
+ ],
41
+ [
42
+ 2,
43
+ 50360
44
+ ]
45
+ ],
46
+ "is_multilingual": true,
47
+ "lang_to_id": {
48
+ "<|af|>": 50327,
49
+ "<|am|>": 50334,
50
+ "<|ar|>": 50272,
51
+ "<|as|>": 50350,
52
+ "<|az|>": 50304,
53
+ "<|ba|>": 50355,
54
+ "<|be|>": 50330,
55
+ "<|bg|>": 50292,
56
+ "<|bn|>": 50302,
57
+ "<|bo|>": 50347,
58
+ "<|br|>": 50309,
59
+ "<|bs|>": 50315,
60
+ "<|ca|>": 50270,
61
+ "<|cs|>": 50283,
62
+ "<|cy|>": 50297,
63
+ "<|da|>": 50285,
64
+ "<|de|>": 50261,
65
+ "<|el|>": 50281,
66
+ "<|en|>": 50259,
67
+ "<|es|>": 50262,
68
+ "<|et|>": 50307,
69
+ "<|eu|>": 50310,
70
+ "<|fa|>": 50300,
71
+ "<|fi|>": 50277,
72
+ "<|fo|>": 50338,
73
+ "<|fr|>": 50265,
74
+ "<|gl|>": 50319,
75
+ "<|gu|>": 50333,
76
+ "<|haw|>": 50352,
77
+ "<|ha|>": 50354,
78
+ "<|he|>": 50279,
79
+ "<|hi|>": 50276,
80
+ "<|hr|>": 50291,
81
+ "<|ht|>": 50339,
82
+ "<|hu|>": 50286,
83
+ "<|hy|>": 50312,
84
+ "<|id|>": 50275,
85
+ "<|is|>": 50311,
86
+ "<|it|>": 50274,
87
+ "<|ja|>": 50266,
88
+ "<|jw|>": 50356,
89
+ "<|ka|>": 50329,
90
+ "<|kk|>": 50316,
91
+ "<|km|>": 50323,
92
+ "<|kn|>": 50306,
93
+ "<|ko|>": 50264,
94
+ "<|la|>": 50294,
95
+ "<|lb|>": 50345,
96
+ "<|ln|>": 50353,
97
+ "<|lo|>": 50336,
98
+ "<|lt|>": 50293,
99
+ "<|lv|>": 50301,
100
+ "<|mg|>": 50349,
101
+ "<|mi|>": 50295,
102
+ "<|mk|>": 50308,
103
+ "<|ml|>": 50296,
104
+ "<|mn|>": 50314,
105
+ "<|mr|>": 50320,
106
+ "<|ms|>": 50282,
107
+ "<|mt|>": 50343,
108
+ "<|my|>": 50346,
109
+ "<|ne|>": 50313,
110
+ "<|nl|>": 50271,
111
+ "<|nn|>": 50342,
112
+ "<|no|>": 50288,
113
+ "<|oc|>": 50328,
114
+ "<|pa|>": 50321,
115
+ "<|pl|>": 50269,
116
+ "<|ps|>": 50340,
117
+ "<|pt|>": 50267,
118
+ "<|ro|>": 50284,
119
+ "<|ru|>": 50263,
120
+ "<|sa|>": 50344,
121
+ "<|sd|>": 50332,
122
+ "<|si|>": 50322,
123
+ "<|sk|>": 50298,
124
+ "<|sl|>": 50305,
125
+ "<|sn|>": 50324,
126
+ "<|so|>": 50326,
127
+ "<|sq|>": 50317,
128
+ "<|sr|>": 50303,
129
+ "<|su|>": 50357,
130
+ "<|sv|>": 50273,
131
+ "<|sw|>": 50318,
132
+ "<|ta|>": 50287,
133
+ "<|te|>": 50299,
134
+ "<|tg|>": 50331,
135
+ "<|th|>": 50289,
136
+ "<|tk|>": 50341,
137
+ "<|tl|>": 50348,
138
+ "<|tr|>": 50268,
139
+ "<|tt|>": 50351,
140
+ "<|uk|>": 50280,
141
+ "<|ur|>": 50290,
142
+ "<|uz|>": 50337,
143
+ "<|vi|>": 50278,
144
+ "<|yi|>": 50335,
145
+ "<|yo|>": 50325,
146
+ "<|yue|>": 50358,
147
+ "<|zh|>": 50260
148
+ },
149
+ "max_initial_timestamp_index": 50,
150
+ "max_length": 448,
151
+ "no_timestamps_token_id": 50364,
152
+ "pad_token_id": 50257,
153
+ "prev_sot_token_id": 50362,
154
+ "return_timestamps": false,
155
+ "suppress_tokens": [],
156
+ "task_to_id": {
157
+ "transcribe": 50360,
158
+ "translate": 50359
159
+ },
160
+ "transformers_version": "4.51.1"
161
+ }