Upload whisper-tiny-en

Browse files

Files changed (6) hide show

.gitattributes +0 -1
README.md +42 -0
config.json +232 -0
model.bin +3 -0
tokenizer.json +0 -0
vocabulary.txt +0 -0

.gitattributes CHANGED Viewed

@@ -25,7 +25,6 @@
 *.safetensors filter=lfs diff=lfs merge=lfs -text
 saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
-*.tar filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
 *.wasm filter=lfs diff=lfs merge=lfs -text

 *.safetensors filter=lfs diff=lfs merge=lfs -text
 saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
 *.wasm filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,42 @@

+---
+language:
+  - en
+tags:
+  - audio
+  - automatic-speech-recognition
+license: mit
+library_name: ctranslate2
+---
+# Whisper tiny.en model for CTranslate2
+This repository contains the conversion of [openai/whisper-tiny.en](https://huggingface.co/openai/whisper-tiny.en) to the [CTranslate2](https://github.com/OpenNMT/CTranslate2) model format.
+This model can be used in CTranslate2 or projects based on CTranslate2 such as [faster-whisper](https://github.com/systran/faster-whisper).
+## Example
+```python
+from faster_whisper import WhisperModel
+model = WhisperModel("tiny.en")
+segments, info = model.transcribe("audio.mp3")
+for segment in segments:
+    print("[%.2fs -> %.2fs] %s" % (segment.start, segment.end, segment.text))
+```
+## Conversion details
+The original model was converted with the following command:
+```
+ct2-transformers-converter --model openai/whisper-tiny.en --output_dir faster-whisper-tiny.en \
+    --copy_files tokenizer.json --quantization float16
+```
+Note that the model weights are saved in FP16. This type can be changed when the model is loaded using the [`compute_type` option in CTranslate2](https://opennmt.net/CTranslate2/quantization.html).
+## More information
+**For more information about the original model, see its [model card](https://huggingface.co/openai/whisper-tiny.en).**

config.json ADDED Viewed

	@@ -0,0 +1,232 @@

+{
+  "alignment_heads": [
+    [
+      1,
+      0
+    ],
+    [
+      2,
+      0
+    ],
+    [
+      2,
+      5
+    ],
+    [
+      3,
+      0
+    ],
+    [
+      3,
+      1
+    ],
+    [
+      3,
+      2
+    ],
+    [
+      3,
+      3
+    ],
+    [
+      3,
+      4
+    ]
+  ],
+  "lang_ids": [
+    50259,
+    50260,
+    50261,
+    50262,
+    50263,
+    50264,
+    50265,
+    50266,
+    50267,
+    50268,
+    50269,
+    50270,
+    50271,
+    50272,
+    50273,
+    50274,
+    50275,
+    50276,
+    50277,
+    50278,
+    50279,
+    50280,
+    50281,
+    50282,
+    50283,
+    50284,
+    50285,
+    50286,
+    50287,
+    50288,
+    50289,
+    50290,
+    50291,
+    50292,
+    50293,
+    50294,
+    50295,
+    50296,
+    50297,
+    50298,
+    50299,
+    50300,
+    50301,
+    50302,
+    50303,
+    50304,
+    50305,
+    50306,
+    50307,
+    50308,
+    50309,
+    50310,
+    50311,
+    50312,
+    50313,
+    50314,
+    50315,
+    50316,
+    50317,
+    50318,
+    50319,
+    50320,
+    50321,
+    50322,
+    50323,
+    50324,
+    50325,
+    50326,
+    50327,
+    50328,
+    50329,
+    50330,
+    50331,
+    50332,
+    50333,
+    50334,
+    50335,
+    50336,
+    50337,
+    50338,
+    50339,
+    50340,
+    50341,
+    50342,
+    50343,
+    50344,
+    50345,
+    50346,
+    50347,
+    50348,
+    50349,
+    50350,
+    50351,
+    50352,
+    50353,
+    50354,
+    50355,
+    50356
+  ],
+  "suppress_ids": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    357,
+    366,
+    438,
+    532,
+    685,
+    705,
+    796,
+    930,
+    1058,
+    1220,
+    1267,
+    1279,
+    1303,
+    1343,
+    1377,
+    1391,
+    1635,
+    1782,
+    1875,
+    2162,
+    2361,
+    2488,
+    3467,
+    4008,
+    4211,
+    4600,
+    4808,
+    5299,
+    5855,
+    6329,
+    7203,
+    9609,
+    9959,
+    10563,
+    10786,
+    11420,
+    11709,
+    11907,
+    13163,
+    13697,
+    13700,
+    14808,
+    15306,
+    16410,
+    16791,
+    17992,
+    19203,
+    19510,
+    20724,
+    22305,
+    22935,
+    27007,
+    30109,
+    30420,
+    33409,
+    34949,
+    40283,
+    40493,
+    40549,
+    47282,
+    49146,
+    50257,
+    50357,
+    50358,
+    50359,
+    50360,
+    50361
+  ],
+  "suppress_ids_begin": [
+    220,
+    50256
+  ]
+}

model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1a5afae06a4db91c975c9a9d78be5cc110ee4ea022ad57d55492e4550e936b2a
+size 75537502

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocabulary.txt ADDED Viewed

The diff for this file is too large to render. See raw diff