first commit - version 202_base-en_v1 model run. trained without augmentation

Files changed (16) hide show

CKPT.yaml +7 -0
brain.ckpt +3 -0
config.json +138 -0
counter.ckpt +3 -0
dataloader-TRAIN.ckpt +3 -0
decoder.ckpt +3 -0
infer_hf_whisper.yaml +75 -0
optimizer.ckpt +3 -0
preprocessor_config.json +0 -0
scaler.ckpt +3 -0
scheduler_whisper.ckpt +3 -0
tokenizer.ckpt +3 -0
tokenizer.json +0 -0
tokenizer_config.json +36 -0
vocab.json +0 -0
whisper.ckpt +3 -0

CKPT.yaml ADDED Viewed

	@@ -0,0 +1,7 @@

+# yamllint disable
+WER: 77.8046811945117
+deletions: 1108
+end-of-epoch: true
+insertions: 1593
+substitutions: 2119
+unixtime: 1682145122.214084

brain.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:222d610754b85e9dcd4fa864baf9e838d3066ea1e79b95355624f47e714729eb
+size 50

config.json ADDED Viewed

	@@ -0,0 +1,138 @@

+{
+  "_name_or_path": "openai/whisper-base.en",
+  "activation_dropout": 0.0,
+  "activation_function": "gelu",
+  "architectures": [
+    "WhisperForConditionalGeneration"
+  ],
+  "attention_dropout": 0.0,
+  "begin_suppress_tokens": [
+    220,
+    50256
+  ],
+  "bos_token_id": 50257,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 50257,
+  "dropout": 0.0,
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 50256,
+  "forced_decoder_ids": [
+    [
+      1,
+      50362
+    ]
+  ],
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "max_length": 448,
+  "max_source_positions": 1500,
+  "max_target_positions": 448,
+  "model_type": "whisper",
+  "num_hidden_layers": 6,
+  "num_mel_bins": 80,
+  "pad_token_id": 50256,
+  "scale_embedding": false,
+  "suppress_tokens": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    357,
+    366,
+    438,
+    532,
+    685,
+    705,
+    796,
+    930,
+    1058,
+    1220,
+    1267,
+    1279,
+    1303,
+    1343,
+    1377,
+    1391,
+    1635,
+    1782,
+    1875,
+    2162,
+    2361,
+    2488,
+    3467,
+    4008,
+    4211,
+    4600,
+    4808,
+    5299,
+    5855,
+    6329,
+    7203,
+    9609,
+    9959,
+    10563,
+    10786,
+    11420,
+    11709,
+    11907,
+    13163,
+    13697,
+    13700,
+    14808,
+    15306,
+    16410,
+    16791,
+    17992,
+    19203,
+    19510,
+    20724,
+    22305,
+    22935,
+    27007,
+    30109,
+    30420,
+    33409,
+    34949,
+    40283,
+    40493,
+    40549,
+    47282,
+    49146,
+    50257,
+    50357,
+    50358,
+    50359,
+    50360,
+    50361
+  ],
+  "torch_dtype": "float32",
+  "transformers_version": "4.27.0.dev0",
+  "use_cache": true,
+  "vocab_size": 51864
+}

counter.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4e07408562bedb8b60ce05c1decfe3ad16b72230967de01f640b7e4729b49fce
+size 1

dataloader-TRAIN.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:721d978b344436223d253b925b1b5fb9965247f76cfddb6172e9f3d7a7c69a62
+size 5

decoder.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:09cd10d25028e9b2ae84581a7579a04a4987b41cfb51a731aedeead7feb06e69
+size 290532565

infer_hf_whisper.yaml ADDED Viewed

	@@ -0,0 +1,75 @@

+# ################################
+# Model: Whisper (Encoder-Decoder) + NLL
+# Augmentation: TimeDomainSpecAugment
+# Authors: Adel Moumen 2022, Titouan Parcollet 2022, Rosy Southwell 2023
+# ################################
+model_src: rosyvs/whisat-base
+model_type: tiny.en
+language: english
+auto_mix_prec: False # TODO: set to True for CUDA
+only_encoder: False
+# These values are only used for the searchers.
+# They needs to be hardcoded and should not be changed with Whisper.
+# They are used as part of the searching process.
+# The bos token of the searcher will be timestamp_index
+# and will be concatenated with the bos, language and task tokens.
+timestamp_index: 50363
+eos_index: 50257
+bos_index: 50258
+# Decoding parameters
+min_decode_ratio: 0.0
+max_decode_ratio: 1.0 # the commonvoice inference yaml uses 0.1
+test_beam_size: 5 # TODO: this was 8, changing to 5 as this is the default used by openAI
+whisper: !new:speechbrain.lobes.models.huggingface_whisper.HuggingFaceWhisper
+    encoder_only: !ref <only_encoder>
+    freeze: True
+    freeze_encoder: True
+    source: !ref <model_src>
+    save_path: !ref <cache_dir>
+    # language: language
+# tokenizer: !new:speechbrain.lobes.models.huggingface_whisper.HuggingFaceWhisper
+#     encoder_only: False
+#     freeze: True
+#     freeze_encoder: True
+#     source: !ref openai/whisper-<model_type>
+#     save_path: !ref <cache_dir>
+# decoder: !new:speechbrain.decoders.seq2seq.S2SWhisperGreedySearch
+#     model: !ref <whisper>
+#     bos_index: !ref <timestamp_index>
+#     eos_index: !ref <eos_index>
+#     min_decode_ratio: !ref <min_decode_ratio>
+#     max_decode_ratio: !ref <max_decode_ratio>
+decoder: !new:speechbrain.decoders.seq2seq.S2SWhisperBeamSearch
+    module: [!ref <whisper>]
+    bos_index: !ref <timestamp_index>
+    eos_index: !ref <eos_index>
+    min_decode_ratio: !ref <min_decode_ratio>
+    max_decode_ratio: !ref <max_decode_ratio>
+    beam_size: !ref <test_beam_size>
+modules:
+    whisper: !ref <whisper>
+    # tokenizer: !ref <tokenizer>
+    decoder: !ref <decoder> # can change to greedy
+pretrainer: !new:speechbrain.utils.parameter_transfer.Pretrainer
+    loadables:
+        whisper: !ref <whisper>
+        # tokenizer: !ref <tokenizer>
+    #     decoder:  !ref <decoder>
+    # paths:
+    #     whisper: !ref <model_src>/whisper.ckpt
+        # tokenizer: !ref openai/whisper-<model_type>
+    #     decoder: !ref <model_src>/whisper.ckpt
+# checkpointer: !new:speechbrain.utils.checkpoints.Checkpointer
+#     checkpoints_dir: !ref <model_src>
+#     recoverables:
+#         whisper: !ref <whisper>

optimizer.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c63a201ce77e2b680fe1cf6957c1dbc5143dc6c23602934a4328a5bdbdcd71ed
+size 580951685

preprocessor_config.json ADDED Viewed

The diff for this file is too large to render. See raw diff

scaler.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:882f6f4d929d3900d5e93202d3d027a1caad91a342f2c084a5e26dad638e087e
+size 557

scheduler_whisper.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:efb8b0ea85f579ac77f43c3d0e01d6bdde8b963e5d0e8d41d9dfbfa452ddb0a0
+size 515

tokenizer.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ee0076ed1a26ff3b91988c6bc709947094aaba3f7571c1c2fca25aa5bc855056
+size 290531013

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "bos_token": {
+    "__type": "AddedToken",
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "__type": "AddedToken",
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "errors": "replace",
+  "model_max_length": 1024,
+  "name_or_path": "openai/whisper-base.en",
+  "pad_token": null,
+  "processor_class": "WhisperProcessor",
+  "return_attention_mask": false,
+  "special_tokens_map_file": null,
+  "tokenizer_class": "WhisperTokenizer",
+  "unk_token": {
+    "__type": "AddedToken",
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

whisper.ckpt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c9ff69b0535385083a7ff21cb8b813b5cb2adeb86a295fd751bdbbe9e474b00e
+size 290529941