LegolasTheElf commited on
Commit
9c1fdc8
·
1 Parent(s): 685a5d9

Bengali Wav2Vec2 with LM !!!

Browse files
.gitattributes CHANGED
@@ -25,3 +25,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
25
  *.zip filter=lfs diff=lfs merge=lfs -text
26
  *.zstandard filter=lfs diff=lfs merge=lfs -text
27
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
25
  *.zip filter=lfs diff=lfs merge=lfs -text
26
  *.zstandard filter=lfs diff=lfs merge=lfs -text
27
  *tfevents* filter=lfs diff=lfs merge=lfs -text
28
+ language_model/unigrams.txt filter=lfs diff=lfs merge=lfs -text
added_tokens.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"<s>": 78, "</s>": 79}
alphabet.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"labels": ["\u09ee", "\u09cd", "\u09ab", "\u099c", "\u09b2", "\u09a5", "\u09ec", "\u09af", "\u0996", "\u0989", "\u09c2", "\u0982", "\u0987", "\u09a1", "\u09aa", "\u09e7", "\u09a0", "\u0993", "\u0997", "\u09c0", "\u09e9", "\u0988", "\u099d", "\u09ce", "\u09c1", "\u09c3", "\u09dc", "\u098f", "\u09ef", "\u09a3", "\u099f", "\u09ed", "\u09b7", "\u09d7", "\u09b6", "\u09a6", "\u09a4", "\u099b", "\u09a7", "\u0995", "\u0981", "\u098b", "\u09df", "\u09f0", "\u09ac", "\u09b0", "\u098a", "\u0990", "\u09ea", "\u09cb", "\u099e", "\u09be", "\u09e8", "\u09ae", "\u0986", "\u0999", "\u09a8", "\u09b9", "\u09dd", "\u09a2", "\u09cc", "\u099a", "\u09bf", "\u09c8", "\u0994", "\u0998", " ", "\u0964", "\u09e6", "\u09eb", "\u09b8", "\u09bc", "\u0983", "\u09c7", "\u0985", "\u09ad", "\u2047", "", "<s>", "</s>"], "is_bpe": false}
language_model/5gram.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:93a6d4a176fff70071bf2d89ea723c5e4955dc8c2b2b98fbd8ee5196cd22fd2b
3
+ size 1658384767
language_model/attrs.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"alpha": 0.5, "beta": 1.0, "unk_score_offset": -10.0, "score_boundary": true}
language_model/unigrams.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aefc41cfe6c7ea3e80caba1cc72754c7bfca3f9a02581804298ac00e28c521fe
3
+ size 27964585
special_tokens_map.json CHANGED
@@ -1 +1 @@
1
- {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]"}
 
1
+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]", "additional_special_tokens": [{"content": "<s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}, {"content": "</s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}]}
tokenizer_config.json CHANGED
@@ -1 +1 @@
1
- {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
 
1
+ {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "special_tokens_map_file": "/root/.cache/huggingface/transformers/b7ed9ed7f9536585036d29aefdefe3a1c12c855a23b117b5460c972dd02babd3.a21d51735cf8667bcd610f057e88548d5d6a381401f6b4501a8bc6c1a9dc8498", "tokenizer_file": null, "name_or_path": "LegolasTheElf/Wav2vec2_XLSR_Bengali", "tokenizer_class": "Wav2Vec2CTCTokenizer"}