tclong commited on
Commit
8af0afd
·
1 Parent(s): 83d4b43

add tokenizer

Browse files
Files changed (3) hide show
  1. special_tokens_map.json +1 -1
  2. tokenizer_config.json +1 -1
  3. vocab.json +1 -1
special_tokens_map.json CHANGED
@@ -1 +1 @@
1
- {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]", "additional_special_tokens": [{"content": "<s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}, {"content": "</s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}]}
 
1
+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]"}
tokenizer_config.json CHANGED
@@ -1 +1 @@
1
- {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": "|", "special_tokens_map_file": null, "tokenizer_file": null, "name_or_path": "./", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
 
1
+ {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": "|", "replace_word_delimiter_char": " ", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
vocab.json CHANGED
@@ -1 +1 @@
1
- {"": 0, "": 1, "": 2, "n": 3, "": 4, "ũ": 5, "": 6, "": 7, "": 8, "": 9, "ư": 10, "à": 11, "": 12, "ĩ": 13, "r": 14, "": 15, "ó": 16, "d": 17, "": 18, "ý": 19, "": 20, "õ": 21, "u": 22, "": 23, "": 24, "a": 25, "": 26, "": 27, "": 28, "": 29, "": 30, "y": 31, "": 32, "ơ": 33, "t": 34, "è": 35, "": 36, "đ": 37, "x": 38, "": 39, "é": 40, "": 41, "ù": 43, "": 44, "": 45, "": 46, "p": 47, "": 48, "â": 49, "": 50, "": 51, "ì": 52, "c": 53, "q": 54, "": 55, "l": 56, "": 57, "": 58, "": 59, "4": 60, "ò": 61, "á": 62, "e": 63, "í": 64, "v": 65, "ú": 66, "ă": 67, "ê": 68, "": 69, "": 70, "": 71, "m": 72, "h": 73, "b": 74, "": 75, "": 76, "ế": 77, "o": 78, "": 79, "s": 80, "g": 81, "": 82, "": 83, "ã": 84, "i": 85, "k": 86, "": 87, "": 88, "": 89, "ô": 90, "|": 42, "[UNK]": 91, "[PAD]": 92}
 
1
+ {"ô": 0, "": 1, "l": 2, "é": 3, "p": 4, "": 5, "": 6, "n": 7, "": 8, "ơ": 9, "e": 10, "": 11, "ă": 12, "â": 13, "": 14, "": 15, "": 16, "": 17, "": 18, "": 19, "ý": 20, "": 21, "à": 22, "g": 23, "ế": 24, "": 25, "": 26, "": 27, "a": 28, "": 29, "è": 30, "b": 31, "k": 32, "r": 33, "o": 34, "v": 35, "": 36, "": 37, "q": 38, "": 39, "": 40, "ũ": 41, "á": 42, "ợ": 43, "": 44, "": 45, "ó": 46, "ĩ": 47, "c": 48, "m": 49, "": 50, "": 52, "": 53, "ù": 54, "ê": 55, "x": 56, "": 57, "": 58, "": 59, "": 60, "": 61, "s": 62, "d": 63, "": 64, "": 65, "í": 66, "": 67, "ì": 68, "": 69, "": 70, "": 71, "h": 72, "u": 73, "ò": 74, "": 75, "ú": 76, "i": 77, "": 78, "õ": 79, "t": 80, "": 81, "ã": 82, "4": 83, "": 84, "đ": 85, "y": 86, "ư": 87, "": 88, "": 89, "": 90, "|": 51, "[UNK]": 91, "[PAD]": 92}