Transformers
amharic_dialect_tokenizer / tokenizer_config.json
Bedru's picture
Train Amharic dialect-aware BPE tokenizer
3cda224 verified
Raw
History Blame Contribute Delete
383 Bytes
{
"add_prefix_space": false,
"backend": "tokenizers",
"bos_token": "<|startoftranscript|>",
"clean_up_tokenization_spaces": false,
"eos_token": "<|endoftext|>",
"language": null,
"model_max_length": 1000000000000000019884624838656,
"pad_token": "<pad>",
"predict_timestamps": false,
"task": null,
"tokenizer_class": "WhisperTokenizer",
"unk_token": "<unk>"
}