GLM-ASR-Nano-2512 / tokenizer_config.json
eustlb's picture
eustlb HF Staff
Upload processor
bda9ff9 verified
raw
history blame contribute delete
817 Bytes
{
"backend": "tokenizers",
"clean_up_tokenization_spaces": false,
"do_lower_case": false,
"eos_token": "<|endoftext|>",
"extra_special_tokens": [
"<|endoftext|>",
"[MASK]",
"[gMASK]",
"[sMASK]",
"<sop>",
"<eop>",
"<|system|>",
"<|user|>",
"<|assistant|>",
"<|observation|>",
"<|begin_of_image|>",
"<|end_of_image|>",
"<|begin_of_video|>",
"<|end_of_video|>",
"<|pad|>",
"<|begin_of_audio|>",
"<|end_of_audio|>"
],
"is_local": false,
"model_input_names": [
"input_ids",
"attention_mask"
],
"model_max_length": 65536,
"model_specific_special_tokens": {},
"pad_token": "<|endoftext|>",
"padding_side": "left",
"processor_class": "GlmAsrProcessor",
"remove_space": false,
"tokenizer_class": "TokenizersBackend"
}