{ "version": "1.0", "truncation": null, "padding": null, "added_tokens": [ { "id": 0, "content": "$", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 97, "content": "[UNK]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 98, "content": "", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 99, "content": "", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true } ], "normalizer": null, "pre_tokenizer": { "type": "Split", "pattern": { "String": "" }, "behavior": "Isolated", "invert": false }, "post_processor": { "type": "TemplateProcessing", "single": [ { "Sequence": { "id": "A", "type_id": 0 } } ], "pair": [ { "Sequence": { "id": "A", "type_id": 0 } }, { "Sequence": { "id": "B", "type_id": 1 } } ], "special_tokens": {} }, "decoder": null, "model": { "type": "WordLevel", "vocab": { "$": 0, "ア": 1, "イ": 2, "ウ": 3, "エ": 4, "オ": 5, "カ": 6, "キ": 7, "ク": 8, "ケ": 9, "コ": 10, "サ": 11, "シ": 12, "ス": 13, "セ": 14, "ソ": 15, "タ": 16, "チ": 17, "ツ": 18, "テ": 19, "ト": 20, "ナ": 21, "ニ": 22, "ヌ": 23, "ネ": 24, "ノ": 25, "ハ": 26, "ヒ": 27, "フ": 28, "ヘ": 29, "ホ": 30, "マ": 31, "ミ": 32, "ム": 33, "メ": 34, "モ": 35, "ヤ": 36, "ユ": 37, "ヨ": 38, "ワ": 39, "ヲ": 40, "ン": 41, "ガ": 42, "ギ": 43, "グ": 44, "ゲ": 45, "ゴ": 46, "ザ": 47, "ジ": 48, "ズ": 49, "ゼ": 50, "ゾ": 51, "ダ": 52, "ヂ": 53, "ヅ": 54, "デ": 55, "ド": 56, "パ": 57, "ピ": 58, "プ": 59, "ペ": 60, "ポ": 61, "ァ": 62, "ィ": 63, "ゥ": 64, "ェ": 65, "ォ": 66, "ャ": 67, "ュ": 68, "ョ": 69, "ヴ": 70, "ー": 71, "_": 72, "/": 73, "'": 74, "ッ": 75, "": 76, "": 77, "": 78, "": 79, "": 80, "": 81, "": 82, "": 83, "": 84, "": 85, "": 86, "": 87, "": 88, "": 89, "": 90, "": 91, "": 92, "": 93, "": 94, "": 95, "": 96, "[UNK]": 97, "": 98, "": 99 }, "unk_token": "[UNK]" } }