Transformers
learning_first_tokenizer / tokenizer.json
Melih1234's picture
Upload tokenizer
62686d8 verified
Raw
History Blame Contribute Delete
11.4 kB
{
"version": "1.0",
"truncation": null,
"padding": null,
"added_tokens": [
{
"id": 0,
"content": "<unk>",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
}
],
"normalizer": null,
"pre_tokenizer": {
"type": "Whitespace"
},
"post_processor": null,
"decoder": null,
"model": {
"type": "BPE",
"dropout": null,
"unk_token": null,
"continuing_subword_prefix": null,
"end_of_word_suffix": null,
"fuse_unk": false,
"byte_fallback": false,
"ignore_merges": false,
"vocab": {
"<unk>": 0,
",": 1,
".": 2,
"a": 3,
"b": 4,
"c": 5,
"d": 6,
"e": 7,
"f": 8,
"g": 9,
"h": 10,
"i": 11,
"k": 12,
"l": 13,
"m": 14,
"n": 15,
"o": 16,
"p": 17,
"q": 18,
"r": 19,
"s": 20,
"t": 21,
"u": 22,
"w": 23,
"y": 24,
"it": 25,
"is": 26,
"th": 27,
"the": 28,
"an": 29,
"al": 30,
"in": 31,
"ital": 32,
"ap": 33,
"cap": 34,
"capital": 35,
"of": 36,
"or": 37,
"on": 38,
"and": 39,
"ar": 40,
"un": 41,
"ur": 42,
"ac": 43,
"ed": 44,
"en": 45,
"no": 46,
"es": 47,
"ma": 48,
"as": 49,
"eac": 50,
"ri": 51,
"ited": 52,
"united": 53,
"each": 54,
"eur": 55,
"op": 56,
"europ": 57,
"europe": 58,
"se": 59,
"wn": 60,
"these": 61,
"not": 62,
"mad": 63,
"ent": 64,
"his": 65,
"tor": 66,
"capitals": 67,
"are": 68,
"histor": 69,
"history": 70,
"at": 71,
"cu": 72,
"lt": 73,
"om": 74,
"st": 75,
"ing": 76,
"ure": 77,
"ates": 78,
"cult": 79,
"states": 80,
"culture": 81,
"ge": 82,
"por": 83,
"its": 84,
"port": 85,
"ch": 86,
"ion": 87,
"lac": 88,
"plac": 89,
"wit": 90,
"ity": 91,
"any": 92,
"rich": 93,
"with": 94,
"co": 95,
"ry": 96,
"try": 97,
"ug": 98,
"art": 99,
"untry": 100,
"ash": 101,
"country": 102,
"ment": 103,
"oge": 104,
"ten": 105,
"toge": 106,
"ther": 107,
"often": 108,
"made": 109,
"ioned": 110,
"mentioned": 111,
"together": 112,
"be": 113,
"for": 114,
"lin": 115,
"own": 116,
"par": 117,
"rom": 118,
"rlin": 119,
"ton": 120,
"wash": 121,
"ington": 122,
"berlin": 123,
"paris": 124,
"rome": 125,
"washington": 126,
"has": 127,
"im": 128,
"kno": 129,
"ant": 130,
"european": 131,
"portant": 132,
"places": 133,
"important": 134,
"known": 135,
"bon": 136,
"don": 137,
"lis": 138,
"lon": 139,
"many": 140,
"rid": 141,
"madrid": 142,
"lisbon": 143,
"london": 144,
"city": 145,
"dent": 146,
"ident": 147,
"ld": 148,
"oug": 149,
"wor": 150,
"thoug": 151,
"althoug": 152,
"identity": 153,
"world": 154,
"although": 155,
"dom": 156,
"king": 157,
"portug": 158,
"kingdom": 159,
"portugal": 160,
"ema": 161,
"hi": 162,
"le": 163,
"rema": 164,
"whi": 165,
"place": 166,
"remain": 167,
"while": 168,
"ain": 169,
"ce": 170,
"fr": 171,
"iq": 172,
"pain": 173,
"rmany": 174,
"spain": 175,
"ue": 176,
"ance": 177,
"italy": 178,
"uniq": 179,
"germany": 180,
"france": 181,
"unique": 182,
"fash": 183,
"fashion": 184,
"am": 185,
"fam": 186,
"ou": 187,
"they": 188,
"famou": 189,
"famous": 190
},
"merges": [
[
"i",
"t"
],
[
"i",
"s"
],
[
"t",
"h"
],
[
"th",
"e"
],
[
"a",
"n"
],
[
"a",
"l"
],
[
"i",
"n"
],
[
"it",
"al"
],
[
"a",
"p"
],
[
"c",
"ap"
],
[
"cap",
"ital"
],
[
"o",
"f"
],
[
"o",
"r"
],
[
"o",
"n"
],
[
"an",
"d"
],
[
"a",
"r"
],
[
"u",
"n"
],
[
"u",
"r"
],
[
"a",
"c"
],
[
"e",
"d"
],
[
"e",
"n"
],
[
"n",
"o"
],
[
"e",
"s"
],
[
"m",
"a"
],
[
"a",
"s"
],
[
"e",
"ac"
],
[
"r",
"i"
],
[
"it",
"ed"
],
[
"un",
"ited"
],
[
"eac",
"h"
],
[
"e",
"ur"
],
[
"o",
"p"
],
[
"eur",
"op"
],
[
"europ",
"e"
],
[
"s",
"e"
],
[
"w",
"n"
],
[
"the",
"se"
],
[
"no",
"t"
],
[
"ma",
"d"
],
[
"en",
"t"
],
[
"h",
"is"
],
[
"t",
"or"
],
[
"capital",
"s"
],
[
"ar",
"e"
],
[
"his",
"tor"
],
[
"histor",
"y"
],
[
"a",
"t"
],
[
"c",
"u"
],
[
"l",
"t"
],
[
"o",
"m"
],
[
"s",
"t"
],
[
"in",
"g"
],
[
"ur",
"e"
],
[
"at",
"es"
],
[
"cu",
"lt"
],
[
"st",
"ates"
],
[
"cult",
"ure"
],
[
"g",
"e"
],
[
"p",
"or"
],
[
"it",
"s"
],
[
"por",
"t"
],
[
"c",
"h"
],
[
"i",
"on"
],
[
"l",
"ac"
],
[
"p",
"lac"
],
[
"w",
"it"
],
[
"it",
"y"
],
[
"an",
"y"
],
[
"ri",
"ch"
],
[
"wit",
"h"
],
[
"c",
"o"
],
[
"r",
"y"
],
[
"t",
"ry"
],
[
"u",
"g"
],
[
"ar",
"t"
],
[
"un",
"try"
],
[
"as",
"h"
],
[
"co",
"untry"
],
[
"m",
"ent"
],
[
"o",
"ge"
],
[
"t",
"en"
],
[
"t",
"oge"
],
[
"the",
"r"
],
[
"of",
"ten"
],
[
"mad",
"e"
],
[
"ion",
"ed"
],
[
"ment",
"ioned"
],
[
"toge",
"ther"
],
[
"b",
"e"
],
[
"f",
"or"
],
[
"l",
"in"
],
[
"o",
"wn"
],
[
"p",
"ar"
],
[
"r",
"om"
],
[
"r",
"lin"
],
[
"t",
"on"
],
[
"w",
"ash"
],
[
"ing",
"ton"
],
[
"be",
"rlin"
],
[
"par",
"is"
],
[
"rom",
"e"
],
[
"wash",
"ington"
],
[
"h",
"as"
],
[
"i",
"m"
],
[
"k",
"no"
],
[
"an",
"t"
],
[
"europe",
"an"
],
[
"port",
"ant"
],
[
"plac",
"es"
],
[
"im",
"portant"
],
[
"kno",
"wn"
],
[
"b",
"on"
],
[
"d",
"on"
],
[
"l",
"is"
],
[
"l",
"on"
],
[
"m",
"any"
],
[
"ri",
"d"
],
[
"mad",
"rid"
],
[
"lis",
"bon"
],
[
"lon",
"don"
],
[
"c",
"ity"
],
[
"d",
"ent"
],
[
"i",
"dent"
],
[
"l",
"d"
],
[
"o",
"ug"
],
[
"w",
"or"
],
[
"th",
"oug"
],
[
"al",
"thoug"
],
[
"ident",
"ity"
],
[
"wor",
"ld"
],
[
"althoug",
"h"
],
[
"d",
"om"
],
[
"k",
"ing"
],
[
"port",
"ug"
],
[
"king",
"dom"
],
[
"portug",
"al"
],
[
"e",
"ma"
],
[
"h",
"i"
],
[
"l",
"e"
],
[
"r",
"ema"
],
[
"w",
"hi"
],
[
"plac",
"e"
],
[
"rema",
"in"
],
[
"whi",
"le"
],
[
"a",
"in"
],
[
"c",
"e"
],
[
"f",
"r"
],
[
"i",
"q"
],
[
"p",
"ain"
],
[
"r",
"many"
],
[
"s",
"pain"
],
[
"u",
"e"
],
[
"an",
"ce"
],
[
"ital",
"y"
],
[
"un",
"iq"
],
[
"ge",
"rmany"
],
[
"fr",
"ance"
],
[
"uniq",
"ue"
],
[
"f",
"ash"
],
[
"fash",
"ion"
],
[
"a",
"m"
],
[
"f",
"am"
],
[
"o",
"u"
],
[
"the",
"y"
],
[
"fam",
"ou"
],
[
"famou",
"s"
]
]
}
}