MEDvqa-GI2T / tokenizer.json
SushantGautam's picture
Upload tokenizer.json
b4a16c2
Raw
History Blame Contribute Delete
12.8 kB
{
"version": "1.0",
"truncation": null,
"padding": null,
"added_tokens": [
{
"id": 0,
"content": "[UNK]",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 1,
"content": "[CLS]",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 2,
"content": "[SEP]",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 3,
"content": "[PAD]",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
},
{
"id": 4,
"content": "[MASK]",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
}
],
"normalizer": null,
"pre_tokenizer": {
"type": "Whitespace"
},
"post_processor": {
"type": "TemplateProcessing",
"single": [
{
"SpecialToken": {
"id": "[CLS]",
"type_id": 0
}
},
{
"Sequence": {
"id": "A",
"type_id": 0
}
},
{
"SpecialToken": {
"id": "[SEP]",
"type_id": 0
}
}
],
"pair": [
{
"SpecialToken": {
"id": "[CLS]",
"type_id": 0
}
},
{
"Sequence": {
"id": "A",
"type_id": 0
}
},
{
"SpecialToken": {
"id": "[SEP]",
"type_id": 0
}
},
{
"Sequence": {
"id": "B",
"type_id": 1
}
},
{
"SpecialToken": {
"id": "[SEP]",
"type_id": 1
}
}
],
"special_tokens": {
"[CLS]": {
"id": "[CLS]",
"ids": [
1
],
"tokens": [
"[CLS]"
]
},
"[SEP]": {
"id": "[SEP]",
"ids": [
2
],
"tokens": [
"[SEP]"
]
}
}
},
"decoder": null,
"model": {
"type": "BPE",
"dropout": null,
"unk_token": null,
"continuing_subword_prefix": null,
"end_of_word_suffix": null,
"fuse_unk": false,
"vocab": {
"[UNK]": 0,
"[CLS]": 1,
"[SEP]": 2,
"[PAD]": 3,
"[MASK]": 4,
"-": 5,
"/": 6,
"0": 7,
"1": 8,
"2": 9,
"3": 10,
"4": 11,
"5": 12,
"6": 13,
"<": 14,
">": 15,
"?": 16,
"A": 17,
"B": 18,
"C": 19,
"G": 20,
"H": 21,
"I": 22,
"L": 23,
"M": 24,
"N": 25,
"O": 26,
"P": 27,
"R": 28,
"T": 29,
"U": 30,
"V": 31,
"W": 32,
"Y": 33,
"Z": 34,
"a": 35,
"b": 36,
"c": 37,
"d": 38,
"e": 39,
"f": 40,
"g": 41,
"h": 42,
"i": 43,
"j": 44,
"k": 45,
"l": 46,
"m": 47,
"n": 48,
"o": 49,
"p": 50,
"r": 51,
"s": 52,
"t": 53,
"u": 54,
"v": 55,
"w": 56,
"x": 57,
"y": 58,
"z": 59,
"he": 60,
"the": 61,
"ma": 62,
"re": 63,
"in": 64,
"is": 65,
"an": 66,
"ge": 67,
"ima": 68,
"image": 69,
"en": 70,
"or": 71,
"at": 72,
"it": 73,
"er": 74,
"ent": 75,
"Whe": 76,
"Where": 77,
"lit": 78,
"ab": 79,
"nor": 80,
"malit": 81,
"abnor": 82,
"abnormalit": 83,
"ol": 84,
"No": 85,
"abnormality": 86,
"le": 87,
"Wh": 88,
"yp": 89,
"What": 90,
"van": 91,
"rele": 92,
"Not": 93,
"vant": 94,
"relevant": 95,
"ow": 96,
"al": 97,
"om": 98,
"there": 99,
"enter": 100,
"olyp": 101,
"cal": 102,
"dma": 103,
"ical": 104,
"lan": 105,
"rk": 106,
"anat": 107,
"omical": 108,
"dmark": 109,
"landmark": 110,
"anatomical": 111,
"polyp": 112,
"st": 113,
"str": 114,
"um": 115,
"instr": 116,
"instrum": 117,
"col": 118,
"color": 119,
"are": 120,
"of": 121,
"Are": 122,
"any": 123,
"es": 124,
"How": 125,
"ny": 126,
"many": 127,
"Is": 128,
"te": 129,
"Center": 130,
"ed": 131,
"instrument": 132,
"Up": 133,
"per": 134,
"Upper": 135,
"ig": 136,
"rig": 137,
"Low": 138,
"Lower": 139,
"ht": 140,
"right": 141,
"ft": 142,
"left": 143,
"pre": 144,
"sent": 145,
"present": 146,
"typ": 147,
"type": 148,
"ac": 149,
"din": 150,
"fin": 151,
"ding": 152,
"finding": 153,
"polyps": 154,
"center": 155,
"Yes": 156,
"ar": 157,
"op": 158,
"ve": 159,
"be": 160,
"th": 161,
"sy": 162,
"instruments": 163,
"iz": 164,
"siz": 165,
"size": 166,
"et": 167,
"gre": 168,
"lac": 169,
"lack": 170,
"ies": 171,
"abnormalities": 172,
"ct": 173,
"findings": 174,
"Ha": 175,
"ak": 176,
"asy": 177,
"bo": 178,
"black": 179,
"ced": 180,
"cop": 181,
"de": 182,
"easy": 183,
"fr": 184,
"fac": 185,
"mo": 186,
"net": 187,
"os": 188,
"oced": 189,
"pr": 190,
"to": 191,
"tak": 192,
"ure": 193,
"ved": 194,
"xt": 195,
"remo": 196,
"all": 197,
"landmarks": 198,
"instrumnet": 199,
"tect": 200,
"tefac": 201,
"text": 202,
"artefac": 203,
"been": 204,
"this": 205,
"green": 206,
"Have": 207,
"box": 208,
"copy": 209,
"detect": 210,
"from": 211,
"oscopy": 212,
"ocedure": 213,
"procedure": 214,
"taken": 215,
"removed": 216,
"instrumnets": 217,
"artefact": 218,
"Pin": 219,
"Pink": 220,
"on": 221,
"Col": 222,
"onoscopy": 223,
"Colonoscopy": 224,
"Red": 225,
"ite": 226,
"White": 227,
"Polyp": 228,
"Par": 229,
"mm": 230,
"Paris": 231,
"0mm": 232,
"Ga": 233,
"stroscopy": 234,
"Gastroscopy": 235,
"Ul": 236,
"co": 237,
"cer": 238,
"ive": 239,
"ative": 240,
"litis": 241,
"Ulcer": 242,
"colitis": 243,
"Ulcerative": 244,
"Oes": 245,
"ag": 246,
"hag": 247,
"itis": 248,
"ophag": 249,
"Oesophag": 250,
"Oesophagitis": 251,
"lin": 252,
"line": 253,
"Tu": 254,
"Tube": 255,
"20mm": 256,
"ip": 257,
"ia": 258,
"iia": 259,
"10mm": 260,
"rigth": 261,
"11": 262,
"Ye": 263,
"ll": 264,
"Yell": 265,
"Yellow": 266,
"5mm": 267,
"Bi": 268,
"ce": 269,
"for": 270,
"ps": 271,
"opsy": 272,
"Biopsy": 273,
"ceps": 274,
"forceps": 275,
"Ce": 276,
"cum": 277,
"Cecum": 278,
"nare": 279,
"snare": 280,
"Gre": 281,
"row": 282,
"rown": 283,
"Or": 284,
"ange": 285,
"Orange": 286,
"Grey": 287,
"Brown": 288,
"Met": 289,
"cl": 290,
"Metal": 291,
"clip": 292,
"grey": 293,
"Black": 294,
"Bl": 295,
"ue": 296,
"Blue": 297,
"ur": 298,
"brown": 299,
"ple": 300,
"urple": 301,
"Ile": 302,
"Ileum": 303,
"purple": 304,
"Py": 305,
"Purple": 306,
"lor": 307,
"us": 308,
"Green": 309,
"Pylor": 310,
"Pylorus": 311,
"16": 312,
"In": 313,
"Bar": 314,
"Pa": 315,
"Vi": 316,
"bur": 317,
"dy": 318,
"eed": 319,
"ect": 320,
"gu": 321,
"ion": 322,
"ject": 323,
"ndy": 324,
"need": 325,
"ts": 326,
"tts": 327,
"retts": 328,
"olet": 329,
"Ink": 330,
"Inject": 331,
"Barretts": 332,
"Pale": 333,
"Violet": 334,
"burgu": 335,
"needle": 336,
"Injection": 337,
"burgundy": 338
},
"merges": [
"h e",
"t he",
"m a",
"r e",
"i n",
"i s",
"a n",
"g e",
"i ma",
"ima ge",
"e n",
"o r",
"a t",
"i t",
"e r",
"en t",
"W he",
"Whe re",
"l it",
"a b",
"n or",
"ma lit",
"ab nor",
"abnor malit",
"o l",
"N o",
"abnormalit y",
"l e",
"W h",
"y p",
"Wh at",
"v an",
"re le",
"No t",
"van t",
"rele vant",
"o w",
"a l",
"o m",
"the re",
"ent er",
"ol yp",
"c al",
"d ma",
"i cal",
"l an",
"r k",
"an at",
"om ical",
"dma rk",
"lan dmark",
"anat omical",
"p olyp",
"s t",
"st r",
"u m",
"in str",
"instr um",
"c ol",
"col or",
"a re",
"o f",
"A re",
"an y",
"e s",
"H ow",
"n y",
"ma ny",
"I s",
"t e",
"C enter",
"e d",
"instrum ent",
"U p",
"p er",
"Up per",
"i g",
"r ig",
"L ow",
"Low er",
"h t",
"rig ht",
"f t",
"le ft",
"p re",
"s ent",
"pre sent",
"t yp",
"typ e",
"a c",
"d in",
"f in",
"din g",
"fin ding",
"polyp s",
"c enter",
"Y es",
"a r",
"o p",
"v e",
"b e",
"t h",
"s y",
"instrument s",
"i z",
"s iz",
"siz e",
"e t",
"g re",
"l ac",
"lac k",
"i es",
"abnormalit ies",
"c t",
"finding s",
"H a",
"a k",
"a sy",
"b o",
"b lack",
"c ed",
"c op",
"d e",
"e asy",
"f r",
"f ac",
"m o",
"n et",
"o s",
"o ced",
"p r",
"t o",
"t ak",
"u re",
"v ed",
"x t",
"re mo",
"al l",
"landmark s",
"instrum net",
"te ct",
"te fac",
"te xt",
"ar tefac",
"be en",
"th is",
"gre en",
"Ha ve",
"bo x",
"cop y",
"de tect",
"fr om",
"os copy",
"oced ure",
"pr ocedure",
"tak en",
"remo ved",
"instrumnet s",
"artefac t",
"P in",
"Pin k",
"o n",
"C ol",
"on oscopy",
"Col onoscopy",
"R ed",
"it e",
"Wh ite",
"P olyp",
"P ar",
"m m",
"Par is",
"0 mm",
"G a",
"str oscopy",
"Ga stroscopy",
"U l",
"c o",
"c er",
"i ve",
"at ive",
"lit is",
"Ul cer",
"co litis",
"Ulcer ative",
"O es",
"a g",
"h ag",
"it is",
"op hag",
"Oes ophag",
"Oesophag itis",
"l in",
"lin e",
"T u",
"Tu be",
"2 0mm",
"i p",
"i a",
"i ia",
"1 0mm",
"rig th",
"1 1",
"Y e",
"l l",
"Ye ll",
"Yell ow",
"5 mm",
"B i",
"c e",
"f or",
"p s",
"op sy",
"Bi opsy",
"ce ps",
"for ceps",
"C e",
"c um",
"Ce cum",
"n are",
"s nare",
"G re",
"r ow",
"row n",
"O r",
"an ge",
"Or ange",
"Gre y",
"B rown",
"M et",
"c l",
"Met al",
"cl ip",
"gre y",
"B lack",
"B l",
"u e",
"Bl ue",
"u r",
"b rown",
"p le",
"ur ple",
"I le",
"Ile um",
"p urple",
"P y",
"P urple",
"l or",
"u s",
"Gre en",
"Py lor",
"Pylor us",
"1 6",
"I n",
"B ar",
"P a",
"V i",
"b ur",
"d y",
"e ed",
"e ct",
"g u",
"i on",
"j ect",
"n dy",
"n eed",
"t s",
"t ts",
"re tts",
"ol et",
"In k",
"In ject",
"Bar retts",
"Pa le",
"Vi olet",
"bur gu",
"need le",
"Inject ion",
"burgu ndy"
]
}
}