Upload 6 files

Browse files

Files changed (6) hide show

model.safetensors +3 -0
special_tokens_map.json +15 -0
tokenizer.json +150 -0
tokenizer_config.json +58 -0
training_args.bin +3 -0
vocab.json +1 -0

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:83a17787139659fafcabb361a78662b68ca140dfbdb9441fa517309ed3f7ed8a
+size 177328292

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token": "<s>",
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "unk_token": "<unk>"
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,150 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "<s>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "</s>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "<mask>",
+      "single_word": false,
+      "lstrip": true,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 4,
+      "content": "<unk>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": null,
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": "(\\[[^\\]]+]|Br?|Cl?|N|O|S|P|F|I|b|c|n|o|s|p|\\(|\\)|\\.|=|#|-|\\+|\\\\|\\/|:|~|@|\\?|>>?|\\*|\\$|%[0-9]{2}|[0-9])"
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "RobertaProcessing",
+    "sep": [
+      "</s>",
+      1
+    ],
+    "cls": [
+      "<s>",
+      0
+    ],
+    "trim_offsets": true,
+    "add_prefix_space": false
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "<s>": 0,
+      "</s>": 1,
+      "<pad>": 2,
+      "<mask>": 3,
+      "<unk>": 4,
+      ")": 5,
+      "c": 6,
+      "C": 7,
+      "O": 8,
+      "6": 9,
+      "=": 10,
+      "N": 11,
+      "\r\n": 12,
+      "n": 13,
+      "5": 14,
+      "[C@@H]": 15,
+      "[C@H]": 16,
+      "F": 17,
+      "S": 18,
+      "Cl": 19,
+      "s": 20,
+      "9": 21,
+      "-": 22,
+      "[NH+]": 23,
+      "o": 24,
+      "/": 25,
+      "%10": 26,
+      "[O-]": 27,
+      "[nH]": 28,
+      "[NH2+]": 29,
+      "3": 30,
+      "#": 31,
+      "Br": 32,
+      "[N+]": 33,
+      "7": 34,
+      "[C@]": 35,
+      "[NH3+]": 36,
+      "[C@@]": 37,
+      "\\": 38,
+      "[nH+]": 39,
+      "8": 40,
+      "4": 41,
+      "%13": 42,
+      "%11": 43,
+      "[N-]": 44,
+      "[S@@]": 45,
+      "[S@]": 46,
+      "I": 47,
+      "[n+]": 48,
+      "%14": 49,
+      "%12": 50,
+      "[S-]": 51,
+      "[n-]": 52,
+      "%17": 53,
+      "%15": 54,
+      "%16": 55,
+      "P": 56,
+      "%18": 57,
+      "[P@]": 58,
+      "%20": 59,
+      "[P@@]": 60,
+      "%19": 61,
+      "%21": 62,
+      "[OH+]": 63,
+      "%22": 64,
+      "[CH-]": 65,
+      "[NH-]": 66,
+      "[SH+]": 67,
+      "[o+]": 68
+    },
+    "unk_token": "<unk>"
+  }
+}

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,58 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<mask>",
+      "lstrip": true,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "tokenizer_class": "RobertaTokenizer",
+  "trim_offsets": true,
+  "unk_token": "<unk>"
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9bf20daccd968dc0c1a695eef872525400b874f71fb92995bf7a56f851885e52
+size 5841

vocab.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"<s>":0,"</s>":1,"<pad>":2,"<mask>":3,"<unk>":4,")":5,"c":6,"C":7,"O":8,"6":9,"=":10,"N":11,"\r\n":12,"n":13,"5":14,"[C@@H]":15,"[C@H]":16,"F":17,"S":18,"Cl":19,"s":20,"9":21,"-":22,"[NH+]":23,"o":24,"/":25,"%10":26,"[O-]":27,"[nH]":28,"[NH2+]":29,"3":30,"#":31,"Br":32,"[N+]":33,"7":34,"[C@]":35,"[NH3+]":36,"[C@@]":37,"\\":38,"[nH+]":39,"8":40,"4":41,"%13":42,"%11":43,"[N-]":44,"[S@@]":45,"[S@]":46,"I":47,"[n+]":48,"%14":49,"%12":50,"[S-]":51,"[n-]":52,"%17":53,"%15":54,"%16":55,"P":56,"%18":57,"[P@]":58,"%20":59,"[P@@]":60,"%19":61,"%21":62,"[OH+]":63,"%22":64,"[CH-]":65,"[NH-]":66,"[SH+]":67,"[o+]":68}