Add custom tokenizer

Files changed (4) hide show

special_tokens_map.json ADDED Viewed

+{
+  "pad_token": "<pad>"
+}

tokenization_vulberta.py ADDED Viewed

+from typing import List
+from tokenizers import NormalizedString, PreTokenizedString
+from tokenizers.pre_tokenizers import PreTokenizer
+from transformers import PreTrainedTokenizerFast
+try:
+    from clang import cindex
+except ModuleNotFoundError as e:
+    raise ModuleNotFoundError(
+        "VulBERTa Clang tokenizer requires `libclang`. Please install it via `pip install libclang`.",
+    ) from e
+class ClangPreTokenizer:
+    cidx = cindex.Index.create()
+    def clang_split(
+        self,
+        i: int,
+        normalized_string: NormalizedString,
+    ) -> List[NormalizedString]:
+        tok = []
+        tu = self.cidx.parse(
+            "tmp.c",
+            args=[""],
+            unsaved_files=[("tmp.c", str(normalized_string.original))],
+            options=0,
+        )
+        for t in tu.get_tokens(extent=tu.cursor.extent):
+            spelling = t.spelling.strip()
+            if spelling == "":
+                continue
+            tok.append(NormalizedString(spelling))
+        return tok
+    def pre_tokenize(self, pretok: PreTokenizedString):
+        pretok.split(self.clang_split)
+class VulBERTaTokenizer(PreTrainedTokenizerFast):
+    def __init__(
+        self,
+        *args,
+        **kwargs,
+    ):
+        super().__init__(
+            *args,
+            **kwargs,
+        )
+        self._tokenizer.pre_tokenizer = PreTokenizer.custom(ClangPreTokenizer())

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

+{
+  "added_tokens_decoder": {
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": true,
+  "max_length": 1024,
+  "model_max_length": 1024,
+  "pad_to_multiple_of": null,
+  "pad_token": "<pad>",
+  "pad_token_type_id": 0,
+  "padding_side": "right",
+  "stride": 0,
+  "tokenizer_class": "VulBERTaTokenizer",
+  "auto_map": {
+    "AutoTokenizer": ["tokenization_vulberta.VulBERTaTokenizer", null]
+  },
+  "truncation_side": "right",
+  "truncation_strategy": "longest_first"
+}