Upload folder using huggingface_hub

Browse files

Files changed (5) hide show

config.json +33 -0
model.safetensors +3 -0
special_tokens_map.json +12 -0
tokenizer.json +163 -0
tokenizer_config.json +82 -0

config.json ADDED Viewed

	@@ -0,0 +1,33 @@

+{
+  "architectures": [
+    "MixtralForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "bos_token_id": 15,
+  "eos_token_id": 16,
+  "head_dim": null,
+  "hidden_act": "silu",
+  "hidden_size": 1024,
+  "initializer_range": 0.01,
+  "intermediate_size": 4096,
+  "iter": 118698,
+  "max_position_embeddings": 1048576,
+  "model_type": "mixtral",
+  "num_attention_heads": 16,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 12,
+  "num_key_value_heads": 8,
+  "num_local_experts": 8,
+  "output_router_logits": false,
+  "pad_token_id": 14,
+  "rms_norm_eps": 1e-05,
+  "rope_theta": 50000000,
+  "router_aux_loss_coef": 0.001,
+  "router_jitter_noise": 0.0,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "torch_dtype": "float32",
+  "transformers_version": "4.52.4",
+  "use_cache": true,
+  "vocab_size": 128
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9a325f2cdf8071f2039f9a3b1073302cdd3d371b4a47722643c01cc8a54db211
+size 4984424416

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,12 @@

+{
+  "additional_special_tokens": [
+    "<EOD>"
+  ],
+  "bos_token": "<s>",
+  "cls_token": "<CLS>",
+  "eos_token": "</s>",
+  "mask_token": "<MASK>",
+  "pad_token": "<PAD>",
+  "sep_token": "<SEP>",
+  "unk_token": "<UNK>"
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,163 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 10,
+      "content": "<CLS>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 11,
+      "content": "<SEP>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 12,
+      "content": "<EOD>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<MASK>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<PAD>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 15,
+      "content": "<s>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 16,
+      "content": "</s>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 17,
+      "content": "<UNK>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": null,
+  "pre_tokenizer": {
+    "type": "ByteLevel",
+    "add_prefix_space": false,
+    "trim_offsets": true,
+    "use_regex": true
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 0
+        }
+      }
+    ],
+    "special_tokens": {
+      "<CLS>": {
+        "id": "<CLS>",
+        "ids": [
+          10
+        ],
+        "tokens": [
+          "<CLS>"
+        ]
+      },
+      "<SEP>": {
+        "id": "<SEP>",
+        "ids": [
+          11
+        ],
+        "tokens": [
+          "<SEP>"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "BPE",
+    "dropout": null,
+    "unk_token": "<UNK>",
+    "continuing_subword_prefix": null,
+    "end_of_word_suffix": null,
+    "fuse_unk": false,
+    "byte_fallback": false,
+    "ignore_merges": false,
+    "vocab": {
+      "[PAD]": 0,
+      "[UNK]": 1,
+      "[CLS]": 2,
+      "[SEP]": 3,
+      "[MASK]": 4,
+      "C": 5,
+      "G": 6,
+      "T": 7,
+      "A": 8,
+      "N": 9,
+      "<CLS>": 10,
+      "<SEP>": 11,
+      "<EOD>": 12,
+      "<MASK>": 13,
+      "<PAD>": 14,
+      "<s>": 15,
+      "</s>": 16,
+      "<UNK>": 17
+    },
+    "merges": []
+  }
+}

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,82 @@

+{
+  "added_tokens_decoder": {
+    "10": {
+      "content": "<CLS>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<SEP>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "<EOD>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<MASK>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<PAD>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "17": {
+      "content": "<UNK>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<EOD>"
+  ],
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "<CLS>",
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "mask_token": "<MASK>",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<PAD>",
+  "sep_token": "<SEP>",
+  "tokenizer_class": "PreTrainedTokenizer",
+  "unk_token": "<UNK>"
+}