Upload model

Files changed (6) hide show

config.json CHANGED Viewed

@@ -1,5 +1,5 @@
 {
-  "_name_or_path": "/home/furuiqin/cvp-ws24/Models/deepseek-ai/deepseek-coder-1.3b-instruct_raw",
   "architectures": [
     "LlamaForCausalLM"
   ],
@@ -18,6 +18,7 @@
   "num_attention_heads": 16,
   "num_hidden_layers": 24,
   "num_key_value_heads": 16,
   "rms_norm_eps": 1e-06,
   "rope_scaling": {
     "factor": 4.0,
@@ -26,8 +27,8 @@
   },
   "rope_theta": 100000,
   "tie_word_embeddings": false,
-  "torch_dtype": "float32",
-  "transformers_version": "4.49.0",
   "use_cache": true,
   "vocab_size": 32256
-}

 {
+  "_name_or_path": "/home/dm/prj/py/nn-gpt/out/upload/Models/ABrain/NNGPT-DeepSeek-Coder-1.3B-Instruct",
   "architectures": [
     "LlamaForCausalLM"
   ],
   "num_attention_heads": 16,
   "num_hidden_layers": 24,
   "num_key_value_heads": 16,
+  "pretraining_tp": 1,
   "rms_norm_eps": 1e-06,
   "rope_scaling": {
     "factor": 4.0,
   },
   "rope_theta": 100000,
   "tie_word_embeddings": false,
+  "torch_dtype": "float16",
+  "transformers_version": "4.48.3",
   "use_cache": true,
   "vocab_size": 32256
+}

generation_config.json CHANGED Viewed

@@ -2,5 +2,5 @@
   "_from_model_config": true,
   "bos_token_id": 32013,
   "eos_token_id": 32021,
-  "transformers_version": "4.49.0"
 }

   "_from_model_config": true,
   "bos_token_id": 32013,
   "eos_token_id": 32021,
+  "transformers_version": "4.48.3"
 }

model-00001-of-00002.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:42c2c7ee11018f56d9104464bda19cf1807f85963994a61b3178d7984da3647c
-size 4986380064

 version https://git-lfs.github.com/spec/v1
+oid sha256:a6fa8663157f2b1ee79d943f48e39ae1cc00f3b36af4914fa5c067a57333778c
+size 4989350312

model-00002-of-00002.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:4cfb3170358c381a6807b09fc00a0ff281ce9e162f2eb1381bccd7690d2c142a
-size 399532808

 version https://git-lfs.github.com/spec/v1
+oid sha256:cc6b0314faf3b20778b12edad6acc738d5ef8ab351204c60cd061ee807110009
+size 132120704

model.safetensors.index.json CHANGED Viewed

@@ -1,6 +1,6 @@
 {
   "metadata": {
-    "total_size": 5385887744
   },
   "weight_map": {
     "lm_head.weight": "model-00002-of-00002.safetensors",
@@ -149,11 +149,11 @@
     "model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
-    "model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
-    "model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
-    "model.layers.23.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
-    "model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
-    "model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
     "model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
@@ -221,6 +221,6 @@
     "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
-    "model.norm.weight": "model-00002-of-00002.safetensors"
   }
 }

 {
   "metadata": {
+    "total_size": 5121445888
   },
   "weight_map": {
     "lm_head.weight": "model-00002-of-00002.safetensors",
     "model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
+    "model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors",
+    "model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
+    "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
+    "model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
+    "model.layers.23.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
     "model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
     "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
+    "model.norm.weight": "model-00001-of-00002.safetensors"
   }
 }

tokenizer_config.json CHANGED Viewed

@@ -189,7 +189,7 @@
   "model_max_length": 16384,
   "pad_token": "<|EOT|>",
   "sp_model_kwargs": {},
-  "tokenizer_class": "LlamaTokenizerFast",
   "unk_token": null,
   "use_default_system_prompt": false
 }

   "model_max_length": 16384,
   "pad_token": "<|EOT|>",
   "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
   "unk_token": null,
   "use_default_system_prompt": false
 }