Kingaimaster commited on 27 days ago

Commit

8e995e0

verified ·

1 Parent(s): 6ccacbf

Upload folder using huggingface_hub

Browse files

Files changed (26) hide show

.gitattributes +1 -0
chat_template.jinja +99 -0
config.json +78 -0
generation_config.json +1 -0
model-00001-of-00019.safetensors +3 -0
model-00002-of-00019.safetensors +3 -0
model-00003-of-00019.safetensors +3 -0
model-00004-of-00019.safetensors +3 -0
model-00005-of-00019.safetensors +3 -0
model-00006-of-00019.safetensors +3 -0
model-00007-of-00019.safetensors +3 -0
model-00008-of-00019.safetensors +3 -0
model-00009-of-00019.safetensors +3 -0
model-00010-of-00019.safetensors +3 -0
model-00011-of-00019.safetensors +3 -0
model-00012-of-00019.safetensors +3 -0
model-00013-of-00019.safetensors +3 -0
model-00014-of-00019.safetensors +3 -0
model-00015-of-00019.safetensors +3 -0
model-00016-of-00019.safetensors +3 -0
model-00017-of-00019.safetensors +3 -0
model-00018-of-00019.safetensors +3 -0
model-00019-of-00019.safetensors +3 -0
model.safetensors.index.json +3 -0
tokenizer.json +0 -0
tokenizer_config.json +31 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+model.safetensors.index.json filter=lfs diff=lfs merge=lfs -text

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,99 @@

+{%- set think_start =  '<think>' -%}
+{%- set think_end =  '</think>' -%}
+{%- macro render_content(msg) -%}
+    {%- set c = msg.get('content') -%}
+    {%- if c is string -%}
+      {{ c | replace(think_start, '*') | replace(think_end, '*') }}
+    {%- endif -%}
+{%- endmacro -%}
+{% macro set_roles(message) -%}
+  {%- set role_name =  message.get('name') or  message['role'] -%}
+  {%- if message['role'] == 'user' -%}
+    <|im_user|>{{role_name}}<|im_middle|>
+  {%- elif message['role'] == 'assistant' -%}
+    <|im_assistant|>{{role_name}}<|im_middle|>
+  {%- else -%}
+    <|im_system|>{{role_name}}<|im_middle|>
+  {%- endif -%}
+{%- endmacro -%}
+{%- macro render_toolcalls(message) -%}
+  <|tool_calls_section_begin|>
+  {%- for tool_call in message['tool_calls'] -%}
+    {%- set formatted_id = tool_call['id'] -%}
+    <|tool_call_begin|>{{ formatted_id }}<|tool_call_argument_begin|>{% if tool_call['function']['arguments'] is string %}{{ tool_call['function']['arguments'] }}{% else %}{{ tool_call['function']['arguments'] | tojson }}{% endif %}<|tool_call_end|>
+  {%- endfor -%}
+  <|tool_calls_section_end|>
+{%- endmacro -%}
+{# Find last non-tool-call assisitant message #}
+{%- set ns = namespace(last_non_tool_call_assistant_msg=-1) -%}
+{%- for idx in range(messages|length-1, -1, -1) -%}
+    {%- if messages[idx]['role'] == 'assistant' and not messages[idx].get('tool_calls') -%}
+        {%- set ns.last_non_tool_call_assistant_msg = idx -%}
+        {%- break -%}
+    {%- endif -%}
+{%- endfor -%}
+{# split all messages into history & suffix, reasoning_content in suffix should be reserved.#}
+{%- set hist_msgs = messages[:ns.last_non_tool_call_assistant_msg+1] -%}
+{%- set suffix_msgs = messages[ns.last_non_tool_call_assistant_msg+1:] -%}
+{%- if tools -%}
+  {%- if tools_ts_str -%}
+    <|im_system|>tool_declare<|im_middle|>{{ tools_ts_str }}<|im_end|>
+  {%- else -%}
+    <|im_system|>tool_declare<|im_middle|>{{ tools | tojson(separators=(',', ':')) }}<|im_end|>
+  {%- endif -%}
+{%- endif -%}
+{%- for message in hist_msgs -%}
+  {{set_roles(message)}}
+  {%- if message['role'] == 'assistant' -%}
+    <think></think>{{render_content(message)}}
+    {%- if message.get('tool_calls') -%}
+      {{render_toolcalls(message)}}
+    {%- endif -%}
+  {%- elif message['role'] == 'tool' -%}
+    {%- set tool_call_id = message.tool_call_id -%}
+    ## Return of {{ tool_call_id }}
+{{render_content(message)}}
+  {%- elif message['content'] is not none -%}
+    {{render_content(message)}}
+  {%- endif -%}
+  <|im_end|>
+{%- endfor -%}
+{%- for message in suffix_msgs -%}
+  {{set_roles(message)}}
+  {%- if message['role'] == 'assistant' -%}
+    {%- if thinking is defined and thinking is false -%}
+    <think></think>{{render_content(message)}}
+    {%- else -%}
+    {%- set rc = message.get('reasoning_content', '') -%}
+    <think>{{rc}}</think>{{render_content(message)}}
+    {%- endif -%}
+    {%- if message.get('tool_calls') -%}
+     {{render_toolcalls(message)}}
+    {%- endif -%}
+  {%- elif message['role'] == 'tool' -%}
+    {%- set tool_call_id = message.tool_call_id -%}
+    ## Return of {{ tool_call_id }}
+{{render_content(message)}}
+  {%- elif message['content'] is not none -%}
+    {{render_content(message)}}
+  {%- endif -%}
+  <|im_end|>
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+  <|im_assistant|>assistant<|im_middle|>
+  {%- if thinking is defined and thinking is false -%}
+  <think></think>
+  {%- else -%}
+  <think>
+  {%- endif -%}
+{%- endif -%}

config.json ADDED Viewed

	@@ -0,0 +1,78 @@

+{
+  "architectures": ["DeepseekV3ForCausalLM"],
+  "bos_token_id": 163584,
+  "dtype": "bfloat16",
+  "eos_token_id": 163585,
+  "hidden_act": "silu",
+  "hidden_size": 7168,
+  "initializer_range": 0.02,
+  "intermediate_size": 18432,
+  "max_position_embeddings": 262144,
+  "model_type": "deepseek_v3",
+  "moe_intermediate_size": 2048,
+  "num_attention_heads": 64,
+  "num_hidden_layers": 61,
+  "num_key_value_heads": 64,
+  "pad_token_id": 163839,
+  "rms_norm_eps": 1e-5,
+  "rope_theta": 50000,
+  "rope_scaling": {
+    "beta_fast": 32,
+    "beta_slow": 1,
+    "factor": 64,
+    "mscale": 1,
+    "mscale_all_dim": 1,
+    "original_max_position_embeddings": 4096,
+    "type": "yarn"
+  },
+  "routed_scaling_factor": 2.827,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 163840,
+  "aux_loss_alpha": 0.001,
+  "attention_bias": false,
+  "attention_dropout": 0,
+  "ep_size": 1,
+  "first_k_dense_replace": 1,
+  "kv_lora_rank": 512,
+  "n_group": 1,
+  "n_routed_experts": 384,
+  "n_shared_experts": 1,
+  "norm_topk_prob": true,
+  "num_experts_per_tok": 8,
+  "moe_layer_freq": 1,
+  "num_nextn_predict_layers": 0,
+  "q_lora_rank": 1536,
+  "qk_nope_head_dim": 128,
+  "qk_rope_head_dim": 64,
+  "scoring_func": "sigmoid",
+  "seq_aux": true,
+  "topk_group": 1,
+  "topk_method": "noaux_tc",
+  "v_head_dim": 128,
+  "quantization_config": {
+    "config_groups": {
+      "group_0": {
+        "input_activations": null,
+        "output_activations": null,
+        "targets": ["Linear"],
+        "weights": {
+          "actorder": null,
+          "block_structure": null,
+          "dynamic": false,
+          "group_size": 32,
+          "num_bits": 4,
+          "observer": "minmax",
+          "observer_kwargs": {},
+          "strategy": "group",
+          "symmetric": true,
+          "type": "int"
+        }
+      }
+    },
+    "format": "pack-quantized",
+    "ignore": ["lm_head", "re:.*self_attn.*", "re:.*shared_experts.*", "re:.*mlp\\.(gate|up|gate_up|down)_proj.*"],
+    "quant_method": "compressed-tensors",
+    "quantization_status": "compressed"
+  }
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"max_length": 262144, "eos_token_id": 163586}

model-00001-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:997e10f3d8d9774e6d096a4d348c95d363757e9e176c2bb0fac700e747078cc9
+size 32207566064

model-00002-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:18620ed9781ce126fc865431760d1887874cfdc0916e69588d9da88cbdc74e04
+size 32209010008

model-00003-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c20f5675d602381daa651ecbb9d14ccdbf330d5921dee82eae8c9416ba2533f8
+size 32209011024

model-00004-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8c285aa6b6ca2e20e62d46e3ab3cd6bfd151ba7525f1454c1b83984b66db7dd4
+size 32207597368

model-00005-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4d2b68038154f6841d4a9e3fd9101a29906b12e9c9e155e552c731a9e860966a
+size 32209019784

model-00006-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:dd1b5e83ea93edc9969140613c735d480debe17f2b28331a14aeb30a2312e77e
+size 32209021432

model-00007-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a89e53f11762b0da426d5e31160d28e91a842bbe45840caee43c614374632dc6
+size 32209019520

model-00008-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:26c3dc9ea05188457597456b3045fb9cc21e65b27390a6d9b5b821bb7060bf2f
+size 32207597648

model-00009-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ab821190ee99e130dab829a540249be9919d25691b9ff1f35a846d04409e0d80
+size 32209021432

model-00010-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:28c72a4c2c8d6bb74438185a9ce5d2a8a75db1d43d302bdb8f026454325d647b
+size 32208102144

model-00011-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c080b5575edeb4087642a6abff4c38698f05fe437ef72a59e9624158e48617d6
+size 32209288472

model-00012-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e7ac7c466570827801f6dc367124a8e4878c1f2523986da48e92fba37fafdd1d
+size 32212833712

model-00013-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:874da2c6179347bb8ea5c662eebeda893a58e49a538a6ee196d364ed23ebcda6
+size 32209021424

model-00014-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:74283d767f168be0db8ce79a896bc4e617332aa41b5c0b186dab9d3c7bb5f6ca
+size 32209019568

model-00015-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0918c6e70159234183eb9e4739cbbc182edd3d2b988c2a2e2934d6b7a11e22ba
+size 32207597592

model-00016-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b150d7b06817ca7f5f1f7f6f5b3e3613ab7b1c532bad7ebff28c7c913f5448bb
+size 32209019784

model-00017-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:82743be5ef520e412f54fdbd1ca593685fdacd705c6427dc5e03c3f253017a4f
+size 32209021432

model-00018-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6e4868844626652f8a8953d2737ca282ac7fc956fc0e65c181869819c7a4a0e1
+size 32208099920

model-00019-of-00019.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7a902ebfd7cceebc8d051d7926534de7b53b625c6bc79a1044603230f805f523
+size 14474008600

model.safetensors.index.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1341ea6020a68cdb9dd8c9f551ebcca6ff9121253ee1572755b03d8e321eaf94
+size 20237088

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "added_tokens_decoder": {
+    "163584": {"content": "[BOS]", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163585": {"content": "[EOS]", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163586": {"content": "<|im_end|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163587": {"content": "<|im_user|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163588": {"content": "<|im_assistant|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163590": {"content": "<|start_header_id|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163591": {"content": "<|end_header_id|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163593": {"content": "[EOT]", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163594": {"content": "<|im_system|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163595": {"content": "<|tool_calls_section_begin|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163596": {"content": "<|tool_calls_section_end|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163597": {"content": "<|tool_call_begin|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163598": {"content": "<|tool_call_argument_begin|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163599": {"content": "<|tool_call_end|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163601": {"content": "<|im_middle|>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163606": {"content": "<think>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163607": {"content": "</think>", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": false},
+    "163838": {"content": "[UNK]", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true},
+    "163839": {"content": "[PAD]", "lstrip": false, "normalized": false, "rstrip": false, "single_word": false, "special": true}
+  },
+  "additional_special_tokens": ["<|im_end|>", "<|im_user|>", "<|im_assistant|>", "<|start_header_id|>", "<|end_header_id|>", "[EOT]", "<|im_system|>", "<|im_middle|>"],
+  "bos_token": "[BOS]",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "[EOS]",
+  "model_max_length": 262144,
+  "pad_token": "[PAD]",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "[UNK]"
+}