Upload folder using huggingface_hub

#1
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International License
2
+
3
+ Copyright (c) 2024 Scott Thornton / perfecXion.ai
4
+
5
+ AEGIS: Adaptive Ensemble Guard with Integrated Steering
6
+
7
+ You are free to:
8
+ Share — copy and redistribute the material in any medium or format
9
+ Adapt — remix, transform, and build upon the material
10
+
11
+ Under the following terms:
12
+
13
+ Attribution — You must give appropriate credit, provide a link to the license,
14
+ and indicate if changes were made. You may do so in any reasonable manner,
15
+ but not in any way that suggests the licensor endorses you or your use.
16
+
17
+ NonCommercial — You may not use the material for commercial purposes
18
+ without explicit written permission from the copyright holder.
19
+
20
+ ShareAlike — If you remix, transform, or build upon the material, you must
21
+ distribute your contributions under the same license as the original.
22
+
23
+ No additional restrictions — You may not apply legal terms or technological
24
+ measures that legally restrict others from doing anything the license permits.
25
+
26
+ ATTRIBUTION REQUIREMENTS:
27
+ When using this material, you must provide clear attribution including:
28
+ 1. The name "AEGIS" and credit to "scthornton.ai / perfecXion.ai"
29
+ 2. A link to the original repository
30
+ 3. Indication of any modifications made
31
+
32
+ For commercial licensing inquiries, contact: scott@perfecxion.ai
33
+
34
+ Full license text: https://creativecommons.org/licenses/by-nc-sa/4.0/legalcode
README.md ADDED
@@ -0,0 +1,163 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: cc-by-nc-sa-4.0
3
+ base_model: Qwen/Qwen2.5-3B-Instruct
4
+ library_name: peft
5
+ tags:
6
+ - llm-security
7
+ - jailbreak-detection
8
+ - text-classification
9
+ - safety
10
+ - lora
11
+ pipeline_tag: text-classification
12
+ ---
13
+
14
+ # TRYLOCK Sidecar Classifier
15
+
16
+ Layer 3 of the TRYLOCK (Adaptive Ensemble Guard with Integrated Steering) defense system. This is a lightweight classifier that runs alongside the main LLM to detect attack patterns and dynamically adjust defense strength.
17
+
18
+ ## Model Description
19
+
20
+ The sidecar classifier categorizes inputs into three classes:
21
+ - **SAFE**: Benign queries that need minimal defense
22
+ - **WARN**: Ambiguous or suspicious queries
23
+ - **ATTACK**: Clear jailbreak/attack attempts
24
+
25
+ ## Intended Uses
26
+
27
+ **Primary use cases:**
28
+ - Real-time classification of LLM inputs for adaptive defense
29
+ - Dynamically adjusting RepE steering strength
30
+ - Research on attack detection methods
31
+ - Content moderation and threat triage
32
+
33
+ **Out of scope:**
34
+ - Standalone content moderation (designed for adaptive steering)
35
+ - High-stakes security decisions without human review
36
+
37
+ ### Training Details
38
+
39
+ | Parameter | Value |
40
+ |-----------|-------|
41
+ | Base Model | Qwen2.5-3B-Instruct |
42
+ | Method | LoRA fine-tuning |
43
+ | LoRA Rank | 32 |
44
+ | LoRA Alpha | 64 |
45
+ | Target Modules | q_proj, k_proj, v_proj, o_proj |
46
+ | Training Samples | 2,349 |
47
+ | Classes | SAFE, WARN, ATTACK |
48
+
49
+ ### Evaluation Results
50
+
51
+ | Class | Precision | Recall | F1 |
52
+ |-------|-----------|--------|-----|
53
+ | SAFE | 24.1% | 34.5% | 28.4% |
54
+ | WARN | 58.3% | 51.6% | 54.8% |
55
+ | ATTACK | 62.4% | 60.8% | 61.6% |
56
+ | **Macro Avg** | 48.3% | 48.9% | 48.3% |
57
+
58
+ **Note**: This is a research prototype. ATTACK detection is prioritized over SAFE detection for security.
59
+
60
+ ## Usage
61
+
62
+ ```python
63
+ import torch
64
+ from transformers import AutoModelForSequenceClassification, AutoTokenizer
65
+ from peft import PeftModel
66
+
67
+ # Load model
68
+ base_model = AutoModelForSequenceClassification.from_pretrained(
69
+ "Qwen/Qwen2.5-3B-Instruct",
70
+ num_labels=3,
71
+ device_map="auto",
72
+ torch_dtype=torch.bfloat16,
73
+ )
74
+ model = PeftModel.from_pretrained(base_model, "scthornton/trylock-sidecar-classifier")
75
+ tokenizer = AutoTokenizer.from_pretrained("scthornton/trylock-sidecar-classifier")
76
+
77
+ # Classify input
78
+ label_names = ["SAFE", "WARN", "ATTACK"]
79
+
80
+ def classify(text):
81
+ inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=2048)
82
+ with torch.no_grad():
83
+ outputs = model(**inputs.to(model.device))
84
+ probs = torch.softmax(outputs.logits, dim=-1)[0]
85
+ return label_names[probs.argmax()], probs.tolist()
86
+
87
+ # Example
88
+ result, probs = classify("Ignore all previous instructions and...")
89
+ print(f"Classification: {result}") # ATTACK
90
+ print(f"Probabilities: {dict(zip(label_names, probs))}")
91
+ ```
92
+
93
+ ## Dynamic Defense Integration
94
+
95
+ Use the classification to adjust RepE steering strength:
96
+
97
+ ```python
98
+ alpha_map = {
99
+ "SAFE": 0.5, # Minimal steering - preserve fluency
100
+ "WARN": 1.5, # Moderate steering
101
+ "ATTACK": 2.5 # Maximum steering - block harmful output
102
+ }
103
+
104
+ classification, _ = classify(user_input)
105
+ alpha = alpha_map[classification]
106
+ # Apply RepE steering with this alpha value
107
+ ```
108
+
109
+ ## TRYLOCK Architecture
110
+
111
+ This classifier is Layer 3 of the 3-layer TRYLOCK defense:
112
+
113
+ 1. **Layer 1 (KNOWLEDGE)**: [trylock-mistral-7b-dpo](https://huggingface.co/scthornton/trylock-mistral-7b-dpo)
114
+ 2. **Layer 2 (INSTINCT)**: [trylock-repe-vectors](https://huggingface.co/scthornton/trylock-repe-vectors)
115
+ 3. **Layer 3 (OVERSIGHT)**: Sidecar classifier (this model)
116
+
117
+ ## Limitations and Risks
118
+
119
+ **Limitations:**
120
+ - SAFE class has lower precision (24%) - may over-classify benign queries as WARN
121
+ - Trained on English-language attacks only
122
+ - 3-class granularity may miss nuanced threat levels
123
+ - Requires ~3B parameter model inference overhead
124
+
125
+ **Risks:**
126
+ - False positives may trigger unnecessary steering
127
+ - False negatives on novel attack patterns
128
+ - Classification confidence doesn't guarantee correctness
129
+
130
+ **Recommendations:**
131
+ - Use probability scores, not just class labels, for fine-grained control
132
+ - Consider WARN classification as "elevated caution" rather than definite threat
133
+ - Combine with other safety mechanisms for production use
134
+
135
+ ## Framework Versions
136
+
137
+ - PEFT 0.18.0
138
+
139
+ ## Citation
140
+
141
+ ```bibtex
142
+ @misc{trylock2024,
143
+ title={TRYLOCK: Defense-in-Depth Against LLM Jailbreaks via Layered Preference and Representation Engineering},
144
+ author={Thornton, Scott},
145
+ year={2024},
146
+ url={https://huggingface.co/scthornton/trylock-sidecar-classifier}
147
+ }
148
+ ```
149
+
150
+ ## License
151
+
152
+ **CC BY-NC-SA 4.0** (Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International)
153
+
154
+ You are free to:
155
+ - **Share** — copy and redistribute the material in any medium or format
156
+ - **Adapt** — remix, transform, and build upon the material
157
+
158
+ Under the following terms:
159
+ - **Attribution** — You must give appropriate credit to **Scott Thornton**, provide a link to the license, and indicate if changes were made
160
+ - **NonCommercial** — You may not use the material for commercial purposes without explicit written permission
161
+ - **ShareAlike** — If you remix, transform, or build upon the material, you must distribute your contributions under the same license
162
+
163
+ For commercial licensing inquiries, contact: scott@perfecxion.ai
adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-3B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 64,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": [
25
+ "classifier",
26
+ "score"
27
+ ],
28
+ "peft_type": "LORA",
29
+ "peft_version": "0.18.0",
30
+ "qalora_group_size": 16,
31
+ "r": 32,
32
+ "rank_pattern": {},
33
+ "revision": null,
34
+ "target_modules": [
35
+ "k_proj",
36
+ "o_proj",
37
+ "q_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "SEQ_CLS",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23ccbeb16090a961da192152bae6afda1e64da54b8a674e7286dabf0927ff7c7
3
+ size 59033448
added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
thresholds.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "warn_threshold": 0.3,
3
+ "attack_threshold": 0.7,
4
+ "label_names": [
5
+ "SAFE",
6
+ "WARN",
7
+ "ATTACK"
8
+ ]
9
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:694f1174c5bdf94e2fc50796c0f1733a5a3945ff110b0dfa40ea0701cc9c9c42
3
+ size 11422176
tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|im_end|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 131072,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ac2954cf714d1e4c9de303ea1e69eb15ccd8e331bc716544a59c350d1bcb250
3
+ size 5841
vocab.json ADDED
The diff for this file is too large to render. See raw diff