nevikw39 commited on
Commit
ba250cb
·
verified ·
1 Parent(s): 56643c9

Upload openvino_detokenizer.xml with huggingface_hub

Browse files
Files changed (1) hide show
  1. openvino_detokenizer.xml +221 -0
openvino_detokenizer.xml ADDED
@@ -0,0 +1,221 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0"?>
2
+ <net name="detokenizer" version="11">
3
+ <layers>
4
+ <layer id="0" name="Parameter_140944" type="Parameter" version="opset1">
5
+ <data shape="?,?" element_type="i64" />
6
+ <output>
7
+ <port id="0" precision="I64" names="Parameter_140944">
8
+ <dim>-1</dim>
9
+ <dim>-1</dim>
10
+ </port>
11
+ </output>
12
+ </layer>
13
+ <layer id="1" name="Convert_141114" type="Convert" version="opset1">
14
+ <data destination_type="i32" />
15
+ <input>
16
+ <port id="0" precision="I64">
17
+ <dim>-1</dim>
18
+ <dim>-1</dim>
19
+ </port>
20
+ </input>
21
+ <output>
22
+ <port id="1" precision="I32">
23
+ <dim>-1</dim>
24
+ <dim>-1</dim>
25
+ </port>
26
+ </output>
27
+ </layer>
28
+ <layer id="2" name="Constant_140946" type="Const" version="opset1">
29
+ <data element_type="i32" shape="151665" offset="0" size="606660" />
30
+ <output>
31
+ <port id="0" precision="I32">
32
+ <dim>151665</dim>
33
+ </port>
34
+ </output>
35
+ </layer>
36
+ <layer id="3" name="Constant_140948" type="Const" version="opset1">
37
+ <data element_type="i32" shape="151665" offset="606660" size="606660" />
38
+ <output>
39
+ <port id="0" precision="I32">
40
+ <dim>151665</dim>
41
+ </port>
42
+ </output>
43
+ </layer>
44
+ <layer id="4" name="Constant_140950" type="Const" version="opset1">
45
+ <data element_type="u8" shape="976273" offset="1213320" size="976273" />
46
+ <output>
47
+ <port id="0" precision="U8">
48
+ <dim>976273</dim>
49
+ </port>
50
+ </output>
51
+ </layer>
52
+ <layer id="5" name="Slice_140955" type="Const" version="opset1">
53
+ <data element_type="i32" shape="9" offset="2189593" size="36" />
54
+ <output>
55
+ <port id="0" precision="I32">
56
+ <dim>9</dim>
57
+ </port>
58
+ </output>
59
+ </layer>
60
+ <layer id="6" name="VocabDecoder_140957" type="VocabDecoder" version="extension">
61
+ <data skip_tokens="" />
62
+ <input>
63
+ <port id="0" precision="I32">
64
+ <dim>-1</dim>
65
+ <dim>-1</dim>
66
+ </port>
67
+ <port id="1" precision="I32">
68
+ <dim>151665</dim>
69
+ </port>
70
+ <port id="2" precision="I32">
71
+ <dim>151665</dim>
72
+ </port>
73
+ <port id="3" precision="U8">
74
+ <dim>976273</dim>
75
+ </port>
76
+ <port id="4" precision="I32">
77
+ <dim>9</dim>
78
+ </port>
79
+ </input>
80
+ <output>
81
+ <port id="5" precision="I32">
82
+ <dim>-1</dim>
83
+ </port>
84
+ <port id="6" precision="I32">
85
+ <dim>-1</dim>
86
+ </port>
87
+ <port id="7" precision="I32">
88
+ <dim>-1</dim>
89
+ </port>
90
+ <port id="8" precision="I32">
91
+ <dim>-1</dim>
92
+ </port>
93
+ <port id="9" precision="U8">
94
+ <dim>-1</dim>
95
+ </port>
96
+ </output>
97
+ </layer>
98
+ <layer id="7" name="FuzeRagged_140958" type="FuzeRagged" version="extension">
99
+ <input>
100
+ <port id="0" precision="I32">
101
+ <dim>-1</dim>
102
+ </port>
103
+ <port id="1" precision="I32">
104
+ <dim>-1</dim>
105
+ </port>
106
+ <port id="2" precision="I32">
107
+ <dim>-1</dim>
108
+ </port>
109
+ <port id="3" precision="I32">
110
+ <dim>-1</dim>
111
+ </port>
112
+ </input>
113
+ <output>
114
+ <port id="4" precision="I32">
115
+ <dim>-1</dim>
116
+ </port>
117
+ <port id="5" precision="I32">
118
+ <dim>-1</dim>
119
+ </port>
120
+ </output>
121
+ </layer>
122
+ <layer id="8" name="UTF8Validate_140959" type="UTF8Validate" version="extension">
123
+ <data replace_mode="true" />
124
+ <input>
125
+ <port id="0" precision="I32">
126
+ <dim>-1</dim>
127
+ </port>
128
+ <port id="1" precision="I32">
129
+ <dim>-1</dim>
130
+ </port>
131
+ <port id="2" precision="U8">
132
+ <dim>-1</dim>
133
+ </port>
134
+ </input>
135
+ <output>
136
+ <port id="3" precision="I32">
137
+ <dim>-1</dim>
138
+ </port>
139
+ <port id="4" precision="I32">
140
+ <dim>-1</dim>
141
+ </port>
142
+ <port id="5" precision="U8">
143
+ <dim>-1</dim>
144
+ </port>
145
+ </output>
146
+ </layer>
147
+ <layer id="9" name="StringTensorPack_140960" type="StringTensorPack" version="opset15">
148
+ <input>
149
+ <port id="0" precision="I32">
150
+ <dim>-1</dim>
151
+ </port>
152
+ <port id="1" precision="I32">
153
+ <dim>-1</dim>
154
+ </port>
155
+ <port id="2" precision="U8">
156
+ <dim>-1</dim>
157
+ </port>
158
+ </input>
159
+ <output>
160
+ <port id="3" precision="STRING" names="Result_140961,string_output">
161
+ <dim>-1</dim>
162
+ </port>
163
+ </output>
164
+ </layer>
165
+ <layer id="10" name="Result_140961" type="Result" version="opset1" output_names="Result_140961,string_output">
166
+ <input>
167
+ <port id="0" precision="STRING">
168
+ <dim>-1</dim>
169
+ </port>
170
+ </input>
171
+ </layer>
172
+ </layers>
173
+ <edges>
174
+ <edge from-layer="0" from-port="0" to-layer="1" to-port="0" />
175
+ <edge from-layer="1" from-port="1" to-layer="6" to-port="0" />
176
+ <edge from-layer="2" from-port="0" to-layer="6" to-port="1" />
177
+ <edge from-layer="3" from-port="0" to-layer="6" to-port="2" />
178
+ <edge from-layer="4" from-port="0" to-layer="6" to-port="3" />
179
+ <edge from-layer="5" from-port="0" to-layer="6" to-port="4" />
180
+ <edge from-layer="6" from-port="7" to-layer="7" to-port="2" />
181
+ <edge from-layer="6" from-port="9" to-layer="8" to-port="2" />
182
+ <edge from-layer="6" from-port="8" to-layer="7" to-port="3" />
183
+ <edge from-layer="6" from-port="6" to-layer="7" to-port="1" />
184
+ <edge from-layer="6" from-port="5" to-layer="7" to-port="0" />
185
+ <edge from-layer="7" from-port="4" to-layer="8" to-port="0" />
186
+ <edge from-layer="7" from-port="5" to-layer="8" to-port="1" />
187
+ <edge from-layer="8" from-port="3" to-layer="9" to-port="0" />
188
+ <edge from-layer="8" from-port="4" to-layer="9" to-port="1" />
189
+ <edge from-layer="8" from-port="5" to-layer="9" to-port="2" />
190
+ <edge from-layer="9" from-port="3" to-layer="10" to-port="0" />
191
+ </edges>
192
+ <rt_info>
193
+ <add_attention_mask value="True" />
194
+ <add_prefix_space />
195
+ <add_special_tokens value="True" />
196
+ <bos_token_id value="151646" />
197
+ <chat_template value="{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'&lt;|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'&lt;|Assistant|>&lt;|tool▁calls▁begin|>&lt;|tool▁call▁begin|>' + tool['type'] + '&lt;|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '&lt;|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\n' + '&lt;|tool▁call▁begin|>' + tool['type'] + '&lt;|tool▁sep|>' + tool['function']['name'] + '\n' + '```json' + '\n' + tool['function']['arguments'] + '\n' + '```' + '&lt;|tool▁call▁end|>'}}{{'&lt;|tool▁calls▁end|>&lt;|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'&lt;|tool▁outputs▁end|>' + message['content'] + '&lt;|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '&lt;/think>' in content %}{% set content = content.split('&lt;/think>')[-1] %}{% endif %}{{'&lt;|Assistant|>' + content + '&lt;|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'&lt;|tool▁outputs▁begin|>&lt;|tool▁output▁begin|>' + message['content'] + '&lt;|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\n&lt;|tool▁output▁begin|>' + message['content'] + '&lt;|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'&lt;|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'&lt;|Assistant|>&lt;think>\n'}}{% endif %}" />
198
+ <clean_up_tokenization_spaces />
199
+ <detokenizer_input_type value="i64" />
200
+ <eos_token_id value="151643" />
201
+ <handle_special_tokens_with_re />
202
+ <max_length />
203
+ <number_of_inputs value="1" />
204
+ <openvino_tokenizers_version value="2025.1.0.0-523-710ddf14de8" />
205
+ <openvino_version value="2025.1.0-18503-6fec06580ab-releases/2025/1" />
206
+ <original_post_processor_template value="{&quot;type&quot;: &quot;TemplateProcessing&quot;, &quot;single&quot;: [{&quot;SpecialToken&quot;: {&quot;id&quot;: &quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;, &quot;type_id&quot;: 0}}, {&quot;Sequence&quot;: {&quot;id&quot;: &quot;A&quot;, &quot;type_id&quot;: 0}}], &quot;pair&quot;: [{&quot;SpecialToken&quot;: {&quot;id&quot;: &quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;, &quot;type_id&quot;: 0}}, {&quot;Sequence&quot;: {&quot;id&quot;: &quot;A&quot;, &quot;type_id&quot;: 0}}, {&quot;SpecialToken&quot;: {&quot;id&quot;: &quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;, &quot;type_id&quot;: 1}}, {&quot;Sequence&quot;: {&quot;id&quot;: &quot;B&quot;, &quot;type_id&quot;: 1}}], &quot;special_tokens&quot;: {&quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;: {&quot;id&quot;: &quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;, &quot;ids&quot;: [151646], &quot;tokens&quot;: [&quot;&lt;\uff5cbegin\u2581of\u2581sentence\uff5c>&quot;]}}}" />
207
+ <original_tokenizer_class value="&lt;class 'transformers.models.llama.tokenization_llama_fast.LlamaTokenizerFast'>" />
208
+ <pad_token_id value="151643" />
209
+ <processed_post_processor_template value="{&quot;single&quot;: {&quot;ids&quot;: [151646, -1], &quot;type_ids&quot;: [0, 0]}, &quot;pair&quot;: {&quot;ids&quot;: [151646, -1, 151646, -2], &quot;type_ids&quot;: [0, 0, 1, 1]}}" />
210
+ <sentencepiece_version value="0.2.0" />
211
+ <skip_special_tokens value="True" />
212
+ <streaming_detokenizer value="False" />
213
+ <tokenizer_output_type value="i64" />
214
+ <tokenizers_version value="0.21.1" />
215
+ <transformers_version value="4.48.3" />
216
+ <use_max_padding value="False" />
217
+ <use_sentencepiece_backend value="False" />
218
+ <utf8_replace_mode value="replace" />
219
+ <with_detokenizer value="True" />
220
+ </rt_info>
221
+ </net>