ryugyosoft commited on
Commit
0c1184e
·
verified ·
1 Parent(s): 125bb0d

Add files using upload-large-folder tool

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model: google/gemma-4-E4B-it
4
+ pipeline_tag: image-text-to-text
5
+ library_name: openvino
6
+ tags:
7
+ - openvino
8
+ - npu
9
+ - intel
10
+ - intel-npu
11
+ - gemma4
12
+ - onw
13
+ language:
14
+ - en
15
+ - ja
16
+ ---
17
+
18
+ # Gemma 4 E4B-it for onw — every component on the Intel NPU
19
+
20
+ [google/gemma-4-E4B-it](https://huggingface.co/google/gemma-4-E4B-it) (text + image) packaged for
21
+ **[onw](https://huggingface.co/ryugyosoft/onw)** (俺のNPUがこんなに動くわけない), the engine that runs LLMs entirely
22
+ on the Intel NPU. Same NPU graphs as the standalone
23
+ [ryugyosoft/gemma-4-E4B-it-npu](https://huggingface.co/ryugyosoft/gemma-4-E4B-it-npu), run by the shared engine.
24
+
25
+ ```bash
26
+ hf download ryugyosoft/onw --local-dir onw && cd onw
27
+ start.bat ryugyosoft/gemma-4-E4B-it-onw # Windows
28
+ bash start.sh ryugyosoft/gemma-4-E4B-it-onw # Ubuntu
29
+ ```
30
+
31
+ | NPU 3720 (Core Ultra 9 285HX) | |
32
+ |---|---|
33
+ | decode | **6.7-6.8 tok/s** (prompt lookup decoding on, exact) |
34
+ | image + question (284 tokens) | vision 1.9 s + prefill 2.0 s (64-token blocks, >= 24 GB RAM) |
35
+ | follow-up turn | only new tokens are processed (turn 2 of an image chat: 1.7 s vs 4.8 s) |
36
+ | memory | ~8 GB with the 64-token block, ~6 GB without |
37
+
38
+ What changed vs the standalone repo: the token embedding and the 2.7 GB per-layer embedding table are looked up
39
+ on the host (they are table reads; no NPU worker process any more), the LM head is one shared INT8 input for all
40
+ block sizes, and prompt lookup decoding / prefix reuse come from the engine.
41
+
42
+ **Download with `hf download` or let `onw` fetch it** — a `git clone` without Git LFS gets pointer files, which
43
+ `onw` detects and reports.
44
+
45
+ ## Files
46
+
47
+ | file | what |
48
+ |---|---|
49
+ | `seg00_S{1,16,64}.xml` | the static Gemma 4 decoder (head_dim-512 attention split into 256-wide heads, host-built masks) for 1 / 16 / 64-token blocks |
50
+ | `seg01_S*.xml` | LM head + logit softcapping (shared INT8 weight) |
51
+ | `vision.xml` | static vision encoder (280 soft tokens) |
52
+ | `shared.bin` | INT8 token embedding (= tied head), per-layer embedding table + id map |
53
+ | `engine.json`, tokenizer / processor files | metadata for onw |
54
+
55
+ ## License
56
+
57
+ Apache 2.0, same as the base model. The weights are re-quantized / restructured from it.
chat_template.jinja ADDED
@@ -0,0 +1,386 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {#
2
+ Template: Google Gemma 4 Canonical Chat Template
3
+ Author: Google Gemma Engineering Team
4
+ Published: 2026-07-09
5
+ Context: Fixed tool-calling loops, turn closures, and thinking content-ordering.
6
+ #}
7
+ {%- macro format_parameters(properties, required, filter_keys=false) -%}
8
+ {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
9
+ {%- set ns = namespace(found_first=false) -%}
10
+ {%- for key, value in properties | dictsort -%}
11
+ {%- set add_comma = false -%}
12
+ {%- if not filter_keys or key not in standard_keys -%}
13
+ {%- if ns.found_first %},{% endif -%}
14
+ {%- set ns.found_first = true -%}
15
+ {{ key }}:{
16
+ {%- if value['description'] -%}
17
+ description:<|"|>{{ value['description'] }}<|"|>
18
+ {%- set add_comma = true -%}
19
+ {%- endif -%}
20
+ {%- if value['type'] | upper == 'STRING' -%}
21
+ {%- if value['enum'] -%}
22
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
23
+ enum:{{ format_argument(value['enum']) }}
24
+ {%- endif -%}
25
+ {%- elif value['type'] | upper == 'ARRAY' -%}
26
+ {%- if value['items'] is mapping and value['items'] -%}
27
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
28
+ items:{
29
+ {%- set ns_items = namespace(found_first=false) -%}
30
+ {%- for item_key, item_value in value['items'] | dictsort -%}
31
+ {%- if item_value is not none -%}
32
+ {%- if ns_items.found_first %},{% endif -%}
33
+ {%- set ns_items.found_first = true -%}
34
+ {%- if item_key == 'properties' -%}
35
+ properties:{
36
+ {%- if item_value is mapping -%}
37
+ {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
38
+ {%- endif -%}
39
+ }
40
+ {%- elif item_key == 'required' -%}
41
+ required:[
42
+ {%- for req_item in item_value -%}
43
+ <|"|>{{- req_item -}}<|"|>
44
+ {%- if not loop.last %},{% endif -%}
45
+ {%- endfor -%}
46
+ ]
47
+ {%- elif item_key == 'type' -%}
48
+ {%- if item_value is string -%}
49
+ type:{{ format_argument(item_value | upper) }}
50
+ {%- else -%}
51
+ type:{{ format_argument(item_value | map('upper') | list) }}
52
+ {%- endif -%}
53
+ {%- else -%}
54
+ {{ item_key }}:{{ format_argument(item_value) }}
55
+ {%- endif -%}
56
+ {%- endif -%}
57
+ {%- endfor -%}
58
+ }
59
+ {%- endif -%}
60
+ {%- endif -%}
61
+ {%- if value['nullable'] %}
62
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
63
+ nullable:true
64
+ {%- endif -%}
65
+ {%- if value['type'] | upper == 'OBJECT' -%}
66
+ {%- if value['properties'] is defined and value['properties'] is mapping -%}
67
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
68
+ properties:{
69
+ {{- format_parameters(value['properties'], value['required'] | default([])) -}}
70
+ }
71
+ {%- elif value is mapping -%}
72
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
73
+ properties:{
74
+ {{- format_parameters(value, value['required'] | default([]), filter_keys=true) -}}
75
+ }
76
+ {%- endif -%}
77
+ {%- if value['required'] -%}
78
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
79
+ required:[
80
+ {%- for item in value['required'] | default([]) -%}
81
+ <|"|>{{- item -}}<|"|>
82
+ {%- if not loop.last %},{% endif -%}
83
+ {%- endfor -%}
84
+ ]
85
+ {%- endif -%}
86
+ {%- endif -%}
87
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
88
+ type:<|"|>{{ value['type'] | upper }}<|"|>}
89
+ {%- endif -%}
90
+ {%- endfor -%}
91
+ {%- endmacro -%}
92
+ {%- macro format_function_declaration(tool_data) -%}
93
+ declaration:{{- tool_data['function']['name'] -}}{description:<|"|>{{- tool_data['function']['description'] -}}<|"|>
94
+ {%- set params = tool_data['function']['parameters'] -%}
95
+ {%- if params -%}
96
+ ,parameters:{
97
+ {%- if params['properties'] -%}
98
+ properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
99
+ {%- endif -%}
100
+ {%- if params['required'] -%}
101
+ required:[
102
+ {%- for item in params['required'] -%}
103
+ <|"|>{{- item -}}<|"|>
104
+ {{- ',' if not loop.last -}}
105
+ {%- endfor -%}
106
+ ],
107
+ {%- endif -%}
108
+ {%- if params['type'] -%}
109
+ type:<|"|>{{- params['type'] | upper -}}<|"|>}
110
+ {%- endif -%}
111
+ {%- endif -%}
112
+ {%- if 'response' in tool_data['function'] -%}
113
+ {%- set response_declaration = tool_data['function']['response'] -%}
114
+ ,response:{
115
+ {%- if response_declaration['description'] -%}
116
+ description:<|"|>{{- response_declaration['description'] -}}<|"|>,
117
+ {%- endif -%}
118
+ {%- if response_declaration['type'] | upper == 'OBJECT' -%}
119
+ type:<|"|>{{- response_declaration['type'] | upper -}}<|"|>}
120
+ {%- endif -%}
121
+ {%- endif -%}
122
+ }
123
+ {%- endmacro -%}
124
+ {%- macro format_argument(argument, escape_keys=True) -%}
125
+ {%- if argument is none -%}
126
+ {{- 'null' -}}
127
+ {%- elif argument is string -%}
128
+ {{- '<|"|>' + argument + '<|"|>' -}}
129
+ {%- elif argument is boolean -%}
130
+ {{- 'true' if argument else 'false' -}}
131
+ {%- elif argument is mapping -%}
132
+ {{- '{' -}}
133
+ {%- set ns = namespace(found_first=false) -%}
134
+ {%- for key, value in argument | dictsort -%}
135
+ {%- if ns.found_first %},{% endif -%}
136
+ {%- set ns.found_first = true -%}
137
+ {%- if escape_keys -%}
138
+ {{- '<|"|>' + key + '<|"|>' -}}
139
+ {%- else -%}
140
+ {{- key -}}
141
+ {%- endif -%}
142
+ :{{- format_argument(value, escape_keys=escape_keys) -}}
143
+ {%- endfor -%}
144
+ {{- '}' -}}
145
+ {%- elif argument is sequence -%}
146
+ {{- '[' -}}
147
+ {%- for item in argument -%}
148
+ {{- format_argument(item, escape_keys=escape_keys) -}}
149
+ {%- if not loop.last %},{% endif -%}
150
+ {%- endfor -%}
151
+ {{- ']' -}}
152
+ {%- else -%}
153
+ {{- argument -}}
154
+ {%- endif -%}
155
+ {%- endmacro -%}
156
+ {%- macro strip_thinking(text) -%}
157
+ {%- set ns = namespace(result='') -%}
158
+ {%- for part in text.split('<channel|>') -%}
159
+ {%- if '<|channel>' in part -%}
160
+ {%- set ns.result = ns.result + part.split('<|channel>')[0] -%}
161
+ {%- else -%}
162
+ {%- set ns.result = ns.result + part -%}
163
+ {%- endif -%}
164
+ {%- endfor -%}
165
+ {{- ns.result | trim -}}
166
+ {%- endmacro -%}
167
+
168
+ {%- macro format_tool_response_block(tool_name, response) -%}
169
+ {{- '<|tool_response>' -}}
170
+ {%- if response is mapping -%}
171
+ {{- 'response:' + tool_name + '{' -}}
172
+ {%- for key, value in response | dictsort -%}
173
+ {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
174
+ {%- if not loop.last %},{% endif -%}
175
+ {%- endfor -%}
176
+ {{- '}' -}}
177
+ {%- else -%}
178
+ {{- 'response:' + tool_name + '{value:' + format_argument(response, escape_keys=False) + '}' -}}
179
+ {%- endif -%}
180
+ {{- '<tool_response|>' -}}
181
+ {%- endmacro -%}
182
+
183
+ {#- ===== SETUP ===== -#}
184
+ {%- set ns = namespace(prev_message_type=None, prev_non_tool_role=None) -%}
185
+ {%- set loop_messages = messages -%}
186
+ {%- set enable_thinking = enable_thinking | default(false) -%}
187
+ {%- set preserve_thinking = preserve_thinking | default(false) -%}
188
+ {{- bos_token -}}
189
+ {#- Handle System/Tool Definitions Block -#}
190
+ {%- if enable_thinking or tools or (messages and messages[0]['role'] in ['system', 'developer']) -%}
191
+ {{- '<|turn>system\n' -}}
192
+ {#- Inject Thinking token at the very top of the FIRST system turn -#}
193
+ {%- if enable_thinking -%}
194
+ {{- '<|think|>\n' -}}
195
+ {%- set ns.prev_message_type = 'think' -%}
196
+ {%- endif -%}
197
+ {%- if messages and messages[0]['role'] in ['system', 'developer'] -%}
198
+ {%- if messages[0]['content'] is string -%}
199
+ {{- messages[0]['content'] | trim -}}
200
+ {%- elif messages[0]['content'] is sequence -%}
201
+ {%- for item in messages[0]['content'] -%}
202
+ {{- item['text'] | trim + ' '-}}
203
+ {%- endfor -%}
204
+ {%- endif -%}
205
+ {%- set loop_messages = messages[1:] -%}
206
+ {%- endif -%}
207
+ {%- if tools -%}
208
+ {%- for tool in tools %}
209
+ {{- '<|tool>' -}}
210
+ {{- format_function_declaration(tool) | trim -}}
211
+ {{- '<tool|>' -}}
212
+ {%- endfor %}
213
+ {%- set ns.prev_message_type = 'tool' -%}
214
+ {%- endif -%}
215
+ {{- '<turn|>\n' -}}
216
+ {%- endif %}
217
+
218
+ {#- Pre-scan: find last user message index for reasoning guard -#}
219
+ {%- set ns_turn = namespace(last_user_idx=-1) -%}
220
+ {%- for i in range(loop_messages | length) -%}
221
+ {%- if loop_messages[i]['role'] == 'user' -%}
222
+ {%- set ns_turn.last_user_idx = i -%}
223
+ {%- endif -%}
224
+ {%- endfor -%}
225
+
226
+ {#- Loop through messages -#}
227
+ {%- for message in loop_messages -%}
228
+ {%- if message['role'] != 'tool' -%}
229
+ {%- set ns.prev_message_type = None -%}
230
+ {%- set role = 'model' if message['role'] == 'assistant' else message['role'] -%}
231
+ {#- Detect continuation using tracked state — O(1) instead of O(n) backward scan -#}
232
+ {%- set continue_same_model_turn = (role == 'model' and ns.prev_non_tool_role == 'assistant') -%}
233
+ {%- if not continue_same_model_turn -%}
234
+ {{- '<|turn>' + role + '\n' }}
235
+ {%- endif -%}
236
+
237
+ {#- Render reasoning/reasoning_content as thinking channel -#}
238
+ {%- set thinking_text = message.get('reasoning') or message.get('reasoning_content') -%}
239
+ {%- set thinking_gate = (loop.index0 > ns_turn.last_user_idx) or (preserve_thinking and message.get('tool_calls')) -%}
240
+ {%- if thinking_text and thinking_gate -%}
241
+ {{- '<|channel>thought\n' + thinking_text + '\n<channel|>' -}}
242
+ {%- endif -%}
243
+
244
+ {%- if message.get('tool_calls') -%}
245
+ {%- for tool_call in message.get('tool_calls') -%}
246
+ {%- set function = tool_call['function'] -%}
247
+ {{- '<|tool_call>call:' + function['name'] + '{' -}}
248
+ {%- if function['arguments'] is mapping -%}
249
+ {%- set ns_args = namespace(found_first=false) -%}
250
+ {%- for key, value in function['arguments'] | dictsort -%}
251
+ {%- if ns_args.found_first %},{% endif -%}
252
+ {%- set ns_args.found_first = true -%}
253
+ {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
254
+ {%- endfor -%}
255
+ {%- elif function['arguments'] is none -%}
256
+ {%- else -%}
257
+ {{- raise_exception(
258
+ "chat_template: tool_calls[].function.arguments must be a "
259
+ "JSON object (mapping), not a string. Deserialize arguments "
260
+ "before passing to the template."
261
+ ) -}}
262
+ {%- endif -%}
263
+ {{- '}<tool_call|>' -}}
264
+ {%- endfor -%}
265
+ {%- set ns.prev_message_type = 'tool_call' -%}
266
+ {%- endif -%}
267
+
268
+ {%- set ns_tr_out = namespace(flag=false) -%}
269
+ {%- if message.get('tool_responses') -%}
270
+ {#- Legacy: tool_responses embedded on the assistant message (Google/Gemma native) -#}
271
+ {%- for tool_response in message.get('tool_responses') -%}
272
+ {{- format_tool_response_block(tool_response['name'] | default('unknown', true), tool_response['response']) -}}
273
+ {%- set ns_tr_out.flag = true -%}
274
+ {%- set ns.prev_message_type = 'tool_response' -%}
275
+ {%- endfor -%}
276
+ {%- elif message.get('tool_calls') -%}
277
+ {#- OpenAI Chat Completions: forward-scan consecutive role:tool messages -#}
278
+ {%- set ns_tool_scan = namespace(stopped=false) -%}
279
+ {%- for k in range(loop.index0 + 1, loop_messages | length) -%}
280
+ {%- if ns_tool_scan.stopped -%}
281
+ {%- elif loop_messages[k]['role'] != 'tool' -%}
282
+ {%- set ns_tool_scan.stopped = true -%}
283
+ {%- else -%}
284
+ {%- set follow = loop_messages[k] -%}
285
+ {#- Resolve tool_call_id to function name -#}
286
+ {%- set ns_tname = namespace(name=follow.get('name') or 'unknown') -%}
287
+ {%- for tc in message.get('tool_calls') -%}
288
+ {%- if tc.get('id') == follow.get('tool_call_id') -%}
289
+ {%- set ns_tname.name = tc['function']['name'] -%}
290
+ {%- endif -%}
291
+ {%- endfor -%}
292
+ {#- Handle content as string or content-parts array -#}
293
+ {%- set tool_body = follow.get('content') -%}
294
+ {%- if tool_body is string -%}
295
+ {{- format_tool_response_block(ns_tname.name, tool_body) -}}
296
+ {%- elif tool_body is sequence and tool_body is not string -%}
297
+ {%- set ns_txt = namespace(s='') -%}
298
+ {%- for part in tool_body -%}
299
+ {%- if part.get('type') == 'text' -%}
300
+ {%- set ns_txt.s = ns_txt.s + (part.get('text') | default('')) -%}
301
+ {%- endif -%}
302
+ {%- endfor -%}
303
+ {{- format_tool_response_block(ns_tname.name, ns_txt.s) -}}
304
+ {%- for part in tool_body -%}
305
+ {%- if part.get('type') in ['image', 'image_url'] -%}
306
+ {{- '<|image|>' -}}
307
+ {%- elif part.get('type') in ['audio', 'input_audio'] -%}
308
+ {{- '<|audio|>' -}}
309
+ {%- elif part.get('type') == 'video' -%}
310
+ {{- '<|video|>' -}}
311
+ {%- endif -%}
312
+ {%- endfor -%}
313
+ {%- else -%}
314
+ {{- format_tool_response_block(ns_tname.name, tool_body) -}}
315
+ {%- endif -%}
316
+ {%- set ns_tr_out.flag = true -%}
317
+ {%- set ns.prev_message_type = 'tool_response' -%}
318
+ {%- endif -%}
319
+ {%- endfor -%}
320
+ {%- endif -%}
321
+
322
+ {%- set captured_content -%}
323
+ {%- if message.get('content') is string -%}
324
+ {%- if role == 'model' -%}
325
+ {{- strip_thinking(message['content']) -}}
326
+ {%- else -%}
327
+ {{- message['content'] | trim -}}
328
+ {%- endif -%}
329
+ {%- elif message.get('content') is sequence -%}
330
+ {%- for item in message['content'] -%}
331
+ {%- if item.get('type') == 'text' -%}
332
+ {%- if role == 'model' -%}
333
+ {{- strip_thinking(item['text']) -}}
334
+ {%- else -%}
335
+ {{- item['text'] | trim -}}
336
+ {%- endif -%}
337
+ {%- elif item.get('type') in ['image', 'image_url'] -%}
338
+ {{- '<|image|>' -}}
339
+ {%- elif item.get('type') in ['audio', 'input_audio'] -%}
340
+ {{- '<|audio|>' -}}
341
+ {%- elif item.get('type') == 'video' -%}
342
+ {{- '<|video|>' -}}
343
+ {%- endif -%}
344
+ {%- endfor -%}
345
+ {%- endif -%}
346
+ {%- endset -%}
347
+
348
+ {{- captured_content -}}
349
+ {%- set has_content = captured_content | trim | length > 0 -%}
350
+
351
+ {#- Forward-scan: find next non-tool message role for continuation detection -#}
352
+ {%- set next_nt = namespace(role=None, found=false) -%}
353
+ {%- for j in range(loop.index0 + 1, loop_messages | length) -%}
354
+ {%- if not next_nt.found -%}
355
+ {%- if loop_messages[j]['role'] != 'tool' -%}
356
+ {%- set next_nt.role = loop_messages[j]['role'] -%}
357
+ {%- set next_nt.found = true -%}
358
+ {%- endif -%}
359
+ {%- endif -%}
360
+ {%- endfor -%}
361
+
362
+ {%- set continues_into_next = (
363
+ role == 'model'
364
+ and next_nt.role == 'assistant'
365
+ and (not message.get('tool_calls') or ns_tr_out.flag)
366
+ ) -%}
367
+
368
+ {%- if ns.prev_message_type == 'tool_call' and not ns_tr_out.flag -%}
369
+ {{- '<|tool_response>' -}}
370
+ {%- elif continues_into_next -%}
371
+ {%- elif not (ns_tr_out.flag and not has_content and not next_nt.found) -%}
372
+ {{- '<turn|>\n' -}}
373
+ {%- endif -%}
374
+
375
+ {#- Track previous non-tool role for next iteration (avoids O(n) backward scan) -#}
376
+ {%- set ns.prev_non_tool_role = message['role'] -%}
377
+ {%- endif -%}
378
+ {%- endfor -%}
379
+
380
+ {%- if add_generation_prompt -%}
381
+ {%- if ns.prev_message_type != 'tool_response' and ns.prev_message_type != 'tool_call' -%}
382
+ {{- '<|turn>model\n' -}}
383
+ {%- elif ns.prev_message_type == 'tool_response' and enable_thinking -%}
384
+ {{- '<|channel>thought\n' -}}
385
+ {%- endif -%}
386
+ {%- endif -%}
config.json ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Gemma4ForConditionalGeneration"
4
+ ],
5
+ "audio_config": {
6
+ "_name_or_path": "",
7
+ "architectures": null,
8
+ "attention_chunk_size": 12,
9
+ "attention_context_left": 13,
10
+ "attention_context_right": 0,
11
+ "attention_invalid_logits_value": -1000000000.0,
12
+ "attention_logit_cap": 50.0,
13
+ "chunk_size_feed_forward": 0,
14
+ "conv_kernel_size": 5,
15
+ "dtype": "bfloat16",
16
+ "gradient_clipping": 10000000000.0,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 1024,
19
+ "id2label": {
20
+ "0": "LABEL_0",
21
+ "1": "LABEL_1"
22
+ },
23
+ "initializer_range": 0.02,
24
+ "is_encoder_decoder": false,
25
+ "label2id": {
26
+ "LABEL_0": 0,
27
+ "LABEL_1": 1
28
+ },
29
+ "model_type": "gemma4_audio",
30
+ "num_attention_heads": 8,
31
+ "num_hidden_layers": 12,
32
+ "output_attentions": false,
33
+ "output_hidden_states": false,
34
+ "output_proj_dims": 1536,
35
+ "problem_type": null,
36
+ "residual_weight": 0.5,
37
+ "return_dict": true,
38
+ "rms_norm_eps": 1e-06,
39
+ "subsampling_conv_channels": [
40
+ 128,
41
+ 32
42
+ ],
43
+ "use_clipped_linears": true
44
+ },
45
+ "audio_token_id": 258881,
46
+ "boa_token_id": 256000,
47
+ "boi_token_id": 255999,
48
+ "dtype": "bfloat16",
49
+ "eoa_token_id": 258883,
50
+ "eoa_token_index": 258883,
51
+ "eoi_token_id": 258882,
52
+ "eos_token_id": [
53
+ 1,
54
+ 106
55
+ ],
56
+ "image_token_id": 258880,
57
+ "initializer_range": 0.02,
58
+ "model_type": "gemma4",
59
+ "text_config": {
60
+ "attention_bias": false,
61
+ "attention_dropout": 0.0,
62
+ "attention_k_eq_v": false,
63
+ "bos_token_id": 2,
64
+ "dtype": "bfloat16",
65
+ "enable_moe_block": false,
66
+ "eos_token_id": 1,
67
+ "expert_intermediate_size": null,
68
+ "final_logit_softcapping": 30.0,
69
+ "global_head_dim": 512,
70
+ "head_dim": 256,
71
+ "hidden_activation": "gelu_pytorch_tanh",
72
+ "hidden_size": 2560,
73
+ "hidden_size_per_layer_input": 256,
74
+ "initializer_range": 0.02,
75
+ "intermediate_size": 10240,
76
+ "layer_types": [
77
+ "sliding_attention",
78
+ "sliding_attention",
79
+ "sliding_attention",
80
+ "sliding_attention",
81
+ "sliding_attention",
82
+ "full_attention",
83
+ "sliding_attention",
84
+ "sliding_attention",
85
+ "sliding_attention",
86
+ "sliding_attention",
87
+ "sliding_attention",
88
+ "full_attention",
89
+ "sliding_attention",
90
+ "sliding_attention",
91
+ "sliding_attention",
92
+ "sliding_attention",
93
+ "sliding_attention",
94
+ "full_attention",
95
+ "sliding_attention",
96
+ "sliding_attention",
97
+ "sliding_attention",
98
+ "sliding_attention",
99
+ "sliding_attention",
100
+ "full_attention",
101
+ "sliding_attention",
102
+ "sliding_attention",
103
+ "sliding_attention",
104
+ "sliding_attention",
105
+ "sliding_attention",
106
+ "full_attention",
107
+ "sliding_attention",
108
+ "sliding_attention",
109
+ "sliding_attention",
110
+ "sliding_attention",
111
+ "sliding_attention",
112
+ "full_attention",
113
+ "sliding_attention",
114
+ "sliding_attention",
115
+ "sliding_attention",
116
+ "sliding_attention",
117
+ "sliding_attention",
118
+ "full_attention"
119
+ ],
120
+ "max_position_embeddings": 131072,
121
+ "model_type": "gemma4_text",
122
+ "moe_intermediate_size": null,
123
+ "num_attention_heads": 8,
124
+ "num_experts": null,
125
+ "num_global_key_value_heads": null,
126
+ "num_hidden_layers": 42,
127
+ "num_key_value_heads": 2,
128
+ "num_kv_shared_layers": 18,
129
+ "pad_token_id": 0,
130
+ "rms_norm_eps": 1e-06,
131
+ "rope_parameters": {
132
+ "full_attention": {
133
+ "partial_rotary_factor": 0.25,
134
+ "rope_theta": 1000000.0,
135
+ "rope_type": "proportional"
136
+ },
137
+ "sliding_attention": {
138
+ "rope_theta": 10000.0,
139
+ "rope_type": "default"
140
+ }
141
+ },
142
+ "sliding_window": 512,
143
+ "tie_word_embeddings": true,
144
+ "top_k_experts": null,
145
+ "use_bidirectional_attention": null,
146
+ "use_cache": true,
147
+ "use_double_wide_mlp": false,
148
+ "vocab_size": 262144,
149
+ "vocab_size_per_layer_input": 262144
150
+ },
151
+ "tie_word_embeddings": true,
152
+ "transformers_version": "5.5.4",
153
+ "video_token_id": 258884,
154
+ "vision_config": {
155
+ "_name_or_path": "",
156
+ "architectures": null,
157
+ "attention_bias": false,
158
+ "attention_dropout": 0.0,
159
+ "chunk_size_feed_forward": 0,
160
+ "default_output_length": 280,
161
+ "dtype": "bfloat16",
162
+ "global_head_dim": 64,
163
+ "head_dim": 64,
164
+ "hidden_activation": "gelu_pytorch_tanh",
165
+ "hidden_size": 768,
166
+ "id2label": {
167
+ "0": "LABEL_0",
168
+ "1": "LABEL_1"
169
+ },
170
+ "initializer_range": 0.02,
171
+ "intermediate_size": 3072,
172
+ "is_encoder_decoder": false,
173
+ "label2id": {
174
+ "LABEL_0": 0,
175
+ "LABEL_1": 1
176
+ },
177
+ "max_position_embeddings": 131072,
178
+ "model_type": "gemma4_vision",
179
+ "num_attention_heads": 12,
180
+ "num_hidden_layers": 16,
181
+ "num_key_value_heads": 12,
182
+ "output_attentions": false,
183
+ "output_hidden_states": false,
184
+ "patch_size": 16,
185
+ "pooling_kernel_size": 3,
186
+ "position_embedding_size": 10240,
187
+ "problem_type": null,
188
+ "return_dict": true,
189
+ "rms_norm_eps": 1e-06,
190
+ "rope_parameters": {
191
+ "rope_theta": 100.0,
192
+ "rope_type": "default"
193
+ },
194
+ "standardize": false,
195
+ "use_clipped_linears": true
196
+ },
197
+ "vision_soft_tokens_per_image": 280
198
+ }
engine.json ADDED
@@ -0,0 +1,197 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arch": "gemma4",
3
+ "T": 1024,
4
+ "sizes": [
5
+ 1,
6
+ 16,
7
+ 64
8
+ ],
9
+ "config": {
10
+ "attention_bias": false,
11
+ "attention_dropout": 0.0,
12
+ "attention_k_eq_v": false,
13
+ "bos_token_id": 2,
14
+ "dtype": "bfloat16",
15
+ "enable_moe_block": false,
16
+ "eos_token_id": 1,
17
+ "expert_intermediate_size": null,
18
+ "final_logit_softcapping": 30.0,
19
+ "global_head_dim": 512,
20
+ "head_dim": 256,
21
+ "hidden_activation": "gelu_pytorch_tanh",
22
+ "hidden_size": 2560,
23
+ "hidden_size_per_layer_input": 256,
24
+ "initializer_range": 0.02,
25
+ "intermediate_size": 10240,
26
+ "layer_types": [
27
+ "sliding_attention",
28
+ "sliding_attention",
29
+ "sliding_attention",
30
+ "sliding_attention",
31
+ "sliding_attention",
32
+ "full_attention",
33
+ "sliding_attention",
34
+ "sliding_attention",
35
+ "sliding_attention",
36
+ "sliding_attention",
37
+ "sliding_attention",
38
+ "full_attention",
39
+ "sliding_attention",
40
+ "sliding_attention",
41
+ "sliding_attention",
42
+ "sliding_attention",
43
+ "sliding_attention",
44
+ "full_attention",
45
+ "sliding_attention",
46
+ "sliding_attention",
47
+ "sliding_attention",
48
+ "sliding_attention",
49
+ "sliding_attention",
50
+ "full_attention",
51
+ "sliding_attention",
52
+ "sliding_attention",
53
+ "sliding_attention",
54
+ "sliding_attention",
55
+ "sliding_attention",
56
+ "full_attention",
57
+ "sliding_attention",
58
+ "sliding_attention",
59
+ "sliding_attention",
60
+ "sliding_attention",
61
+ "sliding_attention",
62
+ "full_attention",
63
+ "sliding_attention",
64
+ "sliding_attention",
65
+ "sliding_attention",
66
+ "sliding_attention",
67
+ "sliding_attention",
68
+ "full_attention"
69
+ ],
70
+ "max_position_embeddings": 131072,
71
+ "model_type": "gemma4_text",
72
+ "moe_intermediate_size": null,
73
+ "num_attention_heads": 8,
74
+ "num_experts": null,
75
+ "num_global_key_value_heads": null,
76
+ "num_hidden_layers": 42,
77
+ "num_key_value_heads": 2,
78
+ "num_kv_shared_layers": 18,
79
+ "pad_token_id": 0,
80
+ "rms_norm_eps": 1e-06,
81
+ "rope_parameters": {
82
+ "full_attention": {
83
+ "partial_rotary_factor": 0.25,
84
+ "rope_theta": 1000000.0,
85
+ "rope_type": "proportional"
86
+ },
87
+ "sliding_attention": {
88
+ "rope_theta": 10000.0,
89
+ "rope_type": "default"
90
+ }
91
+ },
92
+ "sliding_window": 512,
93
+ "tie_word_embeddings": true,
94
+ "top_k_experts": null,
95
+ "use_bidirectional_attention": null,
96
+ "use_cache": true,
97
+ "use_double_wide_mlp": false,
98
+ "vocab_size": 262144,
99
+ "vocab_size_per_layer_input": 262144
100
+ },
101
+ "segments": {
102
+ "seg00_S1.xml": {
103
+ "index": 0,
104
+ "consume_moe": null,
105
+ "router_layer": null,
106
+ "S": 1,
107
+ "slots": 0,
108
+ "head": false
109
+ },
110
+ "seg01_S1.xml": {
111
+ "index": 1,
112
+ "consume_moe": null,
113
+ "router_layer": null,
114
+ "S": 1,
115
+ "slots": 0,
116
+ "head": true
117
+ },
118
+ "seg00_S16.xml": {
119
+ "index": 0,
120
+ "consume_moe": null,
121
+ "router_layer": null,
122
+ "S": 16,
123
+ "slots": 0,
124
+ "head": false
125
+ },
126
+ "seg01_S16.xml": {
127
+ "index": 1,
128
+ "consume_moe": null,
129
+ "router_layer": null,
130
+ "S": 16,
131
+ "slots": 0,
132
+ "head": true
133
+ },
134
+ "seg00_S64.xml": {
135
+ "index": 0,
136
+ "consume_moe": null,
137
+ "router_layer": null,
138
+ "S": 64,
139
+ "slots": 0,
140
+ "head": false
141
+ },
142
+ "seg01_S64.xml": {
143
+ "index": 1,
144
+ "consume_moe": null,
145
+ "router_layer": null,
146
+ "S": 64,
147
+ "slots": 0,
148
+ "head": true
149
+ }
150
+ },
151
+ "masks": {
152
+ "sliding_window": 512
153
+ },
154
+ "host_inputs": {
155
+ "per_layer_inputs": {
156
+ "table": "ple",
157
+ "shape": [
158
+ 42,
159
+ 256
160
+ ]
161
+ }
162
+ },
163
+ "vision": {
164
+ "type": "gemma",
165
+ "soft_tokens": 280,
166
+ "image_processor": {
167
+ "do_convert_rgb": true,
168
+ "do_normalize": false,
169
+ "do_rescale": true,
170
+ "do_resize": true,
171
+ "image_mean": [
172
+ 0.0,
173
+ 0.0,
174
+ 0.0
175
+ ],
176
+ "image_processor_type": "Gemma4ImageProcessor",
177
+ "image_seq_length": 280,
178
+ "image_std": [
179
+ 1.0,
180
+ 1.0,
181
+ 1.0
182
+ ],
183
+ "max_soft_tokens": 280,
184
+ "patch_size": 16,
185
+ "pooling_kernel_size": 3,
186
+ "resample": 3,
187
+ "rescale_factor": 0.00392156862745098
188
+ },
189
+ "image_token_id": 258880,
190
+ "pad_token_id": 0
191
+ },
192
+ "npu_vision_fallback": {
193
+ "NPU_USE_NPUW": "YES",
194
+ "NPUW_DEVICES": "NPU",
195
+ "NPUW_FOLD": "YES"
196
+ }
197
+ }
generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 2,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 1,
6
+ 106,
7
+ 50
8
+ ],
9
+ "pad_token_id": 0,
10
+ "temperature": 1.0,
11
+ "top_k": 64,
12
+ "top_p": 0.95,
13
+ "transformers_version": "5.5.4"
14
+ }
preprocessor_config.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_convert_rgb": true,
3
+ "do_normalize": false,
4
+ "do_rescale": true,
5
+ "do_resize": true,
6
+ "image_mean": [
7
+ 0.0,
8
+ 0.0,
9
+ 0.0
10
+ ],
11
+ "image_processor_type": "Gemma4ImageProcessor",
12
+ "image_seq_length": 280,
13
+ "image_std": [
14
+ 1.0,
15
+ 1.0,
16
+ 1.0
17
+ ],
18
+ "max_soft_tokens": 280,
19
+ "patch_size": 16,
20
+ "pooling_kernel_size": 3,
21
+ "resample": 3,
22
+ "rescale_factor": 0.00392156862745098
23
+ }
processor_config.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "audio_ms_per_token": 40,
3
+ "audio_seq_length": 750,
4
+ "feature_extractor": {
5
+ "dither": 0.0,
6
+ "feature_extractor_type": "Gemma4AudioFeatureExtractor",
7
+ "feature_size": 128,
8
+ "fft_length": 512,
9
+ "fft_overdrive": false,
10
+ "frame_length": 320,
11
+ "hop_length": 160,
12
+ "input_scale_factor": 1.0,
13
+ "max_frequency": 8000.0,
14
+ "mel_floor": 0.001,
15
+ "min_frequency": 0.0,
16
+ "padding_side": "right",
17
+ "padding_value": 0.0,
18
+ "per_bin_mean": null,
19
+ "per_bin_stddev": null,
20
+ "preemphasis": 0.0,
21
+ "preemphasis_htk_flavor": true,
22
+ "return_attention_mask": true,
23
+ "sampling_rate": 16000
24
+ },
25
+ "image_processor": {
26
+ "do_convert_rgb": true,
27
+ "do_normalize": false,
28
+ "do_rescale": true,
29
+ "do_resize": true,
30
+ "image_mean": [
31
+ 0.0,
32
+ 0.0,
33
+ 0.0
34
+ ],
35
+ "image_processor_type": "Gemma4ImageProcessor",
36
+ "image_seq_length": 280,
37
+ "image_std": [
38
+ 1.0,
39
+ 1.0,
40
+ 1.0
41
+ ],
42
+ "max_soft_tokens": 280,
43
+ "patch_size": 16,
44
+ "pooling_kernel_size": 3,
45
+ "resample": 3,
46
+ "rescale_factor": 0.00392156862745098
47
+ },
48
+ "image_seq_length": 280,
49
+ "processor_class": "Gemma4Processor",
50
+ "video_processor": {
51
+ "do_convert_rgb": true,
52
+ "do_normalize": true,
53
+ "do_rescale": true,
54
+ "do_resize": true,
55
+ "do_sample_frames": true,
56
+ "image_mean": [
57
+ 0.0,
58
+ 0.0,
59
+ 0.0
60
+ ],
61
+ "image_std": [
62
+ 1.0,
63
+ 1.0,
64
+ 1.0
65
+ ],
66
+ "max_soft_tokens": 70,
67
+ "num_frames": 32,
68
+ "patch_size": 16,
69
+ "pooling_kernel_size": 3,
70
+ "resample": 3,
71
+ "rescale_factor": 0.00392156862745098,
72
+ "return_metadata": false,
73
+ "video_processor_type": "Gemma4VideoProcessor"
74
+ }
75
+ }
seg00_S1.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38b18a93e134fd55cdcd54772c26b4d966a371c6763eb8ab020e0ce91f87a971
3
+ size 2050670280
seg00_S1.xml ADDED
The diff for this file is too large to render. See raw diff
 
seg00_S16.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38b18a93e134fd55cdcd54772c26b4d966a371c6763eb8ab020e0ce91f87a971
3
+ size 2050670280
seg00_S16.xml ADDED
The diff for this file is too large to render. See raw diff
 
seg00_S64.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38b18a93e134fd55cdcd54772c26b4d966a371c6763eb8ab020e0ce91f87a971
3
+ size 2050670280
seg00_S64.xml ADDED
The diff for this file is too large to render. See raw diff
 
seg01_S1.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e48af94438b6e5de715ec5e3e80e252dc3f66d9ea35b3bdb8dac0650fa8ad71
3
+ size 6
seg01_S1.xml ADDED
@@ -0,0 +1,256 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0"?>
2
+ <net name="gemma4_head_S1" version="11">
3
+ <layers>
4
+ <layer id="0" name="carry.h" type="Parameter" version="opset1">
5
+ <data shape="1,1,2560" element_type="f16" />
6
+ <output>
7
+ <port id="0" precision="FP16" names="carry.h">
8
+ <dim>1</dim>
9
+ <dim>1</dim>
10
+ <dim>2560</dim>
11
+ </port>
12
+ </output>
13
+ </layer>
14
+ <layer id="1" name="shared.head_q" type="Parameter" version="opset1">
15
+ <data shape="262144,2560" element_type="i8" />
16
+ <output>
17
+ <port id="0" precision="I8" names="shared.head_q">
18
+ <dim>262144</dim>
19
+ <dim>2560</dim>
20
+ </port>
21
+ </output>
22
+ </layer>
23
+ <layer id="2" name="shared.head_s" type="Parameter" version="opset1">
24
+ <data shape="262144" element_type="f16" />
25
+ <output>
26
+ <port id="0" precision="FP16" names="shared.head_s">
27
+ <dim>262144</dim>
28
+ </port>
29
+ </output>
30
+ </layer>
31
+ <layer id="3" name="Convert_27003" type="Convert" version="opset1">
32
+ <data destination_type="f32" />
33
+ <input>
34
+ <port id="0" precision="FP16">
35
+ <dim>1</dim>
36
+ <dim>1</dim>
37
+ <dim>2560</dim>
38
+ </port>
39
+ </input>
40
+ <output>
41
+ <port id="1" precision="FP32">
42
+ <dim>1</dim>
43
+ <dim>1</dim>
44
+ <dim>2560</dim>
45
+ </port>
46
+ </output>
47
+ </layer>
48
+ <layer id="4" name="Constant_26988" type="Const" version="opset1">
49
+ <data element_type="f32" shape="" offset="0" size="4" />
50
+ <output>
51
+ <port id="0" precision="FP32" />
52
+ </output>
53
+ </layer>
54
+ <layer id="5" name="Multiply_26989" type="Multiply" version="opset1">
55
+ <data auto_broadcast="numpy" />
56
+ <input>
57
+ <port id="0" precision="FP32">
58
+ <dim>1</dim>
59
+ <dim>1</dim>
60
+ <dim>2560</dim>
61
+ </port>
62
+ <port id="1" precision="FP32" />
63
+ </input>
64
+ <output>
65
+ <port id="2" precision="FP32">
66
+ <dim>1</dim>
67
+ <dim>1</dim>
68
+ <dim>2560</dim>
69
+ </port>
70
+ </output>
71
+ </layer>
72
+ <layer id="6" name="Convert_26992" type="Convert" version="opset1">
73
+ <data destination_type="f16" />
74
+ <input>
75
+ <port id="0" precision="I8">
76
+ <dim>262144</dim>
77
+ <dim>2560</dim>
78
+ </port>
79
+ </input>
80
+ <output>
81
+ <port id="1" precision="FP16">
82
+ <dim>262144</dim>
83
+ <dim>2560</dim>
84
+ </port>
85
+ </output>
86
+ </layer>
87
+ <layer id="7" name="Convert_26993" type="Convert" version="opset1">
88
+ <data destination_type="f32" />
89
+ <input>
90
+ <port id="0" precision="FP16">
91
+ <dim>262144</dim>
92
+ <dim>2560</dim>
93
+ </port>
94
+ </input>
95
+ <output>
96
+ <port id="1" precision="FP32">
97
+ <dim>262144</dim>
98
+ <dim>2560</dim>
99
+ </port>
100
+ </output>
101
+ </layer>
102
+ <layer id="8" name="MatMul_26994" type="MatMul" version="opset1">
103
+ <data transpose_a="false" transpose_b="true" />
104
+ <input>
105
+ <port id="0" precision="FP32">
106
+ <dim>1</dim>
107
+ <dim>1</dim>
108
+ <dim>2560</dim>
109
+ </port>
110
+ <port id="1" precision="FP32">
111
+ <dim>262144</dim>
112
+ <dim>2560</dim>
113
+ </port>
114
+ </input>
115
+ <output>
116
+ <port id="2" precision="FP32">
117
+ <dim>1</dim>
118
+ <dim>1</dim>
119
+ <dim>262144</dim>
120
+ </port>
121
+ </output>
122
+ </layer>
123
+ <layer id="9" name="Convert_26995" type="Convert" version="opset1">
124
+ <data destination_type="f32" />
125
+ <input>
126
+ <port id="0" precision="FP16">
127
+ <dim>262144</dim>
128
+ </port>
129
+ </input>
130
+ <output>
131
+ <port id="1" precision="FP32">
132
+ <dim>262144</dim>
133
+ </port>
134
+ </output>
135
+ </layer>
136
+ <layer id="10" name="Multiply_26996" type="Multiply" version="opset1">
137
+ <data auto_broadcast="numpy" />
138
+ <input>
139
+ <port id="0" precision="FP32">
140
+ <dim>1</dim>
141
+ <dim>1</dim>
142
+ <dim>262144</dim>
143
+ </port>
144
+ <port id="1" precision="FP32">
145
+ <dim>262144</dim>
146
+ </port>
147
+ </input>
148
+ <output>
149
+ <port id="2" precision="FP32">
150
+ <dim>1</dim>
151
+ <dim>1</dim>
152
+ <dim>262144</dim>
153
+ </port>
154
+ </output>
155
+ </layer>
156
+ <layer id="11" name="Tanh_26997" type="Tanh" version="opset1">
157
+ <input>
158
+ <port id="0" precision="FP32">
159
+ <dim>1</dim>
160
+ <dim>1</dim>
161
+ <dim>262144</dim>
162
+ </port>
163
+ </input>
164
+ <output>
165
+ <port id="1" precision="FP32">
166
+ <dim>1</dim>
167
+ <dim>1</dim>
168
+ <dim>262144</dim>
169
+ </port>
170
+ </output>
171
+ </layer>
172
+ <layer id="12" name="Constant_26999_compressed" type="Const" version="opset1">
173
+ <data element_type="f16" shape="" offset="4" size="2" />
174
+ <output>
175
+ <port id="0" precision="FP16" />
176
+ </output>
177
+ </layer>
178
+ <layer id="13" name="Constant_26999" type="Convert" version="opset1">
179
+ <data destination_type="f32" />
180
+ <rt_info>
181
+ <attribute name="decompression" version="0" />
182
+ </rt_info>
183
+ <input>
184
+ <port id="0" precision="FP16" />
185
+ </input>
186
+ <output>
187
+ <port id="1" precision="FP32" />
188
+ </output>
189
+ </layer>
190
+ <layer id="14" name="Multiply_27000" type="Multiply" version="opset1">
191
+ <data auto_broadcast="numpy" />
192
+ <input>
193
+ <port id="0" precision="FP32">
194
+ <dim>1</dim>
195
+ <dim>1</dim>
196
+ <dim>262144</dim>
197
+ </port>
198
+ <port id="1" precision="FP32" />
199
+ </input>
200
+ <output>
201
+ <port id="2" precision="FP32">
202
+ <dim>1</dim>
203
+ <dim>1</dim>
204
+ <dim>262144</dim>
205
+ </port>
206
+ </output>
207
+ </layer>
208
+ <layer id="15" name="Multiply_27000.0" type="Convert" version="opset1">
209
+ <data destination_type="f16" />
210
+ <input>
211
+ <port id="0" precision="FP32">
212
+ <dim>1</dim>
213
+ <dim>1</dim>
214
+ <dim>262144</dim>
215
+ </port>
216
+ </input>
217
+ <output>
218
+ <port id="1" precision="FP16" names="logits">
219
+ <dim>1</dim>
220
+ <dim>1</dim>
221
+ <dim>262144</dim>
222
+ </port>
223
+ </output>
224
+ </layer>
225
+ <layer id="16" name="Result_27001" type="Result" version="opset1" output_names="logits">
226
+ <input>
227
+ <port id="0" precision="FP16">
228
+ <dim>1</dim>
229
+ <dim>1</dim>
230
+ <dim>262144</dim>
231
+ </port>
232
+ </input>
233
+ </layer>
234
+ </layers>
235
+ <edges>
236
+ <edge from-layer="0" from-port="0" to-layer="3" to-port="0" />
237
+ <edge from-layer="1" from-port="0" to-layer="6" to-port="0" />
238
+ <edge from-layer="2" from-port="0" to-layer="9" to-port="0" />
239
+ <edge from-layer="3" from-port="1" to-layer="5" to-port="0" />
240
+ <edge from-layer="4" from-port="0" to-layer="5" to-port="1" />
241
+ <edge from-layer="5" from-port="2" to-layer="8" to-port="0" />
242
+ <edge from-layer="6" from-port="1" to-layer="7" to-port="0" />
243
+ <edge from-layer="7" from-port="1" to-layer="8" to-port="1" />
244
+ <edge from-layer="8" from-port="2" to-layer="10" to-port="0" />
245
+ <edge from-layer="9" from-port="1" to-layer="10" to-port="1" />
246
+ <edge from-layer="10" from-port="2" to-layer="11" to-port="0" />
247
+ <edge from-layer="11" from-port="1" to-layer="14" to-port="0" />
248
+ <edge from-layer="12" from-port="0" to-layer="13" to-port="0" />
249
+ <edge from-layer="13" from-port="1" to-layer="14" to-port="1" />
250
+ <edge from-layer="14" from-port="2" to-layer="15" to-port="0" />
251
+ <edge from-layer="15" from-port="1" to-layer="16" to-port="0" />
252
+ </edges>
253
+ <rt_info>
254
+ <info name="OpenVINO Runtime" value="2026.5.0-23160-6db412d0a27" />
255
+ </rt_info>
256
+ </net>
seg01_S16.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e48af94438b6e5de715ec5e3e80e252dc3f66d9ea35b3bdb8dac0650fa8ad71
3
+ size 6
seg01_S16.xml ADDED
@@ -0,0 +1,256 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0"?>
2
+ <net name="gemma4_head_S16" version="11">
3
+ <layers>
4
+ <layer id="0" name="carry.h" type="Parameter" version="opset1">
5
+ <data shape="1,16,2560" element_type="f16" />
6
+ <output>
7
+ <port id="0" precision="FP16" names="carry.h">
8
+ <dim>1</dim>
9
+ <dim>16</dim>
10
+ <dim>2560</dim>
11
+ </port>
12
+ </output>
13
+ </layer>
14
+ <layer id="1" name="shared.head_q" type="Parameter" version="opset1">
15
+ <data shape="262144,2560" element_type="i8" />
16
+ <output>
17
+ <port id="0" precision="I8" names="shared.head_q">
18
+ <dim>262144</dim>
19
+ <dim>2560</dim>
20
+ </port>
21
+ </output>
22
+ </layer>
23
+ <layer id="2" name="shared.head_s" type="Parameter" version="opset1">
24
+ <data shape="262144" element_type="f16" />
25
+ <output>
26
+ <port id="0" precision="FP16" names="shared.head_s">
27
+ <dim>262144</dim>
28
+ </port>
29
+ </output>
30
+ </layer>
31
+ <layer id="3" name="Convert_54279" type="Convert" version="opset1">
32
+ <data destination_type="f32" />
33
+ <input>
34
+ <port id="0" precision="FP16">
35
+ <dim>1</dim>
36
+ <dim>16</dim>
37
+ <dim>2560</dim>
38
+ </port>
39
+ </input>
40
+ <output>
41
+ <port id="1" precision="FP32">
42
+ <dim>1</dim>
43
+ <dim>16</dim>
44
+ <dim>2560</dim>
45
+ </port>
46
+ </output>
47
+ </layer>
48
+ <layer id="4" name="Constant_54264" type="Const" version="opset1">
49
+ <data element_type="f32" shape="" offset="0" size="4" />
50
+ <output>
51
+ <port id="0" precision="FP32" />
52
+ </output>
53
+ </layer>
54
+ <layer id="5" name="Multiply_54265" type="Multiply" version="opset1">
55
+ <data auto_broadcast="numpy" />
56
+ <input>
57
+ <port id="0" precision="FP32">
58
+ <dim>1</dim>
59
+ <dim>16</dim>
60
+ <dim>2560</dim>
61
+ </port>
62
+ <port id="1" precision="FP32" />
63
+ </input>
64
+ <output>
65
+ <port id="2" precision="FP32">
66
+ <dim>1</dim>
67
+ <dim>16</dim>
68
+ <dim>2560</dim>
69
+ </port>
70
+ </output>
71
+ </layer>
72
+ <layer id="6" name="Convert_54268" type="Convert" version="opset1">
73
+ <data destination_type="f16" />
74
+ <input>
75
+ <port id="0" precision="I8">
76
+ <dim>262144</dim>
77
+ <dim>2560</dim>
78
+ </port>
79
+ </input>
80
+ <output>
81
+ <port id="1" precision="FP16">
82
+ <dim>262144</dim>
83
+ <dim>2560</dim>
84
+ </port>
85
+ </output>
86
+ </layer>
87
+ <layer id="7" name="Convert_54269" type="Convert" version="opset1">
88
+ <data destination_type="f32" />
89
+ <input>
90
+ <port id="0" precision="FP16">
91
+ <dim>262144</dim>
92
+ <dim>2560</dim>
93
+ </port>
94
+ </input>
95
+ <output>
96
+ <port id="1" precision="FP32">
97
+ <dim>262144</dim>
98
+ <dim>2560</dim>
99
+ </port>
100
+ </output>
101
+ </layer>
102
+ <layer id="8" name="MatMul_54270" type="MatMul" version="opset1">
103
+ <data transpose_a="false" transpose_b="true" />
104
+ <input>
105
+ <port id="0" precision="FP32">
106
+ <dim>1</dim>
107
+ <dim>16</dim>
108
+ <dim>2560</dim>
109
+ </port>
110
+ <port id="1" precision="FP32">
111
+ <dim>262144</dim>
112
+ <dim>2560</dim>
113
+ </port>
114
+ </input>
115
+ <output>
116
+ <port id="2" precision="FP32">
117
+ <dim>1</dim>
118
+ <dim>16</dim>
119
+ <dim>262144</dim>
120
+ </port>
121
+ </output>
122
+ </layer>
123
+ <layer id="9" name="Convert_54271" type="Convert" version="opset1">
124
+ <data destination_type="f32" />
125
+ <input>
126
+ <port id="0" precision="FP16">
127
+ <dim>262144</dim>
128
+ </port>
129
+ </input>
130
+ <output>
131
+ <port id="1" precision="FP32">
132
+ <dim>262144</dim>
133
+ </port>
134
+ </output>
135
+ </layer>
136
+ <layer id="10" name="Multiply_54272" type="Multiply" version="opset1">
137
+ <data auto_broadcast="numpy" />
138
+ <input>
139
+ <port id="0" precision="FP32">
140
+ <dim>1</dim>
141
+ <dim>16</dim>
142
+ <dim>262144</dim>
143
+ </port>
144
+ <port id="1" precision="FP32">
145
+ <dim>262144</dim>
146
+ </port>
147
+ </input>
148
+ <output>
149
+ <port id="2" precision="FP32">
150
+ <dim>1</dim>
151
+ <dim>16</dim>
152
+ <dim>262144</dim>
153
+ </port>
154
+ </output>
155
+ </layer>
156
+ <layer id="11" name="Tanh_54273" type="Tanh" version="opset1">
157
+ <input>
158
+ <port id="0" precision="FP32">
159
+ <dim>1</dim>
160
+ <dim>16</dim>
161
+ <dim>262144</dim>
162
+ </port>
163
+ </input>
164
+ <output>
165
+ <port id="1" precision="FP32">
166
+ <dim>1</dim>
167
+ <dim>16</dim>
168
+ <dim>262144</dim>
169
+ </port>
170
+ </output>
171
+ </layer>
172
+ <layer id="12" name="Constant_54275_compressed" type="Const" version="opset1">
173
+ <data element_type="f16" shape="" offset="4" size="2" />
174
+ <output>
175
+ <port id="0" precision="FP16" />
176
+ </output>
177
+ </layer>
178
+ <layer id="13" name="Constant_54275" type="Convert" version="opset1">
179
+ <data destination_type="f32" />
180
+ <rt_info>
181
+ <attribute name="decompression" version="0" />
182
+ </rt_info>
183
+ <input>
184
+ <port id="0" precision="FP16" />
185
+ </input>
186
+ <output>
187
+ <port id="1" precision="FP32" />
188
+ </output>
189
+ </layer>
190
+ <layer id="14" name="Multiply_54276" type="Multiply" version="opset1">
191
+ <data auto_broadcast="numpy" />
192
+ <input>
193
+ <port id="0" precision="FP32">
194
+ <dim>1</dim>
195
+ <dim>16</dim>
196
+ <dim>262144</dim>
197
+ </port>
198
+ <port id="1" precision="FP32" />
199
+ </input>
200
+ <output>
201
+ <port id="2" precision="FP32">
202
+ <dim>1</dim>
203
+ <dim>16</dim>
204
+ <dim>262144</dim>
205
+ </port>
206
+ </output>
207
+ </layer>
208
+ <layer id="15" name="Multiply_54276.0" type="Convert" version="opset1">
209
+ <data destination_type="f16" />
210
+ <input>
211
+ <port id="0" precision="FP32">
212
+ <dim>1</dim>
213
+ <dim>16</dim>
214
+ <dim>262144</dim>
215
+ </port>
216
+ </input>
217
+ <output>
218
+ <port id="1" precision="FP16" names="logits">
219
+ <dim>1</dim>
220
+ <dim>16</dim>
221
+ <dim>262144</dim>
222
+ </port>
223
+ </output>
224
+ </layer>
225
+ <layer id="16" name="Result_54277" type="Result" version="opset1" output_names="logits">
226
+ <input>
227
+ <port id="0" precision="FP16">
228
+ <dim>1</dim>
229
+ <dim>16</dim>
230
+ <dim>262144</dim>
231
+ </port>
232
+ </input>
233
+ </layer>
234
+ </layers>
235
+ <edges>
236
+ <edge from-layer="0" from-port="0" to-layer="3" to-port="0" />
237
+ <edge from-layer="1" from-port="0" to-layer="6" to-port="0" />
238
+ <edge from-layer="2" from-port="0" to-layer="9" to-port="0" />
239
+ <edge from-layer="3" from-port="1" to-layer="5" to-port="0" />
240
+ <edge from-layer="4" from-port="0" to-layer="5" to-port="1" />
241
+ <edge from-layer="5" from-port="2" to-layer="8" to-port="0" />
242
+ <edge from-layer="6" from-port="1" to-layer="7" to-port="0" />
243
+ <edge from-layer="7" from-port="1" to-layer="8" to-port="1" />
244
+ <edge from-layer="8" from-port="2" to-layer="10" to-port="0" />
245
+ <edge from-layer="9" from-port="1" to-layer="10" to-port="1" />
246
+ <edge from-layer="10" from-port="2" to-layer="11" to-port="0" />
247
+ <edge from-layer="11" from-port="1" to-layer="14" to-port="0" />
248
+ <edge from-layer="12" from-port="0" to-layer="13" to-port="0" />
249
+ <edge from-layer="13" from-port="1" to-layer="14" to-port="1" />
250
+ <edge from-layer="14" from-port="2" to-layer="15" to-port="0" />
251
+ <edge from-layer="15" from-port="1" to-layer="16" to-port="0" />
252
+ </edges>
253
+ <rt_info>
254
+ <info name="OpenVINO Runtime" value="2026.5.0-23160-6db412d0a27" />
255
+ </rt_info>
256
+ </net>
seg01_S64.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e48af94438b6e5de715ec5e3e80e252dc3f66d9ea35b3bdb8dac0650fa8ad71
3
+ size 6
seg01_S64.xml ADDED
@@ -0,0 +1,256 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0"?>
2
+ <net name="gemma4_head_S64" version="11">
3
+ <layers>
4
+ <layer id="0" name="carry.h" type="Parameter" version="opset1">
5
+ <data shape="1,64,2560" element_type="f16" />
6
+ <output>
7
+ <port id="0" precision="FP16" names="carry.h">
8
+ <dim>1</dim>
9
+ <dim>64</dim>
10
+ <dim>2560</dim>
11
+ </port>
12
+ </output>
13
+ </layer>
14
+ <layer id="1" name="shared.head_q" type="Parameter" version="opset1">
15
+ <data shape="262144,2560" element_type="i8" />
16
+ <output>
17
+ <port id="0" precision="I8" names="shared.head_q">
18
+ <dim>262144</dim>
19
+ <dim>2560</dim>
20
+ </port>
21
+ </output>
22
+ </layer>
23
+ <layer id="2" name="shared.head_s" type="Parameter" version="opset1">
24
+ <data shape="262144" element_type="f16" />
25
+ <output>
26
+ <port id="0" precision="FP16" names="shared.head_s">
27
+ <dim>262144</dim>
28
+ </port>
29
+ </output>
30
+ </layer>
31
+ <layer id="3" name="Convert_81368" type="Convert" version="opset1">
32
+ <data destination_type="f32" />
33
+ <input>
34
+ <port id="0" precision="FP16">
35
+ <dim>1</dim>
36
+ <dim>64</dim>
37
+ <dim>2560</dim>
38
+ </port>
39
+ </input>
40
+ <output>
41
+ <port id="1" precision="FP32">
42
+ <dim>1</dim>
43
+ <dim>64</dim>
44
+ <dim>2560</dim>
45
+ </port>
46
+ </output>
47
+ </layer>
48
+ <layer id="4" name="Constant_81353" type="Const" version="opset1">
49
+ <data element_type="f32" shape="" offset="0" size="4" />
50
+ <output>
51
+ <port id="0" precision="FP32" />
52
+ </output>
53
+ </layer>
54
+ <layer id="5" name="Multiply_81354" type="Multiply" version="opset1">
55
+ <data auto_broadcast="numpy" />
56
+ <input>
57
+ <port id="0" precision="FP32">
58
+ <dim>1</dim>
59
+ <dim>64</dim>
60
+ <dim>2560</dim>
61
+ </port>
62
+ <port id="1" precision="FP32" />
63
+ </input>
64
+ <output>
65
+ <port id="2" precision="FP32">
66
+ <dim>1</dim>
67
+ <dim>64</dim>
68
+ <dim>2560</dim>
69
+ </port>
70
+ </output>
71
+ </layer>
72
+ <layer id="6" name="Convert_81357" type="Convert" version="opset1">
73
+ <data destination_type="f16" />
74
+ <input>
75
+ <port id="0" precision="I8">
76
+ <dim>262144</dim>
77
+ <dim>2560</dim>
78
+ </port>
79
+ </input>
80
+ <output>
81
+ <port id="1" precision="FP16">
82
+ <dim>262144</dim>
83
+ <dim>2560</dim>
84
+ </port>
85
+ </output>
86
+ </layer>
87
+ <layer id="7" name="Convert_81358" type="Convert" version="opset1">
88
+ <data destination_type="f32" />
89
+ <input>
90
+ <port id="0" precision="FP16">
91
+ <dim>262144</dim>
92
+ <dim>2560</dim>
93
+ </port>
94
+ </input>
95
+ <output>
96
+ <port id="1" precision="FP32">
97
+ <dim>262144</dim>
98
+ <dim>2560</dim>
99
+ </port>
100
+ </output>
101
+ </layer>
102
+ <layer id="8" name="MatMul_81359" type="MatMul" version="opset1">
103
+ <data transpose_a="false" transpose_b="true" />
104
+ <input>
105
+ <port id="0" precision="FP32">
106
+ <dim>1</dim>
107
+ <dim>64</dim>
108
+ <dim>2560</dim>
109
+ </port>
110
+ <port id="1" precision="FP32">
111
+ <dim>262144</dim>
112
+ <dim>2560</dim>
113
+ </port>
114
+ </input>
115
+ <output>
116
+ <port id="2" precision="FP32">
117
+ <dim>1</dim>
118
+ <dim>64</dim>
119
+ <dim>262144</dim>
120
+ </port>
121
+ </output>
122
+ </layer>
123
+ <layer id="9" name="Convert_81360" type="Convert" version="opset1">
124
+ <data destination_type="f32" />
125
+ <input>
126
+ <port id="0" precision="FP16">
127
+ <dim>262144</dim>
128
+ </port>
129
+ </input>
130
+ <output>
131
+ <port id="1" precision="FP32">
132
+ <dim>262144</dim>
133
+ </port>
134
+ </output>
135
+ </layer>
136
+ <layer id="10" name="Multiply_81361" type="Multiply" version="opset1">
137
+ <data auto_broadcast="numpy" />
138
+ <input>
139
+ <port id="0" precision="FP32">
140
+ <dim>1</dim>
141
+ <dim>64</dim>
142
+ <dim>262144</dim>
143
+ </port>
144
+ <port id="1" precision="FP32">
145
+ <dim>262144</dim>
146
+ </port>
147
+ </input>
148
+ <output>
149
+ <port id="2" precision="FP32">
150
+ <dim>1</dim>
151
+ <dim>64</dim>
152
+ <dim>262144</dim>
153
+ </port>
154
+ </output>
155
+ </layer>
156
+ <layer id="11" name="Tanh_81362" type="Tanh" version="opset1">
157
+ <input>
158
+ <port id="0" precision="FP32">
159
+ <dim>1</dim>
160
+ <dim>64</dim>
161
+ <dim>262144</dim>
162
+ </port>
163
+ </input>
164
+ <output>
165
+ <port id="1" precision="FP32">
166
+ <dim>1</dim>
167
+ <dim>64</dim>
168
+ <dim>262144</dim>
169
+ </port>
170
+ </output>
171
+ </layer>
172
+ <layer id="12" name="Constant_81364_compressed" type="Const" version="opset1">
173
+ <data element_type="f16" shape="" offset="4" size="2" />
174
+ <output>
175
+ <port id="0" precision="FP16" />
176
+ </output>
177
+ </layer>
178
+ <layer id="13" name="Constant_81364" type="Convert" version="opset1">
179
+ <data destination_type="f32" />
180
+ <rt_info>
181
+ <attribute name="decompression" version="0" />
182
+ </rt_info>
183
+ <input>
184
+ <port id="0" precision="FP16" />
185
+ </input>
186
+ <output>
187
+ <port id="1" precision="FP32" />
188
+ </output>
189
+ </layer>
190
+ <layer id="14" name="Multiply_81365" type="Multiply" version="opset1">
191
+ <data auto_broadcast="numpy" />
192
+ <input>
193
+ <port id="0" precision="FP32">
194
+ <dim>1</dim>
195
+ <dim>64</dim>
196
+ <dim>262144</dim>
197
+ </port>
198
+ <port id="1" precision="FP32" />
199
+ </input>
200
+ <output>
201
+ <port id="2" precision="FP32">
202
+ <dim>1</dim>
203
+ <dim>64</dim>
204
+ <dim>262144</dim>
205
+ </port>
206
+ </output>
207
+ </layer>
208
+ <layer id="15" name="Multiply_81365.0" type="Convert" version="opset1">
209
+ <data destination_type="f16" />
210
+ <input>
211
+ <port id="0" precision="FP32">
212
+ <dim>1</dim>
213
+ <dim>64</dim>
214
+ <dim>262144</dim>
215
+ </port>
216
+ </input>
217
+ <output>
218
+ <port id="1" precision="FP16" names="logits">
219
+ <dim>1</dim>
220
+ <dim>64</dim>
221
+ <dim>262144</dim>
222
+ </port>
223
+ </output>
224
+ </layer>
225
+ <layer id="16" name="Result_81366" type="Result" version="opset1" output_names="logits">
226
+ <input>
227
+ <port id="0" precision="FP16">
228
+ <dim>1</dim>
229
+ <dim>64</dim>
230
+ <dim>262144</dim>
231
+ </port>
232
+ </input>
233
+ </layer>
234
+ </layers>
235
+ <edges>
236
+ <edge from-layer="0" from-port="0" to-layer="3" to-port="0" />
237
+ <edge from-layer="1" from-port="0" to-layer="6" to-port="0" />
238
+ <edge from-layer="2" from-port="0" to-layer="9" to-port="0" />
239
+ <edge from-layer="3" from-port="1" to-layer="5" to-port="0" />
240
+ <edge from-layer="4" from-port="0" to-layer="5" to-port="1" />
241
+ <edge from-layer="5" from-port="2" to-layer="8" to-port="0" />
242
+ <edge from-layer="6" from-port="1" to-layer="7" to-port="0" />
243
+ <edge from-layer="7" from-port="1" to-layer="8" to-port="1" />
244
+ <edge from-layer="8" from-port="2" to-layer="10" to-port="0" />
245
+ <edge from-layer="9" from-port="1" to-layer="10" to-port="1" />
246
+ <edge from-layer="10" from-port="2" to-layer="11" to-port="0" />
247
+ <edge from-layer="11" from-port="1" to-layer="14" to-port="0" />
248
+ <edge from-layer="12" from-port="0" to-layer="13" to-port="0" />
249
+ <edge from-layer="13" from-port="1" to-layer="14" to-port="1" />
250
+ <edge from-layer="14" from-port="2" to-layer="15" to-port="0" />
251
+ <edge from-layer="15" from-port="1" to-layer="16" to-port="0" />
252
+ </edges>
253
+ <rt_info>
254
+ <info name="OpenVINO Runtime" value="2026.5.0-23160-6db412d0a27" />
255
+ </rt_info>
256
+ </net>
shared.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:402bac867a19a87a5b21800b3a093bc56306b49e52a7bd5a3a9cdf471a7175c6
3
+ size 3492282368
shared.bin.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"tensors": {"embed_q": {"at": [0, 671088640], "type": "i8", "shape": [262144, 2560]}, "embed_s": {"at": [671088640, 524288], "type": "f16", "shape": [262144]}, "head_q": {"alias": "embed_q"}, "head_s": {"at": [671612928, 524288], "type": "f16", "shape": [262144]}, "ple_q": {"at": [672137216, 2818572288], "type": "i8", "shape": [262144, 10752]}, "ple_s": {"at": [3490709504, 524288], "type": "f16", "shape": [262144]}, "ple_map": {"at": [3491233792, 1048576], "type": "i32", "shape": [262144]}}}
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f
3
+ size 32169626
tokenizer_config.json ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "audio_token": "<|audio|>",
3
+ "backend": "tokenizers",
4
+ "boa_token": "<|audio>",
5
+ "boi_token": "<|image>",
6
+ "bos_token": "<bos>",
7
+ "eoa_token": "<audio|>",
8
+ "eoc_token": "<channel|>",
9
+ "eoi_token": "<image|>",
10
+ "eos_token": "<eos>",
11
+ "eot_token": "<turn|>",
12
+ "escape_token": "<|\"|>",
13
+ "etc_token": "<tool_call|>",
14
+ "etd_token": "<tool|>",
15
+ "etr_token": "<tool_response|>",
16
+ "extra_special_tokens": [
17
+ "<|video|>"
18
+ ],
19
+ "image_token": "<|image|>",
20
+ "is_local": false,
21
+ "mask_token": "<mask>",
22
+ "model_max_length": 1000000000000000019884624838656,
23
+ "model_specific_special_tokens": {
24
+ "audio_token": "<|audio|>",
25
+ "boa_token": "<|audio>",
26
+ "boi_token": "<|image>",
27
+ "eoa_token": "<audio|>",
28
+ "eoc_token": "<channel|>",
29
+ "eoi_token": "<image|>",
30
+ "eot_token": "<turn|>",
31
+ "escape_token": "<|\"|>",
32
+ "etc_token": "<tool_call|>",
33
+ "etd_token": "<tool|>",
34
+ "etr_token": "<tool_response|>",
35
+ "image_token": "<|image|>",
36
+ "soc_token": "<|channel>",
37
+ "sot_token": "<|turn>",
38
+ "stc_token": "<|tool_call>",
39
+ "std_token": "<|tool>",
40
+ "str_token": "<|tool_response>",
41
+ "think_token": "<|think|>"
42
+ },
43
+ "pad_token": "<pad>",
44
+ "padding_side": "left",
45
+ "processor_class": "Gemma4Processor",
46
+ "response_schema": {
47
+ "properties": {
48
+ "content": {
49
+ "type": "string"
50
+ },
51
+ "role": {
52
+ "const": "assistant"
53
+ },
54
+ "thinking": {
55
+ "type": "string"
56
+ },
57
+ "tool_calls": {
58
+ "items": {
59
+ "properties": {
60
+ "function": {
61
+ "properties": {
62
+ "arguments": {
63
+ "additionalProperties": {},
64
+ "type": "object",
65
+ "x-parser": "gemma4-tool-call"
66
+ },
67
+ "name": {
68
+ "type": "string"
69
+ }
70
+ },
71
+ "type": "object",
72
+ "x-regex": "call\\:(?P<name>\\w+)(?P<arguments>\\{.*\\})"
73
+ },
74
+ "type": {
75
+ "const": "function"
76
+ }
77
+ },
78
+ "type": "object"
79
+ },
80
+ "type": "array",
81
+ "x-regex-iterator": "<\\|tool_call>(.*?)<tool_call\\|>"
82
+ }
83
+ },
84
+ "type": "object",
85
+ "x-regex": "(\\<\\|channel\\>thought\\n(?P<thinking>.*?)\\<channel\\|\\>)?(?P<tool_calls>\\<\\|tool_call\\>.*\\<tool_call\\|\\>)?(?P<content>(?:(?!\\<turn\\|\\>)(?!\\<\\|tool_response\\>).)+)?(?:\\<turn\\|\\>|\\<\\|tool_response\\>)?"
86
+ },
87
+ "response_template": {
88
+ "defaults": {
89
+ "role": "assistant"
90
+ },
91
+ "fields": {
92
+ "content": {
93
+ "close": [
94
+ "<turn|>",
95
+ "<|tool_response>",
96
+ "<eos>"
97
+ ],
98
+ "content": "text"
99
+ },
100
+ "thinking": {
101
+ "close": "<channel|>",
102
+ "content": "text",
103
+ "open": "<|channel>thought\n"
104
+ },
105
+ "tool_calls": {
106
+ "close": "<tool_call|>",
107
+ "content": "json",
108
+ "content_args": {
109
+ "string_delims": [
110
+ [
111
+ "<|\"|>",
112
+ "<|\"|>"
113
+ ]
114
+ ],
115
+ "unquoted_keys": true
116
+ },
117
+ "open_pattern": "<\\|tool_call>call:(?P<name>\\w+)",
118
+ "repeats": true,
119
+ "transform": {
120
+ "function": {
121
+ "arguments": "{content}",
122
+ "name": "{name}"
123
+ },
124
+ "type": "function"
125
+ }
126
+ }
127
+ },
128
+ "start_anchor": [
129
+ "<|turn>model\n",
130
+ "<tool_response|>"
131
+ ]
132
+ },
133
+ "soc_token": "<|channel>",
134
+ "sot_token": "<|turn>",
135
+ "stc_token": "<|tool_call>",
136
+ "std_token": "<|tool>",
137
+ "str_token": "<|tool_response>",
138
+ "think_token": "<|think|>",
139
+ "tokenizer_class": "GemmaTokenizer",
140
+ "unk_token": "<unk>"
141
+ }
vision.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d7728ad030d97e7ab324ca25e39f587343fda470cf95db97c45fec3035d82980
3
+ size 169812500
vision.xml ADDED
The diff for this file is too large to render. See raw diff