HelloSun commited on
Commit
96e7386
·
verified ·
1 Parent(s): bb14a6b

Upload Qwen-Image-2.1 OpenVINO INT4 (NNCF weight-only, transformer+text_encoder 4bit)

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ processor/tokenizer.json filter=lfs diff=lfs merge=lfs -text
model_index.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Pipeline",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "_name_or_path": "Qwen/Qwen-Image-2.1",
5
+ "processor": [
6
+ "transformers",
7
+ "Qwen3VLProcessor"
8
+ ],
9
+ "scheduler": [
10
+ "diffusers",
11
+ "FlowMatchEulerDiscreteScheduler"
12
+ ],
13
+ "text_encoder": [
14
+ "transformers",
15
+ "Qwen3VLForConditionalGeneration"
16
+ ],
17
+ "transformer": [
18
+ "diffusers",
19
+ "QwenImage21Transformer2DModel"
20
+ ],
21
+ "vae": [
22
+ "diffusers",
23
+ "AutoencoderKLQwenImage21"
24
+ ]
25
+ }
openvino_config.json ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dtype": "int4_int4_int4",
3
+ "input_info": null,
4
+ "optimum_version": "2.3.0",
5
+ "output_attentions": false,
6
+ "quantization_config": {
7
+ "_dataset_kwargs": {},
8
+ "dataset": null,
9
+ "default_config": {
10
+ "quant_method": "default"
11
+ },
12
+ "ignored_scope": null,
13
+ "num_samples": null,
14
+ "processor": null,
15
+ "quantization_configs": {
16
+ "text_encoder": {
17
+ "_dataset_kwargs": {},
18
+ "all_layers": null,
19
+ "backup_precision": null,
20
+ "bits": 4,
21
+ "dataset": null,
22
+ "dq_group_size": null,
23
+ "dtype": "int4",
24
+ "gptq": null,
25
+ "group_size": 128,
26
+ "group_size_fallback": "adjust",
27
+ "ignored_scope": null,
28
+ "lora_correction": null,
29
+ "num_samples": null,
30
+ "processor": null,
31
+ "quant_method": "default",
32
+ "ratio": 1.0,
33
+ "scale_estimation": null,
34
+ "sensitivity_metric": null,
35
+ "statistics_path": null,
36
+ "sym": false,
37
+ "tokenizer": null
38
+ },
39
+ "text_encoder_i2i": {
40
+ "_dataset_kwargs": {},
41
+ "all_layers": null,
42
+ "backup_precision": null,
43
+ "bits": 4,
44
+ "dataset": null,
45
+ "dq_group_size": null,
46
+ "dtype": "int4",
47
+ "gptq": null,
48
+ "group_size": 128,
49
+ "group_size_fallback": "adjust",
50
+ "ignored_scope": null,
51
+ "lora_correction": null,
52
+ "num_samples": null,
53
+ "processor": null,
54
+ "quant_method": "default",
55
+ "ratio": 1.0,
56
+ "scale_estimation": null,
57
+ "sensitivity_metric": null,
58
+ "statistics_path": null,
59
+ "sym": false,
60
+ "tokenizer": null
61
+ },
62
+ "transformer": {
63
+ "_dataset_kwargs": {},
64
+ "all_layers": null,
65
+ "backup_precision": null,
66
+ "bits": 4,
67
+ "dataset": null,
68
+ "dq_group_size": null,
69
+ "dtype": "int4",
70
+ "gptq": null,
71
+ "group_size": 128,
72
+ "group_size_fallback": "adjust",
73
+ "ignored_scope": null,
74
+ "lora_correction": null,
75
+ "num_samples": null,
76
+ "processor": null,
77
+ "quant_method": "default",
78
+ "ratio": 1.0,
79
+ "scale_estimation": null,
80
+ "sensitivity_metric": null,
81
+ "statistics_path": null,
82
+ "sym": false,
83
+ "tokenizer": null
84
+ }
85
+ },
86
+ "tokenizer": null
87
+ },
88
+ "save_onnx_model": false,
89
+ "transformers_version": "5.10.4"
90
+ }
processor/chat_template.jinja ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0].role == 'system' %}
4
+ {%- if messages[0].content is string %}
5
+ {{- messages[0].content }}
6
+ {%- else %}
7
+ {%- for content in messages[0].content %}
8
+ {%- if 'text' in content %}
9
+ {{- content.text }}
10
+ {%- endif %}
11
+ {%- endfor %}
12
+ {%- endif %}
13
+ {{- '\n\n' }}
14
+ {%- endif %}
15
+ {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
16
+ {%- for tool in tools %}
17
+ {{- "\n" }}
18
+ {{- tool | tojson }}
19
+ {%- endfor %}
20
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
21
+ {%- else %}
22
+ {%- if messages[0].role == 'system' %}
23
+ {{- '<|im_start|>system\n' }}
24
+ {%- if messages[0].content is string %}
25
+ {{- messages[0].content }}
26
+ {%- else %}
27
+ {%- for content in messages[0].content %}
28
+ {%- if 'text' in content %}
29
+ {{- content.text }}
30
+ {%- endif %}
31
+ {%- endfor %}
32
+ {%- endif %}
33
+ {{- '<|im_end|>\n' }}
34
+ {%- endif %}
35
+ {%- endif %}
36
+ {%- set image_count = namespace(value=0) %}
37
+ {%- set video_count = namespace(value=0) %}
38
+ {%- for message in messages %}
39
+ {%- if message.role == "user" %}
40
+ {{- '<|im_start|>' + message.role + '\n' }}
41
+ {%- if message.content is string %}
42
+ {{- message.content }}
43
+ {%- else %}
44
+ {%- for content in message.content %}
45
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
46
+ {%- set image_count.value = image_count.value + 1 %}
47
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
48
+ <|vision_start|><|image_pad|><|vision_end|>
49
+ {%- elif content.type == 'video' or 'video' in content %}
50
+ {%- set video_count.value = video_count.value + 1 %}
51
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
52
+ <|vision_start|><|video_pad|><|vision_end|>
53
+ {%- elif 'text' in content %}
54
+ {{- content.text }}
55
+ {%- endif %}
56
+ {%- endfor %}
57
+ {%- endif %}
58
+ {{- '<|im_end|>\n' }}
59
+ {%- elif message.role == "assistant" %}
60
+ {{- '<|im_start|>' + message.role + '\n' }}
61
+ {%- if message.content is string %}
62
+ {{- message.content }}
63
+ {%- else %}
64
+ {%- for content_item in message.content %}
65
+ {%- if 'text' in content_item %}
66
+ {{- content_item.text }}
67
+ {%- endif %}
68
+ {%- endfor %}
69
+ {%- endif %}
70
+ {%- if message.tool_calls %}
71
+ {%- for tool_call in message.tool_calls %}
72
+ {%- if (loop.first and message.content) or (not loop.first) %}
73
+ {{- '\n' }}
74
+ {%- endif %}
75
+ {%- if tool_call.function %}
76
+ {%- set tool_call = tool_call.function %}
77
+ {%- endif %}
78
+ {{- '<tool_call>\n{"name": "' }}
79
+ {{- tool_call.name }}
80
+ {{- '", "arguments": ' }}
81
+ {%- if tool_call.arguments is string %}
82
+ {{- tool_call.arguments }}
83
+ {%- else %}
84
+ {{- tool_call.arguments | tojson }}
85
+ {%- endif %}
86
+ {{- '}\n</tool_call>' }}
87
+ {%- endfor %}
88
+ {%- endif %}
89
+ {{- '<|im_end|>\n' }}
90
+ {%- elif message.role == "tool" %}
91
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
92
+ {{- '<|im_start|>user' }}
93
+ {%- endif %}
94
+ {{- '\n<tool_response>\n' }}
95
+ {%- if message.content is string %}
96
+ {{- message.content }}
97
+ {%- else %}
98
+ {%- for content in message.content %}
99
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
100
+ {%- set image_count.value = image_count.value + 1 %}
101
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
102
+ <|vision_start|><|image_pad|><|vision_end|>
103
+ {%- elif content.type == 'video' or 'video' in content %}
104
+ {%- set video_count.value = video_count.value + 1 %}
105
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
106
+ <|vision_start|><|video_pad|><|vision_end|>
107
+ {%- elif 'text' in content %}
108
+ {{- content.text }}
109
+ {%- endif %}
110
+ {%- endfor %}
111
+ {%- endif %}
112
+ {{- '\n</tool_response>' }}
113
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
114
+ {{- '<|im_end|>\n' }}
115
+ {%- endif %}
116
+ {%- endif %}
117
+ {%- endfor %}
118
+ {%- if add_generation_prompt %}
119
+ {{- '<|im_start|>assistant\n' }}
120
+ {%- endif %}
processor/processor_config.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_processor": {
3
+ "data_format": "channels_first",
4
+ "default_to_square": true,
5
+ "do_convert_rgb": true,
6
+ "do_normalize": true,
7
+ "do_rescale": true,
8
+ "do_resize": true,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_processor_type": "Qwen2VLImageProcessor",
15
+ "image_std": [
16
+ 0.5,
17
+ 0.5,
18
+ 0.5
19
+ ],
20
+ "merge_size": 2,
21
+ "patch_size": 16,
22
+ "resample": 3,
23
+ "rescale_factor": 0.00392156862745098,
24
+ "size": {
25
+ "longest_edge": 16777216,
26
+ "shortest_edge": 65536
27
+ },
28
+ "temporal_patch_size": 2
29
+ },
30
+ "processor_class": "Qwen3VLProcessor",
31
+ "video_processor": {
32
+ "data_format": "channels_first",
33
+ "default_to_square": true,
34
+ "do_convert_rgb": true,
35
+ "do_normalize": true,
36
+ "do_rescale": true,
37
+ "do_resize": true,
38
+ "do_sample_frames": true,
39
+ "fps": 2,
40
+ "image_mean": [
41
+ 0.5,
42
+ 0.5,
43
+ 0.5
44
+ ],
45
+ "image_std": [
46
+ 0.5,
47
+ 0.5,
48
+ 0.5
49
+ ],
50
+ "max_frames": 768,
51
+ "merge_size": 2,
52
+ "min_frames": 4,
53
+ "patch_size": 16,
54
+ "resample": 3,
55
+ "rescale_factor": 0.00392156862745098,
56
+ "return_metadata": false,
57
+ "size": {
58
+ "longest_edge": 25165824,
59
+ "shortest_edge": 4096
60
+ },
61
+ "temporal_patch_size": 2,
62
+ "video_processor_type": "Qwen3VLVideoProcessor"
63
+ }
64
+ }
processor/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
3
+ size 11422650
processor/tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": true,
9
+ "local_files_only": false,
10
+ "model_max_length": 262144,
11
+ "pad_token": "<|endoftext|>",
12
+ "processor_class": "Qwen3VLProcessor",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
scheduler/scheduler_config.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "FlowMatchEulerDiscreteScheduler",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "base_image_seq_len": 256,
5
+ "base_shift": 0.5,
6
+ "invert_sigmas": false,
7
+ "max_image_seq_len": 8192,
8
+ "max_shift": 0.9,
9
+ "num_train_timesteps": 1000,
10
+ "shift": 1.0,
11
+ "shift_terminal": 0.02,
12
+ "stochastic_sampling": false,
13
+ "time_shift_type": "exponential",
14
+ "use_beta_sigmas": false,
15
+ "use_dynamic_shifting": true,
16
+ "use_exponential_sigmas": false,
17
+ "use_karras_sigmas": false
18
+ }
text_encoder/config.json ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "Qwen3VLForConditionalGeneration",
3
+ "_qwenimage21_image_token_id": 151655,
4
+ "attention_bias": false,
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "dtype": "float16",
8
+ "eos_token_id": 151645,
9
+ "head_dim": 128,
10
+ "hidden_act": "silu",
11
+ "hidden_size": 4096,
12
+ "initializer_range": 0.02,
13
+ "intermediate_size": 12288,
14
+ "max_position_embeddings": 262144,
15
+ "model_type": "qwen3_vl_text",
16
+ "num_attention_heads": 32,
17
+ "num_hidden_layers": 36,
18
+ "num_key_value_heads": 8,
19
+ "pad_token_id": null,
20
+ "rms_norm_eps": 1e-06,
21
+ "rope_parameters": {
22
+ "mrope_interleaved": true,
23
+ "mrope_section": [
24
+ 24,
25
+ 20,
26
+ 20
27
+ ],
28
+ "rope_theta": 5000000,
29
+ "rope_type": "default"
30
+ },
31
+ "transformers_version": "5.10.4",
32
+ "use_cache": true,
33
+ "vocab_size": 151936
34
+ }
text_encoder/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c663adbe7ca3fc034420160798dbf7dd263ce484f7011acefd18f0ef91af20d0
3
+ size 4256132893
text_encoder/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff
 
text_encoder_i2i/config.json ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "Qwen3VLForConditionalGeneration",
3
+ "_qwenimage21_image_token_id": 151655,
4
+ "attention_bias": false,
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "dtype": "float16",
8
+ "eos_token_id": 151645,
9
+ "head_dim": 128,
10
+ "hidden_act": "silu",
11
+ "hidden_size": 4096,
12
+ "initializer_range": 0.02,
13
+ "intermediate_size": 12288,
14
+ "max_position_embeddings": 262144,
15
+ "model_type": "qwen3_vl_text",
16
+ "num_attention_heads": 32,
17
+ "num_hidden_layers": 36,
18
+ "num_key_value_heads": 8,
19
+ "pad_token_id": null,
20
+ "rms_norm_eps": 1e-06,
21
+ "rope_parameters": {
22
+ "mrope_interleaved": true,
23
+ "mrope_section": [
24
+ 24,
25
+ 20,
26
+ 20
27
+ ],
28
+ "rope_theta": 5000000,
29
+ "rope_type": "default"
30
+ },
31
+ "transformers_version": "5.10.4",
32
+ "use_cache": true,
33
+ "vocab_size": 151936
34
+ }
text_encoder_i2i/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de2da7c6d5a190f9eb9140b5cc9776eeaabcba1e0159a37749455baf24619061
3
+ size 4256132733
text_encoder_i2i/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff
 
transformer/config.json ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Transformer2DModel",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "_name_or_path": "/root/.cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/790c92633540aa0cb11d9abf19eb46d861714758/transformer",
5
+ "attention_head_dim": 128,
6
+ "axes_dims_rope": [
7
+ 16,
8
+ 56,
9
+ 56
10
+ ],
11
+ "causal_condition": true,
12
+ "context_in_dim": 4096,
13
+ "eps": 1e-06,
14
+ "in_channels": 64,
15
+ "mlp_ratio": 3,
16
+ "num_attention_heads": 32,
17
+ "num_layers": 32,
18
+ "out_channels": 64,
19
+ "patch_size": 1
20
+ }
transformer/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e4cae4db8717b70d904dd67d802e70405093fec93f28c9704caf4917dfe12275
3
+ size 3696679634
transformer/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff
 
vae_decoder/config.json ADDED
@@ -0,0 +1,294 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "AutoencoderKLQwenImage21",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "_name_or_path": "/root/.cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/790c92633540aa0cb11d9abf19eb46d861714758/vae",
5
+ "attn_scales": [],
6
+ "base_dim": 96,
7
+ "decoder_base_dim": 144,
8
+ "dim_mult": [
9
+ 1,
10
+ 2,
11
+ 4,
12
+ 8,
13
+ 8
14
+ ],
15
+ "dropout": 0.0,
16
+ "in_channels": 4,
17
+ "is_residual": true,
18
+ "latents_mean": [
19
+ 0.5126,
20
+ 0.7721,
21
+ -0.0631,
22
+ 1.3506,
23
+ -0.7855,
24
+ -2.1025,
25
+ -0.3458,
26
+ 1.3722,
27
+ 1.8873,
28
+ -1.7177,
29
+ -0.651,
30
+ 0.2732,
31
+ 0.7562,
32
+ -0.6163,
33
+ -1.0277,
34
+ 3.8363,
35
+ 2.021,
36
+ 0.0472,
37
+ 0.932,
38
+ 2.0087,
39
+ 2.4954,
40
+ -0.1391,
41
+ -1.4249,
42
+ 1.8464,
43
+ -0.5236,
44
+ 1.2826,
45
+ 3.7046,
46
+ -1.3035,
47
+ 2.7286,
48
+ -1.4518,
49
+ -1.9036,
50
+ -1.9955,
51
+ -0.0342,
52
+ -1.0265,
53
+ -0.7636,
54
+ 3.0555,
55
+ 0.0746,
56
+ -3.0751,
57
+ -0.1076,
58
+ 1.7376,
59
+ -1.0914,
60
+ -1.9435,
61
+ -0.2784,
62
+ -1.368,
63
+ 0.4809,
64
+ -0.4433,
65
+ 0.3764,
66
+ 0.5729,
67
+ -2.0595,
68
+ 1.096,
69
+ -1.326,
70
+ -2.0211,
71
+ -5.0179,
72
+ 0.5275,
73
+ 4.0162,
74
+ 1.8505,
75
+ 0.3026,
76
+ 1.9373,
77
+ 1.4937,
78
+ 0.2632,
79
+ 0.5547,
80
+ -1.7121,
81
+ -0.1562,
82
+ 0.0304
83
+ ],
84
+ "latents_mean_data": [
85
+ 0.5126,
86
+ 0.7721,
87
+ -0.0631,
88
+ 1.3506,
89
+ -0.7855,
90
+ -2.1025,
91
+ -0.3458,
92
+ 1.3722,
93
+ 1.8873,
94
+ -1.7177,
95
+ -0.651,
96
+ 0.2732,
97
+ 0.7562,
98
+ -0.6163,
99
+ -1.0277,
100
+ 3.8363,
101
+ 2.021,
102
+ 0.0472,
103
+ 0.932,
104
+ 2.0087,
105
+ 2.4954,
106
+ -0.1391,
107
+ -1.4249,
108
+ 1.8464,
109
+ -0.5236,
110
+ 1.2826,
111
+ 3.7046,
112
+ -1.3035,
113
+ 2.7286,
114
+ -1.4518,
115
+ -1.9036,
116
+ -1.9955,
117
+ -0.0342,
118
+ -1.0265,
119
+ -0.7636,
120
+ 3.0555,
121
+ 0.0746,
122
+ -3.0751,
123
+ -0.1076,
124
+ 1.7376,
125
+ -1.0914,
126
+ -1.9435,
127
+ -0.2784,
128
+ -1.368,
129
+ 0.4809,
130
+ -0.4433,
131
+ 0.3764,
132
+ 0.5729,
133
+ -2.0595,
134
+ 1.096,
135
+ -1.326,
136
+ -2.0211,
137
+ -5.0179,
138
+ 0.5275,
139
+ 4.0162,
140
+ 1.8505,
141
+ 0.3026,
142
+ 1.9373,
143
+ 1.4937,
144
+ 0.2632,
145
+ 0.5547,
146
+ -1.7121,
147
+ -0.1562,
148
+ 0.0304
149
+ ],
150
+ "latents_std": [
151
+ 3.2001,
152
+ 3.2936,
153
+ 3.4321,
154
+ 3.0091,
155
+ 3.1061,
156
+ 4.0379,
157
+ 4.0705,
158
+ 3.791,
159
+ 3.0785,
160
+ 3.65,
161
+ 3.9308,
162
+ 3.0904,
163
+ 2.8778,
164
+ 3.7675,
165
+ 3.732,
166
+ 5.0756,
167
+ 3.2864,
168
+ 4.0397,
169
+ 3.1317,
170
+ 4.0443,
171
+ 2.9249,
172
+ 3.9454,
173
+ 3.0988,
174
+ 4.2489,
175
+ 3.4896,
176
+ 3.8513,
177
+ 3.9323,
178
+ 3.4719,
179
+ 3.7498,
180
+ 4.283,
181
+ 3.5694,
182
+ 4.2467,
183
+ 3.9037,
184
+ 3.2947,
185
+ 5.077,
186
+ 3.5075,
187
+ 3.27,
188
+ 3.4767,
189
+ 2.8063,
190
+ 5.1125,
191
+ 3.5327,
192
+ 4.7833,
193
+ 3.1286,
194
+ 4.1819,
195
+ 3.8527,
196
+ 3.8312,
197
+ 3.5605,
198
+ 4.3875,
199
+ 3.9624,
200
+ 4.0168,
201
+ 3.5643,
202
+ 4.055,
203
+ 5.5614,
204
+ 4.2963,
205
+ 4.408,
206
+ 3.4959,
207
+ 3.8747,
208
+ 3.7608,
209
+ 3.5735,
210
+ 3.149,
211
+ 3.7662,
212
+ 3.6746,
213
+ 3.4563,
214
+ 3.8161
215
+ ],
216
+ "latents_std_data": [
217
+ 3.2001,
218
+ 3.2936,
219
+ 3.4321,
220
+ 3.0091,
221
+ 3.1061,
222
+ 4.0379,
223
+ 4.0705,
224
+ 3.791,
225
+ 3.0785,
226
+ 3.65,
227
+ 3.9308,
228
+ 3.0904,
229
+ 2.8778,
230
+ 3.7675,
231
+ 3.732,
232
+ 5.0756,
233
+ 3.2864,
234
+ 4.0397,
235
+ 3.1317,
236
+ 4.0443,
237
+ 2.9249,
238
+ 3.9454,
239
+ 3.0988,
240
+ 4.2489,
241
+ 3.4896,
242
+ 3.8513,
243
+ 3.9323,
244
+ 3.4719,
245
+ 3.7498,
246
+ 4.283,
247
+ 3.5694,
248
+ 4.2467,
249
+ 3.9037,
250
+ 3.2947,
251
+ 5.077,
252
+ 3.5075,
253
+ 3.27,
254
+ 3.4767,
255
+ 2.8063,
256
+ 5.1125,
257
+ 3.5327,
258
+ 4.7833,
259
+ 3.1286,
260
+ 4.1819,
261
+ 3.8527,
262
+ 3.8312,
263
+ 3.5605,
264
+ 4.3875,
265
+ 3.9624,
266
+ 4.0168,
267
+ 3.5643,
268
+ 4.055,
269
+ 5.5614,
270
+ 4.2963,
271
+ 4.408,
272
+ 3.4959,
273
+ 3.8747,
274
+ 3.7608,
275
+ 3.5735,
276
+ 3.149,
277
+ 3.7662,
278
+ 3.6746,
279
+ 3.4563,
280
+ 3.8161
281
+ ],
282
+ "num_res_blocks": 2,
283
+ "out_channels": 4,
284
+ "patch_size": null,
285
+ "scale_factor_spatial": 16,
286
+ "scale_factor_temporal": 8,
287
+ "temperal_downsample": [
288
+ false,
289
+ true,
290
+ true,
291
+ true
292
+ ],
293
+ "z_dim": 64
294
+ }
vae_decoder/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e2df0173563a8fcfede0f5768f913836a00277d4be6bc0d24b601386d2db6eb
3
+ size 253235856
vae_decoder/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff
 
vae_encoder/config.json ADDED
@@ -0,0 +1,162 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "AutoencoderKLQwenImage21",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "_name_or_path": "/root/.cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/790c92633540aa0cb11d9abf19eb46d861714758/vae",
5
+ "attn_scales": [],
6
+ "base_dim": 96,
7
+ "decoder_base_dim": 144,
8
+ "dim_mult": [
9
+ 1,
10
+ 2,
11
+ 4,
12
+ 8,
13
+ 8
14
+ ],
15
+ "dropout": 0.0,
16
+ "in_channels": 4,
17
+ "is_residual": true,
18
+ "latents_mean": [
19
+ 0.5126,
20
+ 0.7721,
21
+ -0.0631,
22
+ 1.3506,
23
+ -0.7855,
24
+ -2.1025,
25
+ -0.3458,
26
+ 1.3722,
27
+ 1.8873,
28
+ -1.7177,
29
+ -0.651,
30
+ 0.2732,
31
+ 0.7562,
32
+ -0.6163,
33
+ -1.0277,
34
+ 3.8363,
35
+ 2.021,
36
+ 0.0472,
37
+ 0.932,
38
+ 2.0087,
39
+ 2.4954,
40
+ -0.1391,
41
+ -1.4249,
42
+ 1.8464,
43
+ -0.5236,
44
+ 1.2826,
45
+ 3.7046,
46
+ -1.3035,
47
+ 2.7286,
48
+ -1.4518,
49
+ -1.9036,
50
+ -1.9955,
51
+ -0.0342,
52
+ -1.0265,
53
+ -0.7636,
54
+ 3.0555,
55
+ 0.0746,
56
+ -3.0751,
57
+ -0.1076,
58
+ 1.7376,
59
+ -1.0914,
60
+ -1.9435,
61
+ -0.2784,
62
+ -1.368,
63
+ 0.4809,
64
+ -0.4433,
65
+ 0.3764,
66
+ 0.5729,
67
+ -2.0595,
68
+ 1.096,
69
+ -1.326,
70
+ -2.0211,
71
+ -5.0179,
72
+ 0.5275,
73
+ 4.0162,
74
+ 1.8505,
75
+ 0.3026,
76
+ 1.9373,
77
+ 1.4937,
78
+ 0.2632,
79
+ 0.5547,
80
+ -1.7121,
81
+ -0.1562,
82
+ 0.0304
83
+ ],
84
+ "latents_std": [
85
+ 3.2001,
86
+ 3.2936,
87
+ 3.4321,
88
+ 3.0091,
89
+ 3.1061,
90
+ 4.0379,
91
+ 4.0705,
92
+ 3.791,
93
+ 3.0785,
94
+ 3.65,
95
+ 3.9308,
96
+ 3.0904,
97
+ 2.8778,
98
+ 3.7675,
99
+ 3.732,
100
+ 5.0756,
101
+ 3.2864,
102
+ 4.0397,
103
+ 3.1317,
104
+ 4.0443,
105
+ 2.9249,
106
+ 3.9454,
107
+ 3.0988,
108
+ 4.2489,
109
+ 3.4896,
110
+ 3.8513,
111
+ 3.9323,
112
+ 3.4719,
113
+ 3.7498,
114
+ 4.283,
115
+ 3.5694,
116
+ 4.2467,
117
+ 3.9037,
118
+ 3.2947,
119
+ 5.077,
120
+ 3.5075,
121
+ 3.27,
122
+ 3.4767,
123
+ 2.8063,
124
+ 5.1125,
125
+ 3.5327,
126
+ 4.7833,
127
+ 3.1286,
128
+ 4.1819,
129
+ 3.8527,
130
+ 3.8312,
131
+ 3.5605,
132
+ 4.3875,
133
+ 3.9624,
134
+ 4.0168,
135
+ 3.5643,
136
+ 4.055,
137
+ 5.5614,
138
+ 4.2963,
139
+ 4.408,
140
+ 3.4959,
141
+ 3.8747,
142
+ 3.7608,
143
+ 3.5735,
144
+ 3.149,
145
+ 3.7662,
146
+ 3.6746,
147
+ 3.4563,
148
+ 3.8161
149
+ ],
150
+ "num_res_blocks": 2,
151
+ "out_channels": 4,
152
+ "patch_size": null,
153
+ "scale_factor_spatial": 16,
154
+ "scale_factor_temporal": 8,
155
+ "temperal_downsample": [
156
+ false,
157
+ true,
158
+ true,
159
+ true
160
+ ],
161
+ "z_dim": 64
162
+ }
vae_encoder/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:52006a17d0d6719c81fd694713aca11b906dbb47c38f9dc9050ff7a36c9e6f61
3
+ size 78002698
vae_encoder/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff
 
vision_encoder/config.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "Qwen3VLVisionModel",
3
+ "deepstack_visual_indexes": [
4
+ 8,
5
+ 16,
6
+ 24
7
+ ],
8
+ "depth": 27,
9
+ "dtype": "float16",
10
+ "hidden_act": "gelu_pytorch_tanh",
11
+ "hidden_size": 1152,
12
+ "in_channels": 3,
13
+ "initializer_range": 0.02,
14
+ "intermediate_size": 4304,
15
+ "model_type": "qwen3_vl_vision",
16
+ "num_heads": 16,
17
+ "num_position_embeddings": 2304,
18
+ "out_hidden_size": 4096,
19
+ "patch_size": 16,
20
+ "spatial_merge_size": 2,
21
+ "temporal_patch_size": 2,
22
+ "transformers_version": "5.10.4"
23
+ }
vision_encoder/openvino_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4dc10d421a4b1f32ce993887063fa4cbfd08c9b349ab5c753f500dcea7688fa3
3
+ size 577780620
vision_encoder/openvino_model.xml ADDED
The diff for this file is too large to render. See raw diff