liamwh commited on
Commit
f6d2344
·
verified ·
1 Parent(s): efc6bf3

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: swift-open-license-1.0
4
+ library_name: transformers
5
+ pipeline_tag: image-text-to-text
6
+ base_model: ukisai/Swift-Qwen3.8-27b
7
+ tags:
8
+ - qwen
9
+ - qwen3.5
10
+ - qwen3.8
11
+ - finetune
12
+ - awq
13
+ - int4
14
+ - w4a16
15
+ - compressed-tensors
16
+ - lmdeploy
17
+ ---
18
+
19
+ # Swift-Qwen3.8-27B-W4A16-AWQ
20
+
21
+ W4A16 (4-bit weights, 16-bit activations) AWQ compressed-tensors quantization
22
+ of [`ukisai/Swift-Qwen3.8-27b`](https://huggingface.co/ukisai/Swift-Qwen3.8-27b).
23
+
24
+ ## Quantization method
25
+
26
+ - **Scheme:** `W4A16_ASYM` — 4-bit asymmetric per-group quantization (group size 128) of all `Linear` weights, stored in the compressed-tensors pack-quantized format (`weight_packed` / `weight_scale` / `weight_zero_point` / `weight_shape`), which LMDeploy `turbomind` auto-detects and loads natively (including the MTP head and vision tower, which stay BF16).
27
+ - **Tooling:** [llmcompressor](https://github.com/vllm-project/llmcompressor) one-shot offline quantization with CPU offloading (`compressed_tensors.offload.load_offloaded_model`), so the full-precision source fits on a 2×16 GB VRAM setup.
28
+ - **AWQ activation smoothing:** `AWQModifier` with the layer-scoped hybrid-attention mappings from `build_hybrid_attention_mappings` — full-attention `input_layernorm` → `self_attn.q/k/v`, `post_attention_layernorm` → `mlp.gate/up`, and `mlp.up_proj` → `mlp.down_proj`, with `duo_scaling="both"` and CPU offload, followed by W4A16 quantization. This layer-scoped recipe is required for hybrid-attention (Qwen3.5-family) architectures — grouped-regex smoothing or mismatched mappings corrupt decoding.
29
+ - Unquantized (kept BF16): embeddings, `lm_head`, norms, `linear_attn.in_proj_a/b`, the vision tower, and the MTP head.
30
+ - **Run command:**
31
+
32
+ ```bash
33
+ CUDA_VISIBLE_DEVICES=1,2 python3 quantize-awq-hybrid.py \
34
+ --model_path ./Swift-Qwen3.8-27b \
35
+ --quant_path ./Swift-Qwen3.8-27b-W4A16-AWQ \
36
+ --offload_dir ./Swift-Qwen3.8-27b-W4A16-AWQ
37
+ ```
38
+
39
+ (Default calibration: UltraChat 200k `train_sft`.)
40
+
41
+ ## Recommended parameters
42
+
43
+ From the [base model](https://huggingface.co/ukisai/Swift-Qwen3.8-27b): temperature 1.0, top_p 0.95, top_k 20, min_p 0, presence_penalty 0, repetition_penalty 1.0.
44
+
45
+ ## Usage
46
+
47
+ Tested with LMDeploy `turbomind`:
48
+
49
+ ```bash
50
+ from lmdeploy import pipeline, TurbomindEngineConfig
51
+
52
+ pipe = pipeline(
53
+ "TheUnderscore/Swift-Qwen3.8-27b-W4A16-AWQ",
54
+ backend_config=TurbomindEngineConfig(
55
+ tp=2,
56
+ model_format="compressed-tensors",
57
+ language_model_only=True,
58
+ ),
59
+ )
60
+ print(pipe("Hello, who are you?").text)
61
+ ```
62
+
63
+ ## License and access
64
+
65
+ Swift weights are distributed through gated access under the **Swift Open License v1.0**;
66
+ this quantization inherits that license from the base model. Personal, research,
67
+ educational, evaluation, and commercial use are free for individuals and organizations
68
+ with annual recurring revenue, including affiliates, of up to US$1,000,000. Above that
69
+ threshold, commercial use requires a separate **Swift Enterprise License**. Contact
70
+ [UkisAI](https://ukisai.com/contact) for terms.
71
+
72
+ ## Files
73
+
74
+ - `quantize-awq-hybrid.py` — the script used to produce this quantization (CPU-offloaded, DDP/torchrun, produces properly numbered `-of-N` shards). `--offload_dir` selects where per-rank CPU offload temp folders live (defaults to the current working directory).
75
+ - `model-nonquant.safetensors` — unquantized tensors (`mtp.*` and `model.visual.*`) preserved BF16 so the full model architecture is loadable.
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,546 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "image_token_id": 248056,
6
+ "language_model_only": false,
7
+ "model_type": "qwen3_5",
8
+ "text_config": {
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "attn_output_gate": true,
12
+ "bos_token_id": 248044,
13
+ "dtype": "bfloat16",
14
+ "eos_token_id": 248044,
15
+ "full_attention_interval": 4,
16
+ "head_dim": 256,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 5120,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 17408,
21
+ "layer_types": [
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention",
42
+ "linear_attention",
43
+ "linear_attention",
44
+ "linear_attention",
45
+ "full_attention",
46
+ "linear_attention",
47
+ "linear_attention",
48
+ "linear_attention",
49
+ "full_attention",
50
+ "linear_attention",
51
+ "linear_attention",
52
+ "linear_attention",
53
+ "full_attention",
54
+ "linear_attention",
55
+ "linear_attention",
56
+ "linear_attention",
57
+ "full_attention",
58
+ "linear_attention",
59
+ "linear_attention",
60
+ "linear_attention",
61
+ "full_attention",
62
+ "linear_attention",
63
+ "linear_attention",
64
+ "linear_attention",
65
+ "full_attention",
66
+ "linear_attention",
67
+ "linear_attention",
68
+ "linear_attention",
69
+ "full_attention",
70
+ "linear_attention",
71
+ "linear_attention",
72
+ "linear_attention",
73
+ "full_attention",
74
+ "linear_attention",
75
+ "linear_attention",
76
+ "linear_attention",
77
+ "full_attention",
78
+ "linear_attention",
79
+ "linear_attention",
80
+ "linear_attention",
81
+ "full_attention",
82
+ "linear_attention",
83
+ "linear_attention",
84
+ "linear_attention",
85
+ "full_attention"
86
+ ],
87
+ "linear_conv_kernel_dim": 4,
88
+ "linear_key_head_dim": 128,
89
+ "linear_num_key_heads": 16,
90
+ "linear_num_value_heads": 48,
91
+ "linear_value_head_dim": 128,
92
+ "mamba_ssm_dtype": "float32",
93
+ "max_position_embeddings": 262144,
94
+ "model_type": "qwen3_5_text",
95
+ "mtp_num_hidden_layers": 1,
96
+ "mtp_use_dedicated_embeddings": false,
97
+ "num_attention_heads": 24,
98
+ "num_hidden_layers": 64,
99
+ "num_key_value_heads": 4,
100
+ "output_gate_type": "swish",
101
+ "pad_token_id": null,
102
+ "partial_rotary_factor": 0.25,
103
+ "rms_norm_eps": 1e-06,
104
+ "rope_parameters": {
105
+ "mrope_interleaved": true,
106
+ "mrope_section": [
107
+ 11,
108
+ 11,
109
+ 10
110
+ ],
111
+ "partial_rotary_factor": 0.25,
112
+ "rope_theta": 10000000,
113
+ "rope_type": "default"
114
+ },
115
+ "tie_word_embeddings": false,
116
+ "use_cache": true,
117
+ "vocab_size": 248320
118
+ },
119
+ "tie_word_embeddings": false,
120
+ "transformers_version": "5.8.0.dev0",
121
+ "video_token_id": 248057,
122
+ "vision_config": {
123
+ "deepstack_visual_indexes": [],
124
+ "depth": 27,
125
+ "hidden_act": "gelu_pytorch_tanh",
126
+ "hidden_size": 1152,
127
+ "in_channels": 3,
128
+ "initializer_range": 0.02,
129
+ "intermediate_size": 4304,
130
+ "model_type": "qwen3_5",
131
+ "num_heads": 16,
132
+ "num_position_embeddings": 2304,
133
+ "out_hidden_size": 5120,
134
+ "patch_size": 16,
135
+ "spatial_merge_size": 2,
136
+ "temporal_patch_size": 2,
137
+ "dtype": "bfloat16"
138
+ },
139
+ "vision_end_token_id": 248054,
140
+ "vision_start_token_id": 248053,
141
+ "quantization_config": {
142
+ "config_groups": {
143
+ "group_0": {
144
+ "format": "pack-quantized",
145
+ "input_activations": null,
146
+ "output_activations": null,
147
+ "targets": [
148
+ "Linear"
149
+ ],
150
+ "weights": {
151
+ "actorder": null,
152
+ "block_structure": null,
153
+ "dynamic": false,
154
+ "group_size": 128,
155
+ "num_bits": 4,
156
+ "observer": "memoryless_minmax",
157
+ "observer_kwargs": {},
158
+ "scale_dtype": null,
159
+ "strategy": "group",
160
+ "symmetric": false,
161
+ "type": "int",
162
+ "zp_dtype": "torch.int8"
163
+ }
164
+ },
165
+ "group_1": {
166
+ "format": "pack-quantized",
167
+ "input_activations": null,
168
+ "output_activations": null,
169
+ "targets": [
170
+ "re:.*lm_head$"
171
+ ],
172
+ "weights": {
173
+ "actorder": null,
174
+ "block_structure": null,
175
+ "dynamic": false,
176
+ "group_size": 128,
177
+ "num_bits": 4,
178
+ "observer": "memoryless_minmax",
179
+ "observer_kwargs": {},
180
+ "scale_dtype": null,
181
+ "strategy": "group",
182
+ "symmetric": true,
183
+ "type": "int",
184
+ "zp_dtype": null
185
+ }
186
+ },
187
+ "group_2": {
188
+ "format": "pack-quantized",
189
+ "input_activations": null,
190
+ "output_activations": null,
191
+ "targets": [
192
+ "re:.*embed_tokens$"
193
+ ],
194
+ "weights": {
195
+ "actorder": null,
196
+ "block_structure": null,
197
+ "dynamic": false,
198
+ "group_size": 128,
199
+ "num_bits": 8,
200
+ "observer": "memoryless_minmax",
201
+ "observer_kwargs": {},
202
+ "scale_dtype": null,
203
+ "strategy": "group",
204
+ "symmetric": true,
205
+ "type": "int",
206
+ "zp_dtype": null
207
+ }
208
+ },
209
+ "group_3": {
210
+ "format": "pack-quantized",
211
+ "input_activations": null,
212
+ "output_activations": null,
213
+ "targets": [
214
+ "re:^mtp\\..*"
215
+ ],
216
+ "weights": {
217
+ "actorder": null,
218
+ "block_structure": null,
219
+ "dynamic": false,
220
+ "group_size": 128,
221
+ "num_bits": 4,
222
+ "observer": "memoryless_minmax",
223
+ "observer_kwargs": {},
224
+ "scale_dtype": null,
225
+ "strategy": "group",
226
+ "symmetric": true,
227
+ "type": "int",
228
+ "zp_dtype": null
229
+ }
230
+ }
231
+ },
232
+ "format": "pack-quantized",
233
+ "global_compression_ratio": null,
234
+ "ignore": [
235
+ "model.visual.blocks.0.attn.qkv",
236
+ "model.visual.blocks.0.attn.proj",
237
+ "model.visual.blocks.0.mlp.linear_fc1",
238
+ "model.visual.blocks.0.mlp.linear_fc2",
239
+ "model.visual.blocks.1.attn.qkv",
240
+ "model.visual.blocks.1.attn.proj",
241
+ "model.visual.blocks.1.mlp.linear_fc1",
242
+ "model.visual.blocks.1.mlp.linear_fc2",
243
+ "model.visual.blocks.2.attn.qkv",
244
+ "model.visual.blocks.2.attn.proj",
245
+ "model.visual.blocks.2.mlp.linear_fc1",
246
+ "model.visual.blocks.2.mlp.linear_fc2",
247
+ "model.visual.blocks.3.attn.qkv",
248
+ "model.visual.blocks.3.attn.proj",
249
+ "model.visual.blocks.3.mlp.linear_fc1",
250
+ "model.visual.blocks.3.mlp.linear_fc2",
251
+ "model.visual.blocks.4.attn.qkv",
252
+ "model.visual.blocks.4.attn.proj",
253
+ "model.visual.blocks.4.mlp.linear_fc1",
254
+ "model.visual.blocks.4.mlp.linear_fc2",
255
+ "model.visual.blocks.5.attn.qkv",
256
+ "model.visual.blocks.5.attn.proj",
257
+ "model.visual.blocks.5.mlp.linear_fc1",
258
+ "model.visual.blocks.5.mlp.linear_fc2",
259
+ "model.visual.blocks.6.attn.qkv",
260
+ "model.visual.blocks.6.attn.proj",
261
+ "model.visual.blocks.6.mlp.linear_fc1",
262
+ "model.visual.blocks.6.mlp.linear_fc2",
263
+ "model.visual.blocks.7.attn.qkv",
264
+ "model.visual.blocks.7.attn.proj",
265
+ "model.visual.blocks.7.mlp.linear_fc1",
266
+ "model.visual.blocks.7.mlp.linear_fc2",
267
+ "model.visual.blocks.8.attn.qkv",
268
+ "model.visual.blocks.8.attn.proj",
269
+ "model.visual.blocks.8.mlp.linear_fc1",
270
+ "model.visual.blocks.8.mlp.linear_fc2",
271
+ "model.visual.blocks.9.attn.qkv",
272
+ "model.visual.blocks.9.attn.proj",
273
+ "model.visual.blocks.9.mlp.linear_fc1",
274
+ "model.visual.blocks.9.mlp.linear_fc2",
275
+ "model.visual.blocks.10.attn.qkv",
276
+ "model.visual.blocks.10.attn.proj",
277
+ "model.visual.blocks.10.mlp.linear_fc1",
278
+ "model.visual.blocks.10.mlp.linear_fc2",
279
+ "model.visual.blocks.11.attn.qkv",
280
+ "model.visual.blocks.11.attn.proj",
281
+ "model.visual.blocks.11.mlp.linear_fc1",
282
+ "model.visual.blocks.11.mlp.linear_fc2",
283
+ "model.visual.blocks.12.attn.qkv",
284
+ "model.visual.blocks.12.attn.proj",
285
+ "model.visual.blocks.12.mlp.linear_fc1",
286
+ "model.visual.blocks.12.mlp.linear_fc2",
287
+ "model.visual.blocks.13.attn.qkv",
288
+ "model.visual.blocks.13.attn.proj",
289
+ "model.visual.blocks.13.mlp.linear_fc1",
290
+ "model.visual.blocks.13.mlp.linear_fc2",
291
+ "model.visual.blocks.14.attn.qkv",
292
+ "model.visual.blocks.14.attn.proj",
293
+ "model.visual.blocks.14.mlp.linear_fc1",
294
+ "model.visual.blocks.14.mlp.linear_fc2",
295
+ "model.visual.blocks.15.attn.qkv",
296
+ "model.visual.blocks.15.attn.proj",
297
+ "model.visual.blocks.15.mlp.linear_fc1",
298
+ "model.visual.blocks.15.mlp.linear_fc2",
299
+ "model.visual.blocks.16.attn.qkv",
300
+ "model.visual.blocks.16.attn.proj",
301
+ "model.visual.blocks.16.mlp.linear_fc1",
302
+ "model.visual.blocks.16.mlp.linear_fc2",
303
+ "model.visual.blocks.17.attn.qkv",
304
+ "model.visual.blocks.17.attn.proj",
305
+ "model.visual.blocks.17.mlp.linear_fc1",
306
+ "model.visual.blocks.17.mlp.linear_fc2",
307
+ "model.visual.blocks.18.attn.qkv",
308
+ "model.visual.blocks.18.attn.proj",
309
+ "model.visual.blocks.18.mlp.linear_fc1",
310
+ "model.visual.blocks.18.mlp.linear_fc2",
311
+ "model.visual.blocks.19.attn.qkv",
312
+ "model.visual.blocks.19.attn.proj",
313
+ "model.visual.blocks.19.mlp.linear_fc1",
314
+ "model.visual.blocks.19.mlp.linear_fc2",
315
+ "model.visual.blocks.20.attn.qkv",
316
+ "model.visual.blocks.20.attn.proj",
317
+ "model.visual.blocks.20.mlp.linear_fc1",
318
+ "model.visual.blocks.20.mlp.linear_fc2",
319
+ "model.visual.blocks.21.attn.qkv",
320
+ "model.visual.blocks.21.attn.proj",
321
+ "model.visual.blocks.21.mlp.linear_fc1",
322
+ "model.visual.blocks.21.mlp.linear_fc2",
323
+ "model.visual.blocks.22.attn.qkv",
324
+ "model.visual.blocks.22.attn.proj",
325
+ "model.visual.blocks.22.mlp.linear_fc1",
326
+ "model.visual.blocks.22.mlp.linear_fc2",
327
+ "model.visual.blocks.23.attn.qkv",
328
+ "model.visual.blocks.23.attn.proj",
329
+ "model.visual.blocks.23.mlp.linear_fc1",
330
+ "model.visual.blocks.23.mlp.linear_fc2",
331
+ "model.visual.blocks.24.attn.qkv",
332
+ "model.visual.blocks.24.attn.proj",
333
+ "model.visual.blocks.24.mlp.linear_fc1",
334
+ "model.visual.blocks.24.mlp.linear_fc2",
335
+ "model.visual.blocks.25.attn.qkv",
336
+ "model.visual.blocks.25.attn.proj",
337
+ "model.visual.blocks.25.mlp.linear_fc1",
338
+ "model.visual.blocks.25.mlp.linear_fc2",
339
+ "model.visual.blocks.26.attn.qkv",
340
+ "model.visual.blocks.26.attn.proj",
341
+ "model.visual.blocks.26.mlp.linear_fc1",
342
+ "model.visual.blocks.26.mlp.linear_fc2",
343
+ "model.visual.merger.linear_fc1",
344
+ "model.visual.merger.linear_fc2",
345
+ "model.language_model.layers.0.linear_attn",
346
+ "model.language_model.layers.0.linear_attn.norm",
347
+ "model.language_model.layers.0.linear_attn.in_proj_b",
348
+ "model.language_model.layers.0.linear_attn.in_proj_a",
349
+ "model.language_model.layers.1.linear_attn",
350
+ "model.language_model.layers.1.linear_attn.norm",
351
+ "model.language_model.layers.1.linear_attn.in_proj_b",
352
+ "model.language_model.layers.1.linear_attn.in_proj_a",
353
+ "model.language_model.layers.2.linear_attn",
354
+ "model.language_model.layers.2.linear_attn.norm",
355
+ "model.language_model.layers.2.linear_attn.in_proj_b",
356
+ "model.language_model.layers.2.linear_attn.in_proj_a",
357
+ "model.language_model.layers.4.linear_attn",
358
+ "model.language_model.layers.4.linear_attn.norm",
359
+ "model.language_model.layers.4.linear_attn.in_proj_b",
360
+ "model.language_model.layers.4.linear_attn.in_proj_a",
361
+ "model.language_model.layers.5.linear_attn",
362
+ "model.language_model.layers.5.linear_attn.norm",
363
+ "model.language_model.layers.5.linear_attn.in_proj_b",
364
+ "model.language_model.layers.5.linear_attn.in_proj_a",
365
+ "model.language_model.layers.6.linear_attn",
366
+ "model.language_model.layers.6.linear_attn.norm",
367
+ "model.language_model.layers.6.linear_attn.in_proj_b",
368
+ "model.language_model.layers.6.linear_attn.in_proj_a",
369
+ "model.language_model.layers.8.linear_attn",
370
+ "model.language_model.layers.8.linear_attn.norm",
371
+ "model.language_model.layers.8.linear_attn.in_proj_b",
372
+ "model.language_model.layers.8.linear_attn.in_proj_a",
373
+ "model.language_model.layers.9.linear_attn",
374
+ "model.language_model.layers.9.linear_attn.norm",
375
+ "model.language_model.layers.9.linear_attn.in_proj_b",
376
+ "model.language_model.layers.9.linear_attn.in_proj_a",
377
+ "model.language_model.layers.10.linear_attn",
378
+ "model.language_model.layers.10.linear_attn.norm",
379
+ "model.language_model.layers.10.linear_attn.in_proj_b",
380
+ "model.language_model.layers.10.linear_attn.in_proj_a",
381
+ "model.language_model.layers.12.linear_attn",
382
+ "model.language_model.layers.12.linear_attn.norm",
383
+ "model.language_model.layers.12.linear_attn.in_proj_b",
384
+ "model.language_model.layers.12.linear_attn.in_proj_a",
385
+ "model.language_model.layers.13.linear_attn",
386
+ "model.language_model.layers.13.linear_attn.norm",
387
+ "model.language_model.layers.13.linear_attn.in_proj_b",
388
+ "model.language_model.layers.13.linear_attn.in_proj_a",
389
+ "model.language_model.layers.14.linear_attn",
390
+ "model.language_model.layers.14.linear_attn.norm",
391
+ "model.language_model.layers.14.linear_attn.in_proj_b",
392
+ "model.language_model.layers.14.linear_attn.in_proj_a",
393
+ "model.language_model.layers.16.linear_attn",
394
+ "model.language_model.layers.16.linear_attn.norm",
395
+ "model.language_model.layers.16.linear_attn.in_proj_b",
396
+ "model.language_model.layers.16.linear_attn.in_proj_a",
397
+ "model.language_model.layers.17.linear_attn",
398
+ "model.language_model.layers.17.linear_attn.norm",
399
+ "model.language_model.layers.17.linear_attn.in_proj_b",
400
+ "model.language_model.layers.17.linear_attn.in_proj_a",
401
+ "model.language_model.layers.18.linear_attn",
402
+ "model.language_model.layers.18.linear_attn.norm",
403
+ "model.language_model.layers.18.linear_attn.in_proj_b",
404
+ "model.language_model.layers.18.linear_attn.in_proj_a",
405
+ "model.language_model.layers.20.linear_attn",
406
+ "model.language_model.layers.20.linear_attn.norm",
407
+ "model.language_model.layers.20.linear_attn.in_proj_b",
408
+ "model.language_model.layers.20.linear_attn.in_proj_a",
409
+ "model.language_model.layers.21.linear_attn",
410
+ "model.language_model.layers.21.linear_attn.norm",
411
+ "model.language_model.layers.21.linear_attn.in_proj_b",
412
+ "model.language_model.layers.21.linear_attn.in_proj_a",
413
+ "model.language_model.layers.22.linear_attn",
414
+ "model.language_model.layers.22.linear_attn.norm",
415
+ "model.language_model.layers.22.linear_attn.in_proj_b",
416
+ "model.language_model.layers.22.linear_attn.in_proj_a",
417
+ "model.language_model.layers.24.linear_attn",
418
+ "model.language_model.layers.24.linear_attn.norm",
419
+ "model.language_model.layers.24.linear_attn.in_proj_b",
420
+ "model.language_model.layers.24.linear_attn.in_proj_a",
421
+ "model.language_model.layers.25.linear_attn",
422
+ "model.language_model.layers.25.linear_attn.norm",
423
+ "model.language_model.layers.25.linear_attn.in_proj_b",
424
+ "model.language_model.layers.25.linear_attn.in_proj_a",
425
+ "model.language_model.layers.26.linear_attn",
426
+ "model.language_model.layers.26.linear_attn.norm",
427
+ "model.language_model.layers.26.linear_attn.in_proj_b",
428
+ "model.language_model.layers.26.linear_attn.in_proj_a",
429
+ "model.language_model.layers.28.linear_attn",
430
+ "model.language_model.layers.28.linear_attn.norm",
431
+ "model.language_model.layers.28.linear_attn.in_proj_b",
432
+ "model.language_model.layers.28.linear_attn.in_proj_a",
433
+ "model.language_model.layers.29.linear_attn",
434
+ "model.language_model.layers.29.linear_attn.norm",
435
+ "model.language_model.layers.29.linear_attn.in_proj_b",
436
+ "model.language_model.layers.29.linear_attn.in_proj_a",
437
+ "model.language_model.layers.30.linear_attn",
438
+ "model.language_model.layers.30.linear_attn.norm",
439
+ "model.language_model.layers.30.linear_attn.in_proj_b",
440
+ "model.language_model.layers.30.linear_attn.in_proj_a",
441
+ "model.language_model.layers.32.linear_attn",
442
+ "model.language_model.layers.32.linear_attn.norm",
443
+ "model.language_model.layers.32.linear_attn.in_proj_b",
444
+ "model.language_model.layers.32.linear_attn.in_proj_a",
445
+ "model.language_model.layers.33.linear_attn",
446
+ "model.language_model.layers.33.linear_attn.norm",
447
+ "model.language_model.layers.33.linear_attn.in_proj_b",
448
+ "model.language_model.layers.33.linear_attn.in_proj_a",
449
+ "model.language_model.layers.34.linear_attn",
450
+ "model.language_model.layers.34.linear_attn.norm",
451
+ "model.language_model.layers.34.linear_attn.in_proj_b",
452
+ "model.language_model.layers.34.linear_attn.in_proj_a",
453
+ "model.language_model.layers.36.linear_attn",
454
+ "model.language_model.layers.36.linear_attn.norm",
455
+ "model.language_model.layers.36.linear_attn.in_proj_b",
456
+ "model.language_model.layers.36.linear_attn.in_proj_a",
457
+ "model.language_model.layers.37.linear_attn",
458
+ "model.language_model.layers.37.linear_attn.norm",
459
+ "model.language_model.layers.37.linear_attn.in_proj_b",
460
+ "model.language_model.layers.37.linear_attn.in_proj_a",
461
+ "model.language_model.layers.38.linear_attn",
462
+ "model.language_model.layers.38.linear_attn.norm",
463
+ "model.language_model.layers.38.linear_attn.in_proj_b",
464
+ "model.language_model.layers.38.linear_attn.in_proj_a",
465
+ "model.language_model.layers.40.linear_attn",
466
+ "model.language_model.layers.40.linear_attn.norm",
467
+ "model.language_model.layers.40.linear_attn.in_proj_b",
468
+ "model.language_model.layers.40.linear_attn.in_proj_a",
469
+ "model.language_model.layers.41.linear_attn",
470
+ "model.language_model.layers.41.linear_attn.norm",
471
+ "model.language_model.layers.41.linear_attn.in_proj_b",
472
+ "model.language_model.layers.41.linear_attn.in_proj_a",
473
+ "model.language_model.layers.42.linear_attn",
474
+ "model.language_model.layers.42.linear_attn.norm",
475
+ "model.language_model.layers.42.linear_attn.in_proj_b",
476
+ "model.language_model.layers.42.linear_attn.in_proj_a",
477
+ "model.language_model.layers.44.linear_attn",
478
+ "model.language_model.layers.44.linear_attn.norm",
479
+ "model.language_model.layers.44.linear_attn.in_proj_b",
480
+ "model.language_model.layers.44.linear_attn.in_proj_a",
481
+ "model.language_model.layers.45.linear_attn",
482
+ "model.language_model.layers.45.linear_attn.norm",
483
+ "model.language_model.layers.45.linear_attn.in_proj_b",
484
+ "model.language_model.layers.45.linear_attn.in_proj_a",
485
+ "model.language_model.layers.46.linear_attn",
486
+ "model.language_model.layers.46.linear_attn.norm",
487
+ "model.language_model.layers.46.linear_attn.in_proj_b",
488
+ "model.language_model.layers.46.linear_attn.in_proj_a",
489
+ "model.language_model.layers.48.linear_attn",
490
+ "model.language_model.layers.48.linear_attn.norm",
491
+ "model.language_model.layers.48.linear_attn.in_proj_b",
492
+ "model.language_model.layers.48.linear_attn.in_proj_a",
493
+ "model.language_model.layers.49.linear_attn",
494
+ "model.language_model.layers.49.linear_attn.norm",
495
+ "model.language_model.layers.49.linear_attn.in_proj_b",
496
+ "model.language_model.layers.49.linear_attn.in_proj_a",
497
+ "model.language_model.layers.50.linear_attn",
498
+ "model.language_model.layers.50.linear_attn.norm",
499
+ "model.language_model.layers.50.linear_attn.in_proj_b",
500
+ "model.language_model.layers.50.linear_attn.in_proj_a",
501
+ "model.language_model.layers.52.linear_attn",
502
+ "model.language_model.layers.52.linear_attn.norm",
503
+ "model.language_model.layers.52.linear_attn.in_proj_b",
504
+ "model.language_model.layers.52.linear_attn.in_proj_a",
505
+ "model.language_model.layers.53.linear_attn",
506
+ "model.language_model.layers.53.linear_attn.norm",
507
+ "model.language_model.layers.53.linear_attn.in_proj_b",
508
+ "model.language_model.layers.53.linear_attn.in_proj_a",
509
+ "model.language_model.layers.54.linear_attn",
510
+ "model.language_model.layers.54.linear_attn.norm",
511
+ "model.language_model.layers.54.linear_attn.in_proj_b",
512
+ "model.language_model.layers.54.linear_attn.in_proj_a",
513
+ "model.language_model.layers.56.linear_attn",
514
+ "model.language_model.layers.56.linear_attn.norm",
515
+ "model.language_model.layers.56.linear_attn.in_proj_b",
516
+ "model.language_model.layers.56.linear_attn.in_proj_a",
517
+ "model.language_model.layers.57.linear_attn",
518
+ "model.language_model.layers.57.linear_attn.norm",
519
+ "model.language_model.layers.57.linear_attn.in_proj_b",
520
+ "model.language_model.layers.57.linear_attn.in_proj_a",
521
+ "model.language_model.layers.58.linear_attn",
522
+ "model.language_model.layers.58.linear_attn.norm",
523
+ "model.language_model.layers.58.linear_attn.in_proj_b",
524
+ "model.language_model.layers.58.linear_attn.in_proj_a",
525
+ "model.language_model.layers.60.linear_attn",
526
+ "model.language_model.layers.60.linear_attn.norm",
527
+ "model.language_model.layers.60.linear_attn.in_proj_b",
528
+ "model.language_model.layers.60.linear_attn.in_proj_a",
529
+ "model.language_model.layers.61.linear_attn",
530
+ "model.language_model.layers.61.linear_attn.norm",
531
+ "model.language_model.layers.61.linear_attn.in_proj_b",
532
+ "model.language_model.layers.61.linear_attn.in_proj_a",
533
+ "model.language_model.layers.62.linear_attn",
534
+ "model.language_model.layers.62.linear_attn.norm",
535
+ "model.language_model.layers.62.linear_attn.in_proj_b",
536
+ "model.language_model.layers.62.linear_attn.in_proj_a"
537
+ ],
538
+ "kv_cache_scheme": null,
539
+ "quant_method": "compressed-tensors",
540
+ "quantization_status": "compressed",
541
+ "sparsity_config": {},
542
+ "transform_config": {},
543
+ "version": "0.18.0"
544
+ },
545
+ "dtype": "bfloat16"
546
+ }
generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 248044,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 248046,
6
+ 248044
7
+ ],
8
+ "pad_token_id": 248044,
9
+ "temperature": 1.0,
10
+ "top_k": 20,
11
+ "top_p": 0.95,
12
+ "min_p": 0,
13
+ "repetition_penalty": 1.0
14
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model-00001-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6872c54bf590710482a55f2a4fc7648a55242e819785ea4f64a91064ee933ba4
3
+ size 2212761608
model-00002-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a44afcd2154f6bac8141a8af48ba7d4788584544a088da6a3112085d695acbfe
3
+ size 2186190664
model-00003-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:07b0a5737261377859be9cb4fa19b0a0dbadb5b158bea0179367da9953de947b
3
+ size 2179699752
model-00004-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bf91ce93aa363b88392fbe32c4f785222de13c06e376768ed56ea60330a9b67e
3
+ size 2179699760
model-00005-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c3deb171898650ccc806f4a2e21c1aee5cea8efdce8fe149ab5f0bfc9049a83f
3
+ size 2179699760
model-00006-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:007e4ccf91e5310b19a335be197342daffbe5bad46fc92dc4a0d2d37abb1c5b4
3
+ size 2186211856
model-00007-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c29cde77d7a301e2434ab5f2bdffb65b04a61f9cfe02bc78d876ea5cac264632
3
+ size 2435435024
model-nonquant.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6036476ba28d3ab8696cc4ae469626d130eadf0f5abb233fbefd930b84d2a5f
3
+ size 52936
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
model_extra_tensors.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:006764c783b8aa3b1dda3b0de8eb10259c0ca1c18347ca149268884eb5b016cc
3
+ size 287295744
mtp_draft_vocab_ids.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8e19e199cdce572ee11d013f92e2c18adf2f0603aa285bb9a8ef585fc842a3f
3
+ size 208701
preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 16777216,
4
+ "shortest_edge": 65536
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "image_processor_type": "Qwen2VLImageProcessorFast"
21
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer_config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": true,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "unk_token": null,
29
+ "video_token": "<|video_pad|>",
30
+ "vision_bos_token": "<|vision_start|>",
31
+ "vision_eos_token": "<|vision_end|>"
32
+ }
video_preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 25165824,
4
+ "shortest_edge": 4096
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "video_processor_type": "Qwen3VLVideoProcessor"
21
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff