Add quantize_int4.py
Browse files- quantize_int4.py +130 -0
quantize_int4.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Qwen-Image-2.1 -> OpenVINO FP16 export + NNCF weight-only INT4 quantization.
|
| 2 |
+
|
| 3 |
+
NOTE on versions (deviation from original pinned env, approved by user):
|
| 4 |
+
Qwen/Qwen-Image-2.1 requires QwenImage21Pipeline (added in diffusers
|
| 5 |
+
commit 6256aa766 "Add Qwen-Image 2.1 (#14804)", post-v0.40.0). Pinned
|
| 6 |
+
diffusers==0.37.1 cannot load it (AttributeError). Therefore:
|
| 7 |
+
- diffusers: 0.41.0.dev0 (git main, includes QwenImage21)
|
| 8 |
+
- transformers: 5.10.4 (pulled by optimum-intel git; supports hub 1.x)
|
| 9 |
+
- huggingface-hub: 1.33.0 (required by git diffusers >=1.31)
|
| 10 |
+
- tokenizers: 0.22.2, optimum: 2.3.0
|
| 11 |
+
- optimum-intel: 2.3.0.dev0+git (main, first version with QwenImage21 support)
|
| 12 |
+
- openvino==2026.4.0 nncf==3.4.0 torch pillow psutil unchanged
|
| 13 |
+
- transformers dependency table patched: hub cap <1.0 -> <2.0
|
| 14 |
+
- optimum-intel modeling_visual_language.py patched for transformers>=5
|
| 15 |
+
(VisionRotaryEmbedding alias + rot_pos_emb try/except, diffusion unused)
|
| 16 |
+
|
| 17 |
+
FP16 export CLI (same pattern as task example):
|
| 18 |
+
optimum-cli export openvino -m Qwen/Qwen-Image-2.1 \
|
| 19 |
+
--task text-to-image --library diffusers --weight-format fp16 \
|
| 20 |
+
/home/user/app/qwen-image-2.1-ov-fp16
|
| 21 |
+
|
| 22 |
+
INT4 quantization:
|
| 23 |
+
OVQuantizer + OVPipelineQuantizationConfig nested in OVConfig:
|
| 24 |
+
transformer + text_encoder (+ text_encoder_i2i, same Qwen3VL arch) with
|
| 25 |
+
OVWeightQuantizationConfig(bits=4, sym=False, group_size=128,
|
| 26 |
+
group_size_fallback="adjust", ratio=1.0),
|
| 27 |
+
rest default INT8. Passed as ov_config=OVConfig(quantization_config=...).
|
| 28 |
+
"""
|
| 29 |
+
import os
|
| 30 |
+
import time
|
| 31 |
+
import json
|
| 32 |
+
|
| 33 |
+
FP16_DIR = "/home/user/app/qwen-image-2.1-ov-fp16"
|
| 34 |
+
INT4_DIR = "/home/user/app/qwen-image-2.1-ov-int4"
|
| 35 |
+
MODEL_ID = "Qwen/Qwen-Image-2.1"
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
def export_fp16():
|
| 39 |
+
if os.path.exists(FP16_DIR) and os.listdir(FP16_DIR):
|
| 40 |
+
print(f"FP16 model already exists at {FP16_DIR}, skipping export...")
|
| 41 |
+
print("CLI: optimum-cli export openvino -m "
|
| 42 |
+
f"{MODEL_ID} --task text-to-image --library diffusers "
|
| 43 |
+
f"--weight-format fp16 {FP16_DIR}")
|
| 44 |
+
return FP16_DIR, 0.0
|
| 45 |
+
# Export is done via optimum-cli (see docstring); this fallback uses API.
|
| 46 |
+
from optimum.intel import OVDiffusionPipeline
|
| 47 |
+
print(f"Exporting {MODEL_ID} to FP16 OpenVINO...")
|
| 48 |
+
start = time.time()
|
| 49 |
+
pipeline = OVDiffusionPipeline.from_pretrained(
|
| 50 |
+
MODEL_ID,
|
| 51 |
+
export=True,
|
| 52 |
+
compile=False,
|
| 53 |
+
weight_format="fp16",
|
| 54 |
+
token=os.environ.get("HF_TOKEN"),
|
| 55 |
+
)
|
| 56 |
+
pipeline.save_pretrained(FP16_DIR)
|
| 57 |
+
dt = time.time() - start
|
| 58 |
+
print(f"FP16 export completed in {dt:.2f}s")
|
| 59 |
+
return FP16_DIR, dt
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def quantize_int4(fp16_dir):
|
| 63 |
+
from optimum.intel import OVDiffusionPipeline
|
| 64 |
+
from optimum.intel.openvino import (
|
| 65 |
+
OVQuantizer,
|
| 66 |
+
OVConfig,
|
| 67 |
+
OVPipelineQuantizationConfig,
|
| 68 |
+
OVWeightQuantizationConfig,
|
| 69 |
+
)
|
| 70 |
+
|
| 71 |
+
print(f"Quantizing {fp16_dir} to INT4...")
|
| 72 |
+
start = time.time()
|
| 73 |
+
pipeline = OVDiffusionPipeline.from_pretrained(fp16_dir, compile=False)
|
| 74 |
+
print("OV submodels:", pipeline._ov_model_names)
|
| 75 |
+
quantizer = OVQuantizer.from_pretrained(pipeline)
|
| 76 |
+
|
| 77 |
+
int4_cfg = OVWeightQuantizationConfig(
|
| 78 |
+
bits=4,
|
| 79 |
+
sym=False,
|
| 80 |
+
group_size=128,
|
| 81 |
+
group_size_fallback="adjust",
|
| 82 |
+
ratio=1.0,
|
| 83 |
+
)
|
| 84 |
+
int8_default = OVWeightQuantizationConfig() # bits=8 default INT8
|
| 85 |
+
|
| 86 |
+
ov_config = OVConfig(
|
| 87 |
+
quantization_config=OVPipelineQuantizationConfig(
|
| 88 |
+
quantization_configs={
|
| 89 |
+
"transformer": int4_cfg,
|
| 90 |
+
"text_encoder": int4_cfg,
|
| 91 |
+
"text_encoder_i2i": int4_cfg,
|
| 92 |
+
},
|
| 93 |
+
default_config=int8_default,
|
| 94 |
+
)
|
| 95 |
+
)
|
| 96 |
+
quantizer.quantize(save_directory=INT4_DIR, ov_config=ov_config)
|
| 97 |
+
dt = time.time() - start
|
| 98 |
+
print(f"INT4 quantization completed in {dt:.2f}s")
|
| 99 |
+
return INT4_DIR, dt
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
def main():
|
| 103 |
+
print("=" * 60)
|
| 104 |
+
print("Qwen-Image-2.1 OpenVINO INT4 Quantization")
|
| 105 |
+
print("=" * 60)
|
| 106 |
+
os.makedirs("/home/user/app/outputs", exist_ok=True)
|
| 107 |
+
fp16_dir, export_time = export_fp16()
|
| 108 |
+
int4_dir, quantize_time = quantize_int4(fp16_dir)
|
| 109 |
+
data = {
|
| 110 |
+
"model_id": MODEL_ID,
|
| 111 |
+
"fp16_export_dir": fp16_dir,
|
| 112 |
+
"int4_dir": int4_dir,
|
| 113 |
+
"export_time_seconds": export_time,
|
| 114 |
+
"quantize_time_seconds": quantize_time,
|
| 115 |
+
"total_time_seconds": export_time + quantize_time,
|
| 116 |
+
"quant_config": {
|
| 117 |
+
"transformer": "OVWeightQuantizationConfig(bits=4,sym=False,group_size=128,group_size_fallback=adjust,ratio=1.0)",
|
| 118 |
+
"text_encoder": "same INT4",
|
| 119 |
+
"text_encoder_i2i": "same INT4 (Qwen-Image-2.1 editing text encoder, same arch)",
|
| 120 |
+
"others": "default INT8 (OVWeightQuantizationConfig bits=8)",
|
| 121 |
+
"ov_config_class": "OVPipelineQuantizationConfig nested in OVConfig",
|
| 122 |
+
},
|
| 123 |
+
}
|
| 124 |
+
with open("/home/user/app/outputs/benchmark_quantization.json", "w") as f:
|
| 125 |
+
json.dump(data, f, indent=2)
|
| 126 |
+
print(f"\nDone. FP16: {fp16_dir} INT4: {int4_dir}")
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
if __name__ == "__main__":
|
| 130 |
+
main()
|