Download quantize_int4.py from HelloSun/Qwen-Image-2.1-OpenVINO-INT4: direct link, hf CLI and curl.
- Browser
- Download file 4.97 kB
-
https://huggingface.co/HelloSun/Qwen-Image-2.1-OpenVINO-INT4/resolve/main/quantize_int4.py
- Command line
-
hf download hf://HelloSun/Qwen-Image-2.1-OpenVINO-INT4/quantize_int4.py
-
curl -L -o quantize_int4.py https://huggingface.co/HelloSun/Qwen-Image-2.1-OpenVINO-INT4/resolve/main/quantize_int4.py
4.97 kB
| """Qwen-Image-2.1 -> OpenVINO FP16 export + NNCF weight-only INT4 quantization. | |
| NOTE on versions (deviation from original pinned env, approved by user): | |
| Qwen/Qwen-Image-2.1 requires QwenImage21Pipeline (added in diffusers | |
| commit 6256aa766 "Add Qwen-Image 2.1 (#14804)", post-v0.40.0). Pinned | |
| diffusers==0.37.1 cannot load it (AttributeError). Therefore: | |
| - diffusers: 0.41.0.dev0 (git main, includes QwenImage21) | |
| - transformers: 5.10.4 (pulled by optimum-intel git; supports hub 1.x) | |
| - huggingface-hub: 1.33.0 (required by git diffusers >=1.31) | |
| - tokenizers: 0.22.2, optimum: 2.3.0 | |
| - optimum-intel: 2.3.0.dev0+git (main, first version with QwenImage21 support) | |
| - openvino==2026.4.0 nncf==3.4.0 torch pillow psutil unchanged | |
| - transformers dependency table patched: hub cap <1.0 -> <2.0 | |
| - optimum-intel modeling_visual_language.py patched for transformers>=5 | |
| (VisionRotaryEmbedding alias + rot_pos_emb try/except, diffusion unused) | |
| FP16 export CLI (same pattern as task example): | |
| optimum-cli export openvino -m Qwen/Qwen-Image-2.1 \ | |
| --task text-to-image --library diffusers --weight-format fp16 \ | |
| ./qwen-image-2.1-ov-fp16 | |
| INT4 quantization: | |
| OVQuantizer + OVPipelineQuantizationConfig nested in OVConfig: | |
| transformer + text_encoder (+ text_encoder_i2i, same Qwen3VL arch) with | |
| OVWeightQuantizationConfig(bits=4, sym=False, group_size=128, | |
| group_size_fallback="adjust", ratio=1.0), | |
| rest default INT8. Passed as ov_config=OVConfig(quantization_config=...). | |
| """ | |
| import os | |
| import time | |
| import json | |
| FP16_DIR = "./qwen-image-2.1-ov-fp16" | |
| INT4_DIR = "./qwen-image-2.1-ov-int4" | |
| MODEL_ID = "Qwen/Qwen-Image-2.1" | |
| def export_fp16(): | |
| if os.path.exists(FP16_DIR) and os.listdir(FP16_DIR): | |
| print(f"FP16 model already exists at {FP16_DIR}, skipping export...") | |
| print("CLI: optimum-cli export openvino -m " | |
| f"{MODEL_ID} --task text-to-image --library diffusers " | |
| f"--weight-format fp16 {FP16_DIR}") | |
| return FP16_DIR, 0.0 | |
| # Export is done via optimum-cli (see docstring); this fallback uses API. | |
| from optimum.intel import OVDiffusionPipeline | |
| print(f"Exporting {MODEL_ID} to FP16 OpenVINO...") | |
| start = time.time() | |
| pipeline = OVDiffusionPipeline.from_pretrained( | |
| MODEL_ID, | |
| export=True, | |
| compile=False, | |
| weight_format="fp16", | |
| token=os.environ.get("HF_TOKEN"), | |
| ) | |
| pipeline.save_pretrained(FP16_DIR) | |
| dt = time.time() - start | |
| print(f"FP16 export completed in {dt:.2f}s") | |
| return FP16_DIR, dt | |
| def quantize_int4(fp16_dir): | |
| from optimum.intel import OVDiffusionPipeline | |
| from optimum.intel.openvino import ( | |
| OVQuantizer, | |
| OVConfig, | |
| OVPipelineQuantizationConfig, | |
| OVWeightQuantizationConfig, | |
| ) | |
| print(f"Quantizing {fp16_dir} to INT4...") | |
| start = time.time() | |
| pipeline = OVDiffusionPipeline.from_pretrained(fp16_dir, compile=False) | |
| print("OV submodels:", pipeline._ov_model_names) | |
| quantizer = OVQuantizer.from_pretrained(pipeline) | |
| int4_cfg = OVWeightQuantizationConfig( | |
| bits=4, | |
| sym=False, | |
| group_size=128, | |
| group_size_fallback="adjust", | |
| ratio=1.0, | |
| ) | |
| int8_default = OVWeightQuantizationConfig() # bits=8 default INT8 | |
| ov_config = OVConfig( | |
| quantization_config=OVPipelineQuantizationConfig( | |
| quantization_configs={ | |
| "transformer": int4_cfg, | |
| "text_encoder": int4_cfg, | |
| "text_encoder_i2i": int4_cfg, | |
| }, | |
| default_config=int8_default, | |
| ) | |
| ) | |
| quantizer.quantize(save_directory=INT4_DIR, ov_config=ov_config) | |
| dt = time.time() - start | |
| print(f"INT4 quantization completed in {dt:.2f}s") | |
| return INT4_DIR, dt | |
| def main(): | |
| print("=" * 60) | |
| print("Qwen-Image-2.1 OpenVINO INT4 Quantization") | |
| print("=" * 60) | |
| os.makedirs(OUTPUT_DIR, exist_ok=True) | |
| fp16_dir, export_time = export_fp16() | |
| int4_dir, quantize_time = quantize_int4(fp16_dir) | |
| data = { | |
| "model_id": MODEL_ID, | |
| "fp16_export_dir": fp16_dir, | |
| "int4_dir": int4_dir, | |
| "export_time_seconds": export_time, | |
| "quantize_time_seconds": quantize_time, | |
| "total_time_seconds": export_time + quantize_time, | |
| "quant_config": { | |
| "transformer": "OVWeightQuantizationConfig(bits=4,sym=False,group_size=128,group_size_fallback=adjust,ratio=1.0)", | |
| "text_encoder": "same INT4", | |
| "text_encoder_i2i": "same INT4 (Qwen-Image-2.1 editing text encoder, same arch)", | |
| "others": "default INT8 (OVWeightQuantizationConfig bits=8)", | |
| "ov_config_class": "OVPipelineQuantizationConfig nested in OVConfig", | |
| }, | |
| } | |
| with open(os.path.join(OUTPUT_DIR, "benchmark_quantization.json"), "w") as f: | |
| json.dump(data, f, indent=2) | |
| print(f"\nDone. FP16: {fp16_dir} INT4: {int4_dir}") | |
| if __name__ == "__main__": | |
| main() | |