{ "bits": 4, "group_size": 128, "desc_act": false, "sym": false, "lm_head": false, "quant_method": "gptq", "checkpoint_format": "gptq", "pack_dtype": "int32", "meta": { "quantizer": [ "gptqmodel:5.7.0-dev" ], "uri": "https://github.com/modelcloud/gptqmodel", "damp_percent": 0.1, "damp_auto_increment": 0.01, "static_groups": false, "true_sequential": true, "mse": 0.0, "gptaq": false, "gptaq_alpha": 0.25, "act_group_aware": true, "failsafe": { "strategy": "rtn", "threshold": "0.5%", "smooth": { "type": "mad", "group_size_threshold": 128, "k": 2.75 } }, "gptaq_memory_device": "auto", "offload_to_disk": true, "offload_to_disk_path": "./gptqmodel_offload/hqdpgrum-rkaakpxx/", "pack_impl": "cpu", "mock_quantization": false, "hessian_chunk_size": null, "hessian_chunk_bytes": null, "hessian_use_bfloat16_staging": false, "vram_strategy": "exclusive" }, "format": "gptq" }