default_stage: default_modifiers: SmoothQuantModifier: smoothing_strength: 0.5 mappings: - !!python/tuple - ['re:.*q_proj', 're:.*k_proj', 're:.*v_proj'] - re:.*input_layernorm - !!python/tuple - ['re:.*gate_proj', 're:.*up_proj'] - re:.*post_attention_layernorm ignore: [] algorithm: smoothquant SpinQuantModifier: rotations: [R1, R2, R4] transform_type: hadamard randomize: false learnable: false precision: torch.float64 transform_block_size: 128 transform_config: config_groups: R1: type: hadamard apply: - targets: ['re:.*embed_tokens$', 're:.*o_proj$', 're:.*down_proj$'] location: weight_output inverse: false ignore: [] - targets: ['re:.*q_proj$', 're:.*k_proj$', 're:.*v_proj$', 're:.*up_proj$', 're:.*gate_proj$', lm_head] location: weight_input inverse: true ignore: [] randomize: false requires_grad: false head_dim: 128 precision: torch.float64 R2: type: hadamard apply: - targets: ['re:.*v_proj$'] location: weight_output inverse: false ignore: [] - targets: ['re:.*o_proj$'] location: weight_input inverse: true ignore: [] randomize: false requires_grad: false head_dim: 128 precision: torch.float64 R4: type: hadamard apply: - targets: ['re:.*down_proj$'] location: input inverse: false ignore: [] - targets: ['re:.*down_proj$'] location: weight_input inverse: true ignore: [] randomize: false requires_grad: false head_dim: 128 precision: torch.float64 GPTQModifier: config_groups: group_0: targets: [Linear] weights: num_bits: 8 type: int symmetric: true group_size: null strategy: tensor block_structure: null dynamic: false actorder: !!python/object/apply:compressed_tensors.quantization.quant_args.ActivationOrdering [ static] scale_dtype: null zp_dtype: null observer: memoryless_minmax observer_kwargs: {} input_activations: num_bits: 8 type: int symmetric: false group_size: null strategy: tensor block_structure: null dynamic: false actorder: null scale_dtype: null zp_dtype: torch.int8 observer: memoryless_minmax observer_kwargs: {} output_activations: null format: null targets: [Linear] ignore: [lm_head] bypass_divisibility_checks: false block_size: 128 dampening_frac: 0.01 actorder: static offload_hessians: false