FabioTrindade's picture
Upload folder using huggingface_hub
c55f5b8 verified
Raw History Blame Contribute Delete
3.27 kB
default_stage:
default_modifiers:
SmoothQuantModifier:
smoothing_strength: 0.5
mappings:
- !!python/tuple
- ['re:.*q_proj', 're:.*k_proj', 're:.*v_proj']
- re:.*input_layernorm
- !!python/tuple
- ['re:.*gate_proj', 're:.*up_proj']
- re:.*post_attention_layernorm
ignore: []
algorithm: smoothquant
SpinQuantModifier:
rotations: [R1, R2, R4]
transform_type: hadamard
randomize: false
learnable: false
precision: torch.float64
transform_block_size: 128
transform_config:
config_groups:
R1:
type: hadamard
apply:
- targets: ['re:.*embed_tokens$', 're:.*o_proj$', 're:.*down_proj$']
location: weight_output
inverse: false
ignore: []
- targets: ['re:.*q_proj$', 're:.*k_proj$', 're:.*v_proj$', 're:.*up_proj$', 're:.*gate_proj$',
lm_head]
location: weight_input
inverse: true
ignore: []
randomize: false
requires_grad: false
head_dim: 128
precision: torch.float64
R2:
type: hadamard
apply:
- targets: ['re:.*v_proj$']
location: weight_output
inverse: false
ignore: []
- targets: ['re:.*o_proj$']
location: weight_input
inverse: true
ignore: []
randomize: false
requires_grad: false
head_dim: 128
precision: torch.float64
R4:
type: hadamard
apply:
- targets: ['re:.*down_proj$']
location: input
inverse: false
ignore: []
- targets: ['re:.*down_proj$']
location: weight_input
inverse: true
ignore: []
randomize: false
requires_grad: false
head_dim: 128
precision: torch.float64
GPTQModifier:
config_groups:
group_0:
targets: [Linear]
weights:
num_bits: 8
type: int
symmetric: true
group_size: null
strategy: tensor
block_structure: null
dynamic: false
actorder: !!python/object/apply:compressed_tensors.quantization.quant_args.ActivationOrdering [
static]
scale_dtype: null
zp_dtype: null
observer: memoryless_minmax
observer_kwargs: {}
input_activations:
num_bits: 8
type: int
symmetric: false
group_size: null
strategy: tensor
block_structure: null
dynamic: false
actorder: null
scale_dtype: null
zp_dtype: torch.int8
observer: memoryless_minmax
observer_kwargs: {}
output_activations: null
format: null
targets: [Linear]
ignore: [lm_head]
bypass_divisibility_checks: false
block_size: 128
dampening_frac: 0.01
actorder: static
offload_hessians: false