File size: 1,100 Bytes
37e7e5e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
"""Quark recipe matching AMD's published Qwen3.8-27B AWQ configuration."""
from quark.torch.quantization.config.config import (
    AWQConfig,
    Int4PerGroupSpec,
    QConfig,
    QLayerConfig,
)


def make_config():
    return QConfig(
        global_quant_config=QLayerConfig(
            weight=Int4PerGroupSpec(ch_axis=-1, group_size=128).to_quantization_spec()
        ),
        exclude=["model.visual.*", "lm_head", "mtp.*"],
        algo_config=[
            AWQConfig(
                model_decoder_layers="model.language_model.layers",
                scaling_layers=[
                    {
                        "prev_op": "post_attention_layernorm",
                        "layers": ["mlp.gate_proj", "mlp.up_proj"],
                        "inp": "mlp.gate_proj",
                        "module2inspect": "mlp",
                    },
                    {
                        "prev_op": "mlp.up_proj",
                        "layers": ["mlp.down_proj"],
                        "inp": "mlp.down_proj",
                    },
                ],
            )
        ],
    )