File size: 923 Bytes
b51cd45
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
name: fp8_dynamic_moe
scheme: FP8_DYNAMIC  # FP8 weights, FP8 dynamic per-token activations — data-free, vLLM-native
engine: llmcompressor

# MoE variant of fp8_dynamic.yaml for qwen3_5_moe bases (Ornith-1.0-35B, Qwen3.6-35B-A3B).
# FP8_DYNAMIC needs no calibration data; this tiny placeholder is ignored by DataFreePipeline.
calibration:
  dataset: neuralmagic/calibration
  config: LLM
  split: train
  num_samples: 4
  max_seq_length: 512

ignore:
  - lm_head
  - "re:.*visual.*"
  - "re:.*linear_attn.*"            # Mamba/SSM block stays in bf16 (landmine 5)
  - "re:.*mtp.*"                    # no-op on Ornith (base ships 0 mtp.*), harmless
  - "re:.*mlp.gate$"               # MoE router — keep bf16 (MoE-only, landmine 24)
  - "re:.*mlp.shared_expert_gate$" # shared-expert gate — keep bf16
  # The 256 routed mlp.experts.N.* and the mlp.shared_expert.* FFNs ARE quantised.

export:
  save_compressed: true