File size: 923 Bytes
b51cd45 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 | name: fp8_dynamic_moe
scheme: FP8_DYNAMIC # FP8 weights, FP8 dynamic per-token activations — data-free, vLLM-native
engine: llmcompressor
# MoE variant of fp8_dynamic.yaml for qwen3_5_moe bases (Ornith-1.0-35B, Qwen3.6-35B-A3B).
# FP8_DYNAMIC needs no calibration data; this tiny placeholder is ignored by DataFreePipeline.
calibration:
dataset: neuralmagic/calibration
config: LLM
split: train
num_samples: 4
max_seq_length: 512
ignore:
- lm_head
- "re:.*visual.*"
- "re:.*linear_attn.*" # Mamba/SSM block stays in bf16 (landmine 5)
- "re:.*mtp.*" # no-op on Ornith (base ships 0 mtp.*), harmless
- "re:.*mlp.gate$" # MoE router — keep bf16 (MoE-only, landmine 24)
- "re:.*mlp.shared_expert_gate$" # shared-expert gate — keep bf16
# The 256 routed mlp.experts.N.* and the mlp.shared_expert.* FFNs ARE quantised.
export:
save_compressed: true
|