name: fp8_dynamic_moe scheme: FP8_DYNAMIC # FP8 weights, FP8 dynamic per-token activations — data-free, vLLM-native engine: llmcompressor # MoE variant of fp8_dynamic.yaml for qwen3_5_moe bases (Ornith-1.0-35B, Qwen3.6-35B-A3B). # FP8_DYNAMIC needs no calibration data; this tiny placeholder is ignored by DataFreePipeline. calibration: dataset: neuralmagic/calibration config: LLM split: train num_samples: 4 max_seq_length: 512 ignore: - lm_head - "re:.*visual.*" - "re:.*linear_attn.*" # Mamba/SSM block stays in bf16 (landmine 5) - "re:.*mtp.*" # no-op on Ornith (base ships 0 mtp.*), harmless - "re:.*mlp.gate$" # MoE router — keep bf16 (MoE-only, landmine 24) - "re:.*mlp.shared_expert_gate$" # shared-expert gate — keep bf16 # The 256 routed mlp.experts.N.* and the mlp.shared_expert.* FFNs ARE quantised. export: save_compressed: true