Qwen3.8-27B-heretic-ara-DFlash2-fp8 / fp8_profile_ab_comparison.json
magiccodingman's picture
Upload folder using huggingface_hub
b6bbe8a verified
Raw History Blame Contribute Delete
1.09 kB
{
"decision": "select mlp-only profile",
"reason": "Preserving attention output projections costs approximately 105 MB but improves proposal KL, top-1 agreement, and hidden-state fidelity.",
"profiles": {
"mlp_plus_attention_output_fp8": {
"tensor_payload_bytes": 2407368960,
"quantized_matrices": 20,
"proposal_kl_mean": 0.0026533013419426316,
"proposal_top1_agreement": 0.9285714030265808,
"top16_candidate_overlap": 0.9754464285714286,
"draft_hidden_relative_rmse": 0.03659051767294803,
"draft_hidden_cosine_similarity": 0.9993306398391724,
"selector_path_token_agreement": 0.8571429252624512
},
"mlp_only_fp8": {
"tensor_payload_bytes": 2512200960,
"quantized_matrices": 15,
"proposal_kl_mean": 0.0018929995239503794,
"proposal_top1_agreement": 0.9642857313156128,
"top16_candidate_overlap": 0.9765625,
"draft_hidden_relative_rmse": 0.03063499420217177,
"draft_hidden_cosine_similarity": 0.9995341897010803,
"selector_path_token_agreement": 0.8571429252624512
}
}
}