Qwen3.8-27B-heretic-ara-DFlash2-fp8 / fp8_target_conditioned_validation.json
magiccodingman's picture
Upload folder using huggingface_hub
b6bbe8a verified
Raw History Blame Contribute Delete
1.25 kB
{
"status": "pass",
"comparison": "BF16 DFlash2 vs selective block-FP8 weights dequantized to BF16",
"samples": 8,
"draft_positions": 56,
"proposal_kl_mean": 0.0018929995239503794,
"proposal_kl_median": 0.001594925176043174,
"proposal_kl_p95": 0.00300999847240746,
"proposal_kl_max": 0.008521616478903316,
"proposal_top1_agreement": 0.9642857313156128,
"selector_path_token_agreement": 0.8571429252624512,
"selector_full_path_agreement": 0.75,
"top16_candidate_overlap": 0.9765625,
"draft_hidden_relative_rmse": 0.03063499420217177,
"draft_hidden_cosine_similarity": 0.9995341897010803,
"teacher_mean_forward_seconds": 0.20252476473979186,
"student_mean_forward_seconds": 0.5098142223869218,
"per_sample_selector_token_agreement": [
1.0,
1.0,
1.0,
1.0,
0.0,
1.0,
0.8571429252624512,
1.0
],
"gates": {
"proposal_kl_mean_max": 0.02,
"proposal_top1_agreement_min": 0.9,
"selector_path_token_agreement_min": 0.85,
"top16_candidate_overlap_min": 0.95
},
"limitation": "RTX 3090 dequantizes FP8 weights to BF16. This test uses real target hidden states and the exact target LM head, but it is not an end-to-end native W8A8 throughput or acceptance benchmark."
}