{ "status": "pass", "comparison": "BF16 DFlash2 vs selective block-FP8 weights dequantized to BF16", "samples": 8, "draft_positions": 56, "proposal_kl_mean": 0.0018929995239503794, "proposal_kl_median": 0.001594925176043174, "proposal_kl_p95": 0.00300999847240746, "proposal_kl_max": 0.008521616478903316, "proposal_top1_agreement": 0.9642857313156128, "selector_path_token_agreement": 0.8571429252624512, "selector_full_path_agreement": 0.75, "top16_candidate_overlap": 0.9765625, "draft_hidden_relative_rmse": 0.03063499420217177, "draft_hidden_cosine_similarity": 0.9995341897010803, "teacher_mean_forward_seconds": 0.20252476473979186, "student_mean_forward_seconds": 0.5098142223869218, "per_sample_selector_token_agreement": [ 1.0, 1.0, 1.0, 1.0, 0.0, 1.0, 0.8571429252624512, 1.0 ], "gates": { "proposal_kl_mean_max": 0.02, "proposal_top1_agreement_min": 0.9, "selector_path_token_agreement_min": 0.85, "top16_candidate_overlap_min": 0.95 }, "limitation": "RTX 3090 dequantizes FP8 weights to BF16. This test uses real target hidden states and the exact target LM head, but it is not an end-to-end native W8A8 throughput or acceptance benchmark." }