{ "decision": "select mlp-only profile", "reason": "Preserving attention output projections costs approximately 105 MB but improves proposal KL, top-1 agreement, and hidden-state fidelity.", "profiles": { "mlp_plus_attention_output_fp8": { "tensor_payload_bytes": 2407368960, "quantized_matrices": 20, "proposal_kl_mean": 0.0026533013419426316, "proposal_top1_agreement": 0.9285714030265808, "top16_candidate_overlap": 0.9754464285714286, "draft_hidden_relative_rmse": 0.03659051767294803, "draft_hidden_cosine_similarity": 0.9993306398391724, "selector_path_token_agreement": 0.8571429252624512 }, "mlp_only_fp8": { "tensor_payload_bytes": 2512200960, "quantized_matrices": 15, "proposal_kl_mean": 0.0018929995239503794, "proposal_top1_agreement": 0.9642857313156128, "top16_candidate_overlap": 0.9765625, "draft_hidden_relative_rmse": 0.03063499420217177, "draft_hidden_cosine_similarity": 0.9995341897010803, "selector_path_token_agreement": 0.8571429252624512 } } }