Feature Extraction
MLX
Safetensors
qwen3
apple-silicon
quantized
mixed-precision
axquant
axq
development
MXFP8
embedding
sentence-similarity
8-bit precision
Instructions to use AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8 with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] hf download AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8 --local-dir AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download axquant_plan.json from AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8: direct link, hf CLI and curl.
- Browser
- Download file 352 kB
-
https://huggingface.co/AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8/resolve/main/axquant_plan.json
- Command line
-
hf download hf://AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8/axquant_plan.json
-
curl -L -o axquant_plan.json https://huggingface.co/AutomatosX/AX-Qwen3-Embedding-4B-MLX-AXQ-MXFP8/resolve/main/axquant_plan.json
352 kB
| { | |
| "analysis_sha256": "274968a93f24b1211cf0fe611509477db75ab3aaf9e28a27e960d31bba60985a", | |
| "architecture_profile": { | |
| "adapter_id": "qwen3-dense-v1", | |
| "audio_present": false, | |
| "config_model_type": "qwen3", | |
| "dense": true, | |
| "mtp_declared": false, | |
| "notes": [ | |
| "Qwen3 dense (model_type=qwen3), including Qwen3-Embedding retrieval models.", | |
| "Embedding checkpoints share the causal backbone layout; use embedding runtimes for retrieval quality — do not claim generative or MTP metrics." | |
| ], | |
| "optimization_scope": "text-path", | |
| "product_family": "qwen3", | |
| "support_level": "supported", | |
| "support_tier": "convertible", | |
| "text_layer_count": 36, | |
| "vision_present": false | |
| }, | |
| "assignments": [ | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "embed_tokens", | |
| "outlier_strategy": "none", | |
| "parameters": 388262400, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule embedding-8: Embedding floor 8-bit affine", | |
| "role": "embedding", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "embed_tokens.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.0.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.0.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.0.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.0.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.0.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.0.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.1.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.1.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.1.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.1.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.1.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.1.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.10.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.10.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.10.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.10.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.10.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.10.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.11.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.11.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.11.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.11.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.11.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.11.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.12.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.12.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.12.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.12.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.12.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.12.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.13.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.13.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.13.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.13.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.13.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.13.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.14.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.14.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.14.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.14.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.14.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.14.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.15.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.15.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.15.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.15.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.15.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.15.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.16.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.16.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.16.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.16.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.16.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.16.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.17.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.17.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.17.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.17.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.17.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.17.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.18.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.18.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.18.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.18.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.18.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.18.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.19.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.19.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.19.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.19.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.19.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.19.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.2.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.2.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.2.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.2.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.2.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.2.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.20.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.20.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.3.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.3.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.3.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.3.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.3.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.3.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.4.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.4.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.4.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.4.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.4.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.4.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.5.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.5.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.5.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.5.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.5.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.5.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.6.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.6.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.6.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.6.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.6.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.6.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.7.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.7.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.7.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.7.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.7.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.7.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.8.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.8.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.8.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.8.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.8.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.8.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.9.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.9.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.9.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.9.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.9.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.9.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.20.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.20.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.20.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.20.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.21.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.21.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.21.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.21.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.21.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.21.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.22.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.22.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.22.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.22.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.22.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.22.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.23.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.23.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.23.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.23.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.23.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.23.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.24.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.24.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.24.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.24.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.24.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.24.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.25.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.25.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.25.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.25.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.25.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.25.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.26.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.26.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.26.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.26.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.26.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.26.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.27.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.27.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.27.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.27.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.27.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.27.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.28.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.28.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.28.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.28.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.28.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.28.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.29.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.29.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.29.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.29.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.29.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.29.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.30.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.30.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.30.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.30.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.30.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.30.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.31.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.31.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.31.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.31.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.31.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.31.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.32.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.32.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.32.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.32.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.32.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.32.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.33.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.33.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.33.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.33.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.33.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.33.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.34.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.34.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.34.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.34.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.34.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.34.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.input_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.35.input_layernorm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.mlp.down_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.mlp.down_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.mlp.gate_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.mlp.gate_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.mlp.up_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 24903680, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule mlp-mxfp8: AXQ-mxfp8 MLP trunk", | |
| "role": "mlp", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.mlp.up_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.post_attention_layernorm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.35.post_attention_layernorm.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.k_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.35.self_attn.k_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.k_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.self_attn.k_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.o_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.self_attn.o_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.q_norm", | |
| "outlier_strategy": "none", | |
| "parameters": 128, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "layers.35.self_attn.q_norm.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.q_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 10485760, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.self_attn.q_proj.weight" | |
| }, | |
| { | |
| "bits": 8, | |
| "group_size": 32, | |
| "method": "affine", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "layers.35.self_attn.v_proj", | |
| "outlier_strategy": "none", | |
| "parameters": 2621440, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule attention-mxfp8: AXQ-mxfp8 attention trunk", | |
| "role": "attention", | |
| "scale_strategy": "group-affine", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 9.0 | |
| }, | |
| "tensor": "layers.35.self_attn.v_proj.weight" | |
| }, | |
| { | |
| "bits": 16, | |
| "group_size": null, | |
| "method": "bf16", | |
| "metrics": { | |
| "cosine_distance": 0.0, | |
| "decode_latency_cost": 0.0, | |
| "hidden_state_error": 0.0, | |
| "long_context_loss": 0.0, | |
| "mtp_acceptance_loss": 0.0, | |
| "output_kl": 0.0, | |
| "peak_memory_cost": 0.0, | |
| "prefill_latency_cost": 0.0, | |
| "task_loss_delta": 0.0, | |
| "token_disagreement": 0.0 | |
| }, | |
| "module_path": "norm", | |
| "outlier_strategy": "none", | |
| "parameters": 2560, | |
| "predicted_loss": 0.0, | |
| "reason": "manual rule protect-norms: Protection floor — norms stay BF16", | |
| "role": "norm", | |
| "scale_strategy": "none", | |
| "strategy_metadata": { | |
| "selected_from_candidates": 1, | |
| "storage_bpw": 16.0 | |
| }, | |
| "tensor": "norm.weight" | |
| } | |
| ], | |
| "calibration": null, | |
| "candidate_bits": [ | |
| 8, | |
| 16 | |
| ], | |
| "candidate_group_sizes": [], | |
| "constraints": { | |
| "effective_bpw_limit": 16.0, | |
| "lm_head_min_bits": 16, | |
| "max_model_size_ratio_to_uniform4": 1.1, | |
| "minimum_mtp_acceptance_retention": 0.95, | |
| "minimum_mtp_speedup": 1.2, | |
| "minimum_quality_retention": 0.98 | |
| }, | |
| "cost_model": "abstract-bpw", | |
| "created_at": "2026-10-04T02:06:47.618121Z", | |
| "effective_bpw": 9.000341310050072, | |
| "evidence_kind": "architecture_prior", | |
| "global_validation_required": true, | |
| "group_size": 32, | |
| "hardware": { | |
| "name": "ax-engine-apple-silicon-affine-dwq-v3", | |
| "runtime": "ax-engine", | |
| "supported_bits": [ | |
| 2, | |
| 3, | |
| 4, | |
| 6, | |
| 8, | |
| 16 | |
| ], | |
| "supported_group_sizes": [ | |
| 32, | |
| 64, | |
| 128 | |
| ], | |
| "supported_methods": [ | |
| "affine", | |
| "awq", | |
| "dwq", | |
| "gptq", | |
| "gptq-act", | |
| "bf16" | |
| ] | |
| }, | |
| "kernel_latency_host_id": null, | |
| "kernel_latency_sha256": null, | |
| "kv_cache": null, | |
| "method_near_ties": [], | |
| "method_near_ties_omitted": 0, | |
| "mtp": { | |
| "candidate_bits": [ | |
| 8, | |
| 16 | |
| ], | |
| "min_bits": 8, | |
| "mode": "protected", | |
| "optimize_for_acceptance": true, | |
| "preserve_external_sidecar": true, | |
| "protect_norms": true, | |
| "protect_output_head": true | |
| }, | |
| "mtp_distribution": {}, | |
| "nominal_bpw": 8.000390068628654, | |
| "objective": { | |
| "cosine_distance": 0.03, | |
| "decode_latency_cost": 0.07, | |
| "hidden_state_error": 0.1, | |
| "long_context_loss": 0.05, | |
| "mtp_acceptance_loss": 0.22, | |
| "output_kl": 0.15, | |
| "peak_memory_cost": 0.03, | |
| "prefill_latency_cost": 0.03, | |
| "task_loss_delta": 0.2, | |
| "token_disagreement": 0.12 | |
| }, | |
| "primary_runtime": "ax-engine", | |
| "profile": "agent-coding", | |
| "quantizer": "axquant", | |
| "random_seed": 20260728, | |
| "schema_version": "axquant.plan.v2", | |
| "software_versions": { | |
| "ax_engine": "7.5.7", | |
| "axquant": "1.9.0", | |
| "mlx": "0.32.1", | |
| "mlx_lm": "0.31.3", | |
| "pydantic": "2.13.4", | |
| "python": "3.12.13", | |
| "safetensors": "0.8.0" | |
| }, | |
| "source_model": { | |
| "architecture": "Qwen3ForCausalLM", | |
| "format": "mlx", | |
| "local_path": null, | |
| "model_id": "Qwen/Qwen3-Embedding-4B", | |
| "revision": "5cf2132abc99cad020ac570b19d031efec650f2b" | |
| }, | |
| "status": "planned", | |
| "target_bpw": 16.0, | |
| "target_class": "16p0bpw", | |
| "target_mode": "low-memory", | |
| "warnings": [ | |
| "Manual assignments are unmeasured development evidence.", | |
| "Conversion requires --allow-unmeasured and cannot pass publication gates." | |
| ], | |
| "weight_distribution": { | |
| "8bit": { | |
| "fraction": 0.9999512414214182, | |
| "parameters": 4021578240 | |
| }, | |
| "bf16": { | |
| "fraction": 4.8758578581769534e-05, | |
| "parameters": 196096 | |
| } | |
| } | |
| } | |