{ "format": "qwen35-9b-ttnn-native-text-v1", "scope": "text-only-no-vision-with-mtp", "precision_groups": { "torch.bfloat16": { "tensors": 193, "parameters": 1261466880, "serialized_tensor_bytes": 2522954504 }, "torch.float32": { "tensors": 48, "parameters": 3840, "serialized_tensor_bytes": 20160 }, "bfp8": { "tensors": 195, "parameters": 5217714176, "serialized_tensor_bytes": 5543891512 }, "bfp4": { "tensors": 54, "parameters": 2717908992, "serialized_tensor_bytes": 1528843248 } }, "artifact_bytes": 9629415772, "tensor_parameters": 9197093888, "quantizer": "TTNN native BFP4_B/BFP8_B rounding with selectively retained precision; not the Unsloth quantizer", "calibration": "300941 chat-templated input tokens; activation second moments propose precision, validation output KL selects it", "embedding": "Original BF16 retained", "nonlinear": "All 48 original FP32 GDN norm/A_log tensors retained losslessly", "limitations": [ "Single Blackhole P150; text-only with MTP-1, no vision", "Native TT format, not GGUF or bitsandbytes", "Local validation gates are not official Unsloth benchmark thresholds" ], "mtp": { "tensors": 15, "serialized_bytes": 486582864, "storage": "All original draft tensors preserved losslessly at source dtype", "runtime": "Existing draft attention and MLP BFP4 policy retained; FC and norms BF16; shared embedding BF16 and shared target LM head BFP8", "runtime_precision_binding": "Mode-specific equivalence runtime_environment and runtime_sources; target precision map does not apply to draft layers" } }