File size: 1,702 Bytes
12f320c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
{
  "format": "qwen35-9b-ttnn-native-text-v1",
  "scope": "text-only-no-vision-with-mtp",
  "precision_groups": {
    "torch.bfloat16": {
      "tensors": 193,
      "parameters": 1261466880,
      "serialized_tensor_bytes": 2522954504
    },
    "torch.float32": {
      "tensors": 48,
      "parameters": 3840,
      "serialized_tensor_bytes": 20160
    },
    "bfp8": {
      "tensors": 195,
      "parameters": 5217714176,
      "serialized_tensor_bytes": 5543891512
    },
    "bfp4": {
      "tensors": 54,
      "parameters": 2717908992,
      "serialized_tensor_bytes": 1528843248
    }
  },
  "artifact_bytes": 9629415772,
  "tensor_parameters": 9197093888,
  "quantizer": "TTNN native BFP4_B/BFP8_B rounding with selectively retained precision; not the Unsloth quantizer",
  "calibration": "300941 chat-templated input tokens; activation second moments propose precision, validation output KL selects it",
  "embedding": "Original BF16 retained",
  "nonlinear": "All 48 original FP32 GDN norm/A_log tensors retained losslessly",
  "limitations": [
    "Single Blackhole P150; text-only with MTP-1, no vision",
    "Native TT format, not GGUF or bitsandbytes",
    "Local validation gates are not official Unsloth benchmark thresholds"
  ],
  "mtp": {
    "tensors": 15,
    "serialized_bytes": 486582864,
    "storage": "All original draft tensors preserved losslessly at source dtype",
    "runtime": "Existing draft attention and MLP BFP4 policy retained; FC and norms BF16; shared embedding BF16 and shared target LM head BFP8",
    "runtime_precision_binding": "Mode-specific equivalence runtime_environment and runtime_sources; target precision map does not apply to draft layers"
  }
}