File size: 3,008 Bytes
12f320c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
{
  "goal": "Select a reproducible TT-native mixed-precision Qwen3.5-9B quantization using full-distribution output KL and held-out language-model loss",
  "upstream": {
    "repo_id": "Qwen/Qwen3.5-9B",
    "revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a"
  },
  "reference": "Original text-model weights on Transformers CPU, BF16 compute with original sensitive FP32 parameters preserved; not a claim of exact real-arithmetic reference",
  "corpus": "Instruction corpus preparation in progress; corpus.jsonl WikiText now diagnostic only, not precision-selection calibration",
  "provenance": "corpus-provenance.json",
  "metrics": [
    "mean KL(reference || candidate), natural logarithms, full vocabulary",
    "next-token negative log likelihood",
    "perplexity",
    "reference top-1 agreement",
    "KL p99.9 and maximum",
    "heldout generation behavior and trajectory divergence"
  ],
  "alignment": "For each independent sequence, row i predicts token i+1 from the exact input prefix through token i; reset recurrent and KV state between sequences",
  "split_roles": {
    "calibration": "Screen supported precision choices and identify sensitive families",
    "validation": "Choose among calibration candidates",
    "heldout": "Final evaluation only; never use these scores to select precision"
  },
  "controls": [
    "CPU reference repeat",
    "TT higher-precision control on matched generic kernels",
    "TT BFP4 on matched generic kernels",
    "Current packed serving runtime, MTP disabled",
    "Selected candidate repeatability and serving-path confirmation"
  ],
  "selection": "Prefer the smaller/faster candidate when quality is comparable; retain measured quality-memory-speed tradeoffs rather than assume a fixed bit width is sufficient",
  "rejection_conditions": [
    "Nonfinite logits",
    "Misaligned tokenization or vocabulary",
    "State leaking across records",
    "Improvement only on calibration with material validation regression",
    "Runtime error incorrectly presented as quantization error"
  ],
  "coverage_limitations": [
    "Completed 8192-token WikiText pass is a diagnostic only, not proof of broad instruct fidelity",
    "TT native precision selection follows Unsloth principles but is not Unsloth GGUF/NF4 or its proprietary calibration algorithm"
  ],
  "publication_policy": "Do not replace the current server or publish a quality claim until candidate artifacts and measured results agree",
  "methodology_sources": [
    "https://unsloth.ai/docs/basics/dynamic-3.0-ggufs.md",
    "https://unsloth.ai/docs/models/qwen3.5/gguf-benchmarks.md"
  ],
  "calibration_requirements": {
    "minimum_tokens": 300000,
    "chat_template": "exact pinned original tokenizer template",
    "domains": [
      "conversation",
      "code",
      "multilingual"
    ],
    "statistics": "activation second moments for precision ranking, no QAT/QAD",
    "holdout": "deduplicated separate instruction examples plus independent WikiText diagnostics"
  }
}