File size: 3,008 Bytes
12f320c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 | {
"goal": "Select a reproducible TT-native mixed-precision Qwen3.5-9B quantization using full-distribution output KL and held-out language-model loss",
"upstream": {
"repo_id": "Qwen/Qwen3.5-9B",
"revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a"
},
"reference": "Original text-model weights on Transformers CPU, BF16 compute with original sensitive FP32 parameters preserved; not a claim of exact real-arithmetic reference",
"corpus": "Instruction corpus preparation in progress; corpus.jsonl WikiText now diagnostic only, not precision-selection calibration",
"provenance": "corpus-provenance.json",
"metrics": [
"mean KL(reference || candidate), natural logarithms, full vocabulary",
"next-token negative log likelihood",
"perplexity",
"reference top-1 agreement",
"KL p99.9 and maximum",
"heldout generation behavior and trajectory divergence"
],
"alignment": "For each independent sequence, row i predicts token i+1 from the exact input prefix through token i; reset recurrent and KV state between sequences",
"split_roles": {
"calibration": "Screen supported precision choices and identify sensitive families",
"validation": "Choose among calibration candidates",
"heldout": "Final evaluation only; never use these scores to select precision"
},
"controls": [
"CPU reference repeat",
"TT higher-precision control on matched generic kernels",
"TT BFP4 on matched generic kernels",
"Current packed serving runtime, MTP disabled",
"Selected candidate repeatability and serving-path confirmation"
],
"selection": "Prefer the smaller/faster candidate when quality is comparable; retain measured quality-memory-speed tradeoffs rather than assume a fixed bit width is sufficient",
"rejection_conditions": [
"Nonfinite logits",
"Misaligned tokenization or vocabulary",
"State leaking across records",
"Improvement only on calibration with material validation regression",
"Runtime error incorrectly presented as quantization error"
],
"coverage_limitations": [
"Completed 8192-token WikiText pass is a diagnostic only, not proof of broad instruct fidelity",
"TT native precision selection follows Unsloth principles but is not Unsloth GGUF/NF4 or its proprietary calibration algorithm"
],
"publication_policy": "Do not replace the current server or publish a quality claim until candidate artifacts and measured results agree",
"methodology_sources": [
"https://unsloth.ai/docs/basics/dynamic-3.0-ggufs.md",
"https://unsloth.ai/docs/models/qwen3.5/gguf-benchmarks.md"
],
"calibration_requirements": {
"minimum_tokens": 300000,
"chat_template": "exact pinned original tokenizer template",
"domains": [
"conversation",
"code",
"multilingual"
],
"statistics": "activation second moments for precision ranking, no QAT/QAD",
"holdout": "deduplicated separate instruction examples plus independent WikiText diagnostics"
}
}
|