Qwen3.5-9B-TT-Mixed-BFP4-BFP8-P150 / evaluation /evaluation-protocol.json
Lottolabs's picture
Upload verified mixed BFP4/BFP8 checkpoint with MTP and evaluation evidence
12f320c verified
Raw History Blame Contribute Delete
3.01 kB
{
"goal": "Select a reproducible TT-native mixed-precision Qwen3.5-9B quantization using full-distribution output KL and held-out language-model loss",
"upstream": {
"repo_id": "Qwen/Qwen3.5-9B",
"revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a"
},
"reference": "Original text-model weights on Transformers CPU, BF16 compute with original sensitive FP32 parameters preserved; not a claim of exact real-arithmetic reference",
"corpus": "Instruction corpus preparation in progress; corpus.jsonl WikiText now diagnostic only, not precision-selection calibration",
"provenance": "corpus-provenance.json",
"metrics": [
"mean KL(reference || candidate), natural logarithms, full vocabulary",
"next-token negative log likelihood",
"perplexity",
"reference top-1 agreement",
"KL p99.9 and maximum",
"heldout generation behavior and trajectory divergence"
],
"alignment": "For each independent sequence, row i predicts token i+1 from the exact input prefix through token i; reset recurrent and KV state between sequences",
"split_roles": {
"calibration": "Screen supported precision choices and identify sensitive families",
"validation": "Choose among calibration candidates",
"heldout": "Final evaluation only; never use these scores to select precision"
},
"controls": [
"CPU reference repeat",
"TT higher-precision control on matched generic kernels",
"TT BFP4 on matched generic kernels",
"Current packed serving runtime, MTP disabled",
"Selected candidate repeatability and serving-path confirmation"
],
"selection": "Prefer the smaller/faster candidate when quality is comparable; retain measured quality-memory-speed tradeoffs rather than assume a fixed bit width is sufficient",
"rejection_conditions": [
"Nonfinite logits",
"Misaligned tokenization or vocabulary",
"State leaking across records",
"Improvement only on calibration with material validation regression",
"Runtime error incorrectly presented as quantization error"
],
"coverage_limitations": [
"Completed 8192-token WikiText pass is a diagnostic only, not proof of broad instruct fidelity",
"TT native precision selection follows Unsloth principles but is not Unsloth GGUF/NF4 or its proprietary calibration algorithm"
],
"publication_policy": "Do not replace the current server or publish a quality claim until candidate artifacts and measured results agree",
"methodology_sources": [
"https://unsloth.ai/docs/basics/dynamic-3.0-ggufs.md",
"https://unsloth.ai/docs/models/qwen3.5/gguf-benchmarks.md"
],
"calibration_requirements": {
"minimum_tokens": 300000,
"chat_template": "exact pinned original tokenizer template",
"domains": [
"conversation",
"code",
"multilingual"
],
"statistics": "activation second moments for precision ranking, no QAT/QAD",
"holdout": "deduplicated separate instruction examples plus independent WikiText diagnostics"
}
}