{ "goal": "Select a reproducible TT-native mixed-precision Qwen3.5-9B quantization using full-distribution output KL and held-out language-model loss", "upstream": { "repo_id": "Qwen/Qwen3.5-9B", "revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a" }, "reference": "Original text-model weights on Transformers CPU, BF16 compute with original sensitive FP32 parameters preserved; not a claim of exact real-arithmetic reference", "corpus": "Instruction corpus preparation in progress; corpus.jsonl WikiText now diagnostic only, not precision-selection calibration", "provenance": "corpus-provenance.json", "metrics": [ "mean KL(reference || candidate), natural logarithms, full vocabulary", "next-token negative log likelihood", "perplexity", "reference top-1 agreement", "KL p99.9 and maximum", "heldout generation behavior and trajectory divergence" ], "alignment": "For each independent sequence, row i predicts token i+1 from the exact input prefix through token i; reset recurrent and KV state between sequences", "split_roles": { "calibration": "Screen supported precision choices and identify sensitive families", "validation": "Choose among calibration candidates", "heldout": "Final evaluation only; never use these scores to select precision" }, "controls": [ "CPU reference repeat", "TT higher-precision control on matched generic kernels", "TT BFP4 on matched generic kernels", "Current packed serving runtime, MTP disabled", "Selected candidate repeatability and serving-path confirmation" ], "selection": "Prefer the smaller/faster candidate when quality is comparable; retain measured quality-memory-speed tradeoffs rather than assume a fixed bit width is sufficient", "rejection_conditions": [ "Nonfinite logits", "Misaligned tokenization or vocabulary", "State leaking across records", "Improvement only on calibration with material validation regression", "Runtime error incorrectly presented as quantization error" ], "coverage_limitations": [ "Completed 8192-token WikiText pass is a diagnostic only, not proof of broad instruct fidelity", "TT native precision selection follows Unsloth principles but is not Unsloth GGUF/NF4 or its proprietary calibration algorithm" ], "publication_policy": "Do not replace the current server or publish a quality claim until candidate artifacts and measured results agree", "methodology_sources": [ "https://unsloth.ai/docs/basics/dynamic-3.0-ggufs.md", "https://unsloth.ai/docs/models/qwen3.5/gguf-benchmarks.md" ], "calibration_requirements": { "minimum_tokens": 300000, "chat_template": "exact pinned original tokenizer template", "domains": [ "conversation", "code", "multilingual" ], "statistics": "activation second moments for precision ranking, no QAT/QAD", "holdout": "deduplicated separate instruction examples plus independent WikiText diagnostics" } }