Qwen3.5-9B-TT-Mixed-BFP4-BFP8-P150 / evaluation /instruction-provenance.json
Lottolabs's picture
Upload verified mixed BFP4/BFP8 checkpoint with MTP and evaluation evidence
12f320c verified
Raw History Blame Contribute Delete
9.5 kB
{
"outputs": {
"instruction-calibration.jsonl": {
"counts": {
"calibration/chat": {
"languages": {
"en": 90
},
"prediction_tokens": 90442,
"records": 90,
"tokens": 90532
},
"calibration/code": {
"languages": {
"en": 170
},
"prediction_tokens": 90007,
"records": 170,
"tokens": 90177
},
"calibration/multilingual": {
"languages": {
"acq": 1,
"amh": 2,
"arb": 8,
"ary": 12,
"arz": 1,
"ben": 4,
"ell": 2,
"eng": 6,
"eus": 3,
"fil": 4,
"fin": 3,
"fra": 1,
"guj": 11,
"hau": 6,
"hin": 1,
"ibo": 3,
"ita": 1,
"jpn": 17,
"kir": 14,
"lit": 1,
"mal": 5,
"mar": 6,
"nld": 4,
"npi": 5,
"nya": 1,
"pan": 10,
"pes": 2,
"plt": 24,
"por": 10,
"rus": 1,
"sin": 25,
"sna": 1,
"snd": 1,
"som": 15,
"spa": 4,
"swe": 1,
"swh": 1,
"tam": 25,
"tel": 14,
"tur": 12,
"ukr": 1,
"urd": 4,
"vie": 12,
"wol": 4,
"xho": 1,
"yor": 15,
"zho": 6,
"zsm": 16,
"zul": 4
},
"prediction_tokens": 89762,
"records": 331,
"tokens": 90093
},
"calibration/tools": {
"languages": {
"en": 23
},
"prediction_tokens": 30116,
"records": 23,
"tokens": 30139
}
},
"max_record_tokens": 1979,
"min_record_tokens": 32,
"records": 614,
"sha256": "c65c34c37633381a7322ff433a292f9e615d414bb834d1951b243ee16ecb310e",
"tokens": 300941
},
"instruction-eval.jsonl": {
"counts": {
"heldout/chat": {
"languages": {
"en": 2
},
"prediction_tokens": 950,
"records": 2,
"tokens": 952
},
"heldout/code": {
"languages": {
"en": 4
},
"prediction_tokens": 953,
"records": 4,
"tokens": 957
},
"heldout/multilingual": {
"languages": {
"arb": 1,
"eng": 2,
"por": 1,
"tel": 1,
"tur": 1,
"zho": 1
},
"prediction_tokens": 1010,
"records": 7,
"tokens": 1017
},
"heldout/tools": {
"languages": {
"en": 1
},
"prediction_tokens": 1020,
"records": 1,
"tokens": 1021
},
"validation/chat": {
"languages": {
"en": 1
},
"prediction_tokens": 994,
"records": 1,
"tokens": 995
},
"validation/code": {
"languages": {
"en": 3
},
"prediction_tokens": 914,
"records": 3,
"tokens": 917
},
"validation/multilingual": {
"languages": {
"arb": 1,
"eng": 1,
"por": 1,
"yor": 1
},
"prediction_tokens": 963,
"records": 4,
"tokens": 967
},
"validation/tools": {
"languages": {
"en": 1
},
"prediction_tokens": 937,
"records": 1,
"tokens": 938
}
},
"max_record_tokens": 1021,
"min_record_tokens": 38,
"records": 23,
"sha256": "a53a6e18c4d33b787f15bb1aad6ce88de97f25239ad2236617a62256b297f979",
"tokens": 7764
},
"instruction-long-context.jsonl": {
"counts": {
"heldout/chat": {
"languages": {
"en": 1
},
"prediction_tokens": 2241,
"records": 1,
"tokens": 2242
},
"heldout/multilingual": {
"languages": {
"arb": 1
},
"prediction_tokens": 2294,
"records": 1,
"tokens": 2295
},
"heldout/tools": {
"languages": {
"en": 1
},
"prediction_tokens": 2829,
"records": 1,
"tokens": 2830
},
"validation/chat": {
"languages": {
"en": 1
},
"prediction_tokens": 2377,
"records": 1,
"tokens": 2378
},
"validation/multilingual": {
"languages": {
"arb": 1
},
"prediction_tokens": 2475,
"records": 1,
"tokens": 2476
},
"validation/tools": {
"languages": {
"en": 1
},
"prediction_tokens": 2331,
"records": 1,
"tokens": 2332
}
},
"max_record_tokens": 2830,
"min_record_tokens": 2242,
"records": 6,
"sha256": "6ca2b86259d40fe30ab5db341afec97f9a0efd8074f0754f8978988b0d3f6d3f",
"tokens": 14553
}
},
"packages": {
"jinja2": "3.1.6",
"tokenizers": "0.22.2",
"transformers": "5.12.1"
},
"purpose": "Diverse exact-template instruct PTQ calibration for TT sensitivity/precision ranking; not official Unsloth data or a llama.cpp imatrix; no training/QAT/QAD",
"rejected_candidates": {
"chat:outside_32_4096": 2,
"multilingual:normalized_prompt_duplicate": 1,
"multilingual:outside_32_4096": 116,
"tools:conversion:Expecting value: line 1 column 1 (char 0)": 326,
"tools:conversion:No real tool schemas": 24,
"tools:conversion:Require a complete user-to-assistant conversation": 1,
"tools:conversion:Tool response/call order mismatch": 63
},
"schema_version": 1,
"script_sha256": "c5abd94571e74b281cff61c342c5037742e5384ed3a478ce6c9c3df7d42fc5fa",
"selection": {
"calibration_target_tokens": {
"chat": 90000,
"code": 90000,
"multilingual": 90000,
"tools": 30000
},
"candidate_order": "ascending SHA256(id + ':order-v1')",
"context_policy": "No truncation or packing. Entire conversations only; 32..2048 calibration, 32..1024 small KL, 2049..4096 separate long contexts. Oversized or malformed/truncated source rows rejected.",
"dedup": "Global exact normalized user-prompt SHA256 (Unicode NFKC, casefold, whitespace collapse), across all domains and all splits including long contexts; reject entire conversation if any user prompt already used. Semantic/paraphrase dedup not claimed.",
"eval_independence": "No calibration prompt overlap; official heldout source splits where available, deterministic disjoint row partitions otherwise. Same dataset families, NOT independent-source generalization proof.",
"kl_total_token_cap": 8192,
"multilingual_eval_order": "Stable language-round-robin over hash order: first occurrence of every language before second occurrences, etc.",
"source_rows": "50-row windows at evenly spaced integer offsets spanning each entire source split; source order preserved inside each window",
"split_assignment": "Official train/test for UltraChat and Aya; test hashed 50/50 validation/heldout. Code/Hermes train rows hashed 80/10/10. SHA256(domain:source_split:row_idx:split-v1) first8 hex mod10."
},
"snapshot_manifest_sha256": "d73630cfe8466e5545fdeba82f9dd342a9ea5d4232facfe94f93f4119a9f5b58",
"sources": {
"chat": {
"config": "default",
"dataset": "HuggingFaceH4/ultrachat_200k",
"license": "mit",
"revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
"splits": {
"test_sft": 10,
"train_sft": 20
}
},
"code": {
"config": "default",
"dataset": "m-a-p/CodeFeedback-Filtered-Instruction",
"license": "apache-2.0",
"revision": "a08c213a9748c66c15d0225814be80a2e77adf4a",
"splits": {
"train": 24
}
},
"multilingual": {
"config": "default",
"dataset": "CohereLabs/aya_dataset",
"license": "apache-2.0",
"revision": "f9ea04583f02a8f86404ff6c58bf75fe637df8a2",
"splits": {
"test": 10,
"train": 40
}
},
"tools": {
"config": "func_calling",
"dataset": "NousResearch/hermes-function-calling-v1",
"license": "apache-2.0",
"revision": "dae3e1d28cfbcf4b915c04ea1e072030529b4bda",
"splits": {
"train": 16
}
}
},
"tokenizer": {
"add_generation_prompt": false,
"chat_template_applied": true,
"enable_thinking": false,
"files": {
"chat_template.jinja": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715",
"tokenizer.json": "5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42",
"tokenizer_config.json": "316230d6a809701f4db5ea8f8fc862bc3a6f3229c937c174e674ff3ca0a64ac8"
},
"model": "Qwen/Qwen3.5-9B",
"revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a",
"tools": "Real Hermes function schemas, assistant.tool_calls argument objects and linked tool responses; source system tool boilerplate replaced by exact Qwen template tools rendering"
}
}