{ "base": null, "builder": { "module_sha256": "7f74839fb11762d66fa4cdc9657e7dfa8f4041bb20293dedbf69d54640fe0f0d", "source_commit": "e903af5b6ff68f9efa724bca68dc42b076f396d1" }, "calibration": null, "card": { "assets_receipt_sha256": "9ab6fdcb6c429a8f4df8cad60ee765cbe6ab5bf03ad437b8038b21d4f67230db", "figures_sha256": { "assets/banner.png": "2363db1a11ea3d8c61656a078c2b2cdf10623177e54ca3ac5a48b22ccb0be724", "assets/index-areas.png": "1088285ed8cd75a4063752db5350513f55ba42864a1afecf8cd3b32d390e5064", "assets/index-pareto.png": "241cdf685562dc58e8c30a321cb2056f682cacd33ddc711349b0bec0e01d26e8", "assets/jevarena-types.png": "34b007d8c56bacca1898d05b804db5a10c72113fe117ce3fb1e6b497f1f55316", "assets/jevarena.png": "2e26f0972f90bb274b28f4d929bdfcbe58f147c257f4787019bf92000d371f7d" }, "index_sha256": "bf2bd02d1359de5f1bdd871cb3e240b6f5022cc2e3d6638580e921cd4386ad4a", "readme_sha256": "1dd9bc15c0b6dfe008befdc8b29b22a9564afb7fce1a4a5ef1b79f1b6fe90cd6" }, "files_sha256": { "LICENSE": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a", "README.md": "1dd9bc15c0b6dfe008befdc8b29b22a9564afb7fce1a4a5ef1b79f1b6fe90cd6", "assets/banner.png": "2363db1a11ea3d8c61656a078c2b2cdf10623177e54ca3ac5a48b22ccb0be724", "assets/index-areas.png": "1088285ed8cd75a4063752db5350513f55ba42864a1afecf8cd3b32d390e5064", "assets/index-pareto.png": "241cdf685562dc58e8c30a321cb2056f682cacd33ddc711349b0bec0e01d26e8", "assets/jevarena-types.png": "34b007d8c56bacca1898d05b804db5a10c72113fe117ce3fb1e6b497f1f55316", "assets/jevarena.png": "2e26f0972f90bb274b28f4d929bdfcbe58f147c257f4787019bf92000d371f7d", "backbone/config.json": "071e97d8291168ba712237744c9a60e733acce554e6164322d22550dc96de19b", "backbone/model-00001-of-00002.safetensors": "17ba386b5f9bab7c648ef2509988165cdc341bb75c35a2fa5e8b27faece88fe0", "backbone/model-00002-of-00002.safetensors": "aec9a7eaadce4e7ad764614a56a8183d67ee94eb2e63969165ba129d07a0281a", "backbone/model.safetensors.index.json": "7607bb56aa11afe3a32bba1855e6bc3e54767b4a29c28216765f368b01da45a2", "chat_template.jinja": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80", "config.json": "60c6f61ba93b64fec2d62697cfdb8a474bd0c0299d049592615e36dfde705d03", "configuration_decision2.py": "b88ce8380cb0f210776f3f3ccc06219489b96502c520317790f3f4253e2a9367", "decision2/__init__.py": "10fea99a87686a8f5d99e9a8a91573ce1ddb7bc702f22b1fa21d3de258a4054a", "decision2/_vendor/__init__.py": "fe3babbdd871fa7c0cf919320a66efd1249253abe6bd7cdc5e33352dc46adeef", "decision2/_vendor/dev2model/__init__.py": "4611a85789fae8f06395702a2975592e66e16a8632487c5f9255c0ce68c700d1", "decision2/_vendor/dev2model/calibration.py": "71633a04533c7022e1606ed2734107f33839454c34c993fe16d80c8413d1c561", "decision2/_vendor/dev2model/data.py": "632bd60555f63459ff08ffe82c8263ca6a96f75360bc8c06469fcb10f1e2e99c", "decision2/_vendor/dev2model/decision_model.py": "1ab1e49bccc8b84632a32565ce666a6db94298372ee891dee5b68e54f80957fb", "decision2/_vendor/dev2model/infer.py": "9d5a58324cc758ff92b74d00cfb2cf211b27a1d49b4025d4984f20c0b13f8eb3", "decision2/_vendor/dev2model/lora.py": "7049e35e2a7bf50dd5888902cd0b7030d9bbd448d7d7aba8875ceb89aebff231", "decision2/_vendor/dev2model/source.py": "ef7b30171c4befb26163ea4d3d6e5add9dc602228a476d6f0ead9d40b450c7e3", "decision2/api.py": "3157c0509e8fb6c56038486f19b42ca09f44edcfce2ef32bc9ed5f4768829b0f", "decision2/qwen.py": "97ba565f95d5a5ac9af6632e7397c05cfe9ab47fde68b56df19aabfa2d8fb187", "decision_config.json": "0b6c3429ee06032739d7386659c332ed7bb62d0b96ccddfcad4fad999e22cd4f", "decision_head.safetensors": "33b6541bb6636677eb91a4d8e06acd4db81152b11097088840796f11d49c707a", "modeling_decision2.py": "a3f700b3d2a5deb344821802248a5ea7ea06ccedc5af1823dca4e3e813c17fbb", "pipeline_decision2.py": "3393faec770ddaa9045a7e41092ad17cee5310fe74187deb6f58bbe718b0c1c3", "tokenizer.json": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523", "tokenizer_config.json": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" }, "identity": { "fingerprint_files": { "backbone/config.json": "071e97d8291168ba712237744c9a60e733acce554e6164322d22550dc96de19b", "backbone/model-00001-of-00002.safetensors": "17ba386b5f9bab7c648ef2509988165cdc341bb75c35a2fa5e8b27faece88fe0", "backbone/model-00002-of-00002.safetensors": "aec9a7eaadce4e7ad764614a56a8183d67ee94eb2e63969165ba129d07a0281a", "backbone/model.safetensors.index.json": "7607bb56aa11afe3a32bba1855e6bc3e54767b4a29c28216765f368b01da45a2", "chat_template.jinja": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80", "decision_config.json": "0b6c3429ee06032739d7386659c332ed7bb62d0b96ccddfcad4fad999e22cd4f", "decision_head.safetensors": "33b6541bb6636677eb91a4d8e06acd4db81152b11097088840796f11d49c707a", "tokenizer.json": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523", "tokenizer_config.json": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" }, "model_sha256": "32872f2968e38aa99901797aaab5e12e32e281bf67bb924037d443394411170d" }, "kind": "release", "licence": { "components": [ { "component": "Decision-2.0-Sol-2B weights, decision head, package runtime, card and artwork", "licence": "apache-2.0", "source": "llm-semantic-router/Decision-2.0-Sol-2B" }, { "component": "Decision 1.0 Sol-2B weights (direct weight origin, fully fine-tuned)", "licence": "apache-2.0", "source": "llm-semantic-router/Decision-1.0-Sol-2B@ce0c018a" }, { "component": "Qwen3.5-2B text backbone and tokenizer (upstream of Decision 1.0 Sol)", "licence": "apache-2.0", "source": "Qwen/Qwen3.5-2B@15852e8c" } ], "spdx": "apache-2.0" }, "max_input_tokens": 16384, "model_files": [ "backbone/config.json", "backbone/model-00001-of-00002.safetensors", "backbone/model-00002-of-00002.safetensors", "backbone/model.safetensors.index.json", "chat_template.jinja", "decision_config.json", "decision_head.safetensors", "tokenizer.json", "tokenizer_config.json" ], "model_name": "Decision-2.0-Sol-2B", "origin": { "relation": "finetune", "repo_id": "llm-semantic-router/Decision-1.0-Sol-2B", "revision": "ce0c018a28de16d6639b1cd203b761bf643b89e6", "summary": "Every weight of Decision 1.0 Sol was fine-tuned (nothing frozen, no adapter), with Decision 1.0 Sol's own answer probabilities as soft targets, and the release is the uniform average of three seeds of that fine-tune. Decision 1.0 Sol is itself a text-only fine-tune of [Qwen/Qwen3.5-2B](https://huggingface.co/Qwen/Qwen3.5-2B) at `15852e8c16360a2fea060d615a32b45270f8a8fc` (Apache-2.0), whose text backbone and tokenizer this model inherits; the Qwen3.5 vision tower is not part of it." }, "package_schema": "dev2-package/1", "parameters": { "external_base_text": null, "loaded": 1883930944, "packaged": { "backbone": 1881825088, "head": 2105856, "residual": 0 }, "packaged_files": { "backbone": [ "backbone/model-00001-of-00002.safetensors", "backbone/model-00002-of-00002.safetensors" ], "head": [ "decision_head.safetensors" ], "residual": [] }, "source": "safetensors header element counts; the runtime asserts the loaded count" }, "profile": "qwen-full", "remote_code": { "architectures": [ "Decision2Model" ], "auto_map": { "AutoConfig": "configuration_decision2.Decision2Config", "AutoModel": "modeling_decision2.Decision2Model" }, "automap_source": { "commit": "99432d1a7da5adbc70212df78ae7ebf7e50b41e4", "content_manifest_sha256": "3bede2ca1fc0b805c167bf5ae6b2bd824a3b64d1b6ae98478d04add1cc1444ff", "tree": "fbf37474b1ce2efd228b7f50803848c6f60ebbd2" }, "custom_pipelines": { "decision": { "impl": "pipeline_decision2.Decision2Pipeline", "pt": [ "AutoModel" ], "type": "text" } }, "files": { "configuration_decision2.py": { "sha256": "b88ce8380cb0f210776f3f3ccc06219489b96502c520317790f3f4253e2a9367", "source": "v2/release/automap/configuration_decision2.py", "source_sha256": "b88ce8380cb0f210776f3f3ccc06219489b96502c520317790f3f4253e2a9367" }, "modeling_decision2.py": { "sha256": "a3f700b3d2a5deb344821802248a5ea7ea06ccedc5af1823dca4e3e813c17fbb", "source": "v2/release/automap/modeling_decision2.py", "source_sha256": "a3f700b3d2a5deb344821802248a5ea7ea06ccedc5af1823dca4e3e813c17fbb" }, "pipeline_decision2.py": { "sha256": "3393faec770ddaa9045a7e41092ad17cee5310fe74187deb6f58bbe718b0c1c3", "source": "v2/release/automap/pipeline_decision2.py", "source_sha256": "3393faec770ddaa9045a7e41092ad17cee5310fe74187deb6f58bbe718b0c1c3" } }, "model_type": "decision2", "tested": [ "5.17.0", "5.18.0" ] }, "repo_id": "llm-semantic-router/Decision-2.0-Sol-2B", "runtime": { "equivalence": "decision2/qwen.py loads this full checkpoint with the vendored training/model sources whose SHA-256 equal the scored adapter sources (checked at build time) and applies the per-item batching, BF16-backbone / FP32-head execution, raw probabilities (temperature 1; no calibration file) and answer normalization of v2.dec.infer_dec, which for this non-residual checkpoint wraps the same DecisionModel without extra readouts. The scored checkpoint (073bd1f2) stored every tensor in FP32; this package (v2.release.bf16_copy, receipt a5229ef1) stores its 186 Linear projection matrices in BF16 exactly as BF16 autocast rounds them and every other tensor bit for bit in FP32. From this revision the runtime holds the backbone's BF16-exact Linear weights in BF16, the values BF16 autocast multiplies with, instead of FP32 copies cast before every matmul; it was checked with 0 answer changes on every scored prompt and on mlx-diag (2,275). From this revision the package also ships 🤗 Transformers remote code (configuration_decision2.py, modeling_decision2.py, pipeline_decision2.py; config.json gains model_type, auto_map and custom_pipelines): AutoModel with trust_remote_code loads the package through this runtime, which now also refuses Transformers' remote-code prompt while loading the tokenizer; it was checked with 0 answer changes against the native runtime on every scored prompt and on mlx-diag (2,275). From this revision the runtime runs a request whose padded question batch would put more than 2**30 elements in a gated-delta q / k / v tensor (the FLA kernels' 32-bit offsets) as several GPU-sized batches, longest questions first; requests within that budget, every scored prompt among them, keep the single-batch path, and it was checked with 0 answer changes on every scored prompt and on mlx-diag (2,275). Checked on one GPU of the scoring node against the T = 1 predictions derived exactly from the sealed CAL698 predictions of every scored prompt (typed-final 1,600, css15 6,547, public231 231) and of the mlx-diag diagnostic (2,275) by release.sh --parity, with the scored run's persisted Triton autotune cache.", "requirements": { "causal-conv1d": "1.7.0 (GPU convolution kernels)", "flash-linear-attention": "0.5.2 (GPU gated-delta kernels)", "python": "3.12.13", "safetensors": "0.8.0", "tokenizers": "0.23.2", "torch": "2.12.0+git6bbd260 (ROCm 7.2 build; one ROCm GPU tested)", "transformers": "5.17.0", "triton": "3.7.1 (persisted autotune cache)" }, "runtime_source": { "commit": "99432d1a7da5adbc70212df78ae7ebf7e50b41e4", "content_manifest_sha256": "3bede2ca1fc0b805c167bf5ae6b2bd824a3b64d1b6ae98478d04add1cc1444ff", "tree": "fbf37474b1ce2efd228b7f50803848c6f60ebbd2" }, "scored_runtime_check": { "checked": { "training/model/calibration.py": "71633a04533c7022e1606ed2734107f33839454c34c993fe16d80c8413d1c561", "training/model/data.py": "632bd60555f63459ff08ffe82c8263ca6a96f75360bc8c06469fcb10f1e2e99c", "training/model/decision_model.py": "1ab1e49bccc8b84632a32565ce666a6db94298372ee891dee5b68e54f80957fb", "training/model/infer.py": "9d5a58324cc758ff92b74d00cfb2cf211b27a1d49b4025d4984f20c0b13f8eb3", "training/model/lora.py": "7049e35e2a7bf50dd5888902cd0b7030d9bbd448d7d7aba8875ceb89aebff231", "training/model/source.py": "ef7b30171c4befb26163ea4d3d6e5add9dc602228a476d6f0ead9d40b450c7e3" }, "native_manifest_sha256": "49e4112fc01a107f6d553a3992f74670b4082f721ec009243c1a4e89bb281fcb" }, "vendor_source": { "commit": "33de83cea695e8988e385866ee603d492f7f0e3a", "content_manifest_sha256": "5125a28a08e13dbfd901f057349151e2caaf3d289d0c43c01463f8f9441c37cd", "tree": "0d3298c33f6f1f59216191d28e51b6cc671061d9" } }, "runtime_files": { "decision2/__init__.py": { "rewritten": false, "sha256": "10fea99a87686a8f5d99e9a8a91573ce1ddb7bc702f22b1fa21d3de258a4054a", "source": "v2/release/runtime/__init__.py", "source_sha256": "10fea99a87686a8f5d99e9a8a91573ce1ddb7bc702f22b1fa21d3de258a4054a" }, "decision2/_vendor/__init__.py": { "rewritten": false, "sha256": "fe3babbdd871fa7c0cf919320a66efd1249253abe6bd7cdc5e33352dc46adeef", "source": null }, "decision2/_vendor/dev2model/__init__.py": { "rewritten": false, "sha256": "4611a85789fae8f06395702a2975592e66e16a8632487c5f9255c0ce68c700d1", "source": null }, "decision2/_vendor/dev2model/calibration.py": { "rewritten": false, "sha256": "71633a04533c7022e1606ed2734107f33839454c34c993fe16d80c8413d1c561", "source": "training/model/calibration.py", "source_sha256": "71633a04533c7022e1606ed2734107f33839454c34c993fe16d80c8413d1c561" }, "decision2/_vendor/dev2model/data.py": { "rewritten": false, "sha256": "632bd60555f63459ff08ffe82c8263ca6a96f75360bc8c06469fcb10f1e2e99c", "source": "training/model/data.py", "source_sha256": "632bd60555f63459ff08ffe82c8263ca6a96f75360bc8c06469fcb10f1e2e99c" }, "decision2/_vendor/dev2model/decision_model.py": { "rewritten": false, "sha256": "1ab1e49bccc8b84632a32565ce666a6db94298372ee891dee5b68e54f80957fb", "source": "training/model/decision_model.py", "source_sha256": "1ab1e49bccc8b84632a32565ce666a6db94298372ee891dee5b68e54f80957fb" }, "decision2/_vendor/dev2model/infer.py": { "rewritten": false, "sha256": "9d5a58324cc758ff92b74d00cfb2cf211b27a1d49b4025d4984f20c0b13f8eb3", "source": "training/model/infer.py", "source_sha256": "9d5a58324cc758ff92b74d00cfb2cf211b27a1d49b4025d4984f20c0b13f8eb3" }, "decision2/_vendor/dev2model/lora.py": { "rewritten": false, "sha256": "7049e35e2a7bf50dd5888902cd0b7030d9bbd448d7d7aba8875ceb89aebff231", "source": "training/model/lora.py", "source_sha256": "7049e35e2a7bf50dd5888902cd0b7030d9bbd448d7d7aba8875ceb89aebff231" }, "decision2/_vendor/dev2model/source.py": { "rewritten": false, "sha256": "ef7b30171c4befb26163ea4d3d6e5add9dc602228a476d6f0ead9d40b450c7e3", "source": "training/model/source.py", "source_sha256": "ef7b30171c4befb26163ea4d3d6e5add9dc602228a476d6f0ead9d40b450c7e3" }, "decision2/api.py": { "rewritten": false, "sha256": "3157c0509e8fb6c56038486f19b42ca09f44edcfce2ef32bc9ed5f4768829b0f", "source": "v2/release/runtime/api.py", "source_sha256": "3157c0509e8fb6c56038486f19b42ca09f44edcfce2ef32bc9ed5f4768829b0f" }, "decision2/qwen.py": { "rewritten": false, "sha256": "97ba565f95d5a5ac9af6632e7397c05cfe9ab47fde68b56df19aabfa2d8fb187", "source": "v2/release/runtime/qwen.py", "source_sha256": "97ba565f95d5a5ac9af6632e7397c05cfe9ab47fde68b56df19aabfa2d8fb187" } }, "schema": "dev2-package-manifest/1", "scored": { "label": "post-key same-panel run m3-S2T-soup-nodeA at T = 1 (predictions derived from the sealed CAL698 run by undoing its temperatures; adopted and sealed)", "paired_sha256": "bca44e5f286312f2e96ac69b590347ddf37d81899dd762f5ce48475365c824ab", "predictions_sha256": { "css15": "2cdcf415bbd891f64f3c865067394b3e9a7479e2b26059e55d8ced3a719cf1c7", "mlx-diag": "736a984af6304bf91bd6bfb0950522a09788236dd23978100dd06e76e767bf48", "public231": "f3af4853b325f38754a3ee0f977c1aaf943ee04159a28ccd55c1faa68dd46b29", "typed-final": "177e62c1ed5263f484ed43285a50c481c4103dd9d7f4056a3dd34f9d76d64647" }, "report_sha256": "08745acf9fc753bc1d3b0335dc3e341c697a657a9e6c7c26d85eebfebf4d160c", "seal_sha256": "91a3df051724344b4c4d18bddb2215a81636af98de0d74f9eb9e7f286d88beab" }, "tier": "2B" }