Correct question-scaling measurement documentation
Browse files- MATERIALS.json +4 -3
- QUESTION-SCALING.md +1 -1
- README.md +1 -1
- metrics/question-scaling.json +13 -1
- release-manifest.json +15 -14
MATERIALS.json
CHANGED
|
@@ -8,8 +8,8 @@
|
|
| 8 |
".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
|
| 9 |
"DIAGNOSTICS.md": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73",
|
| 10 |
"EVALUATION.md": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e",
|
| 11 |
-
"QUESTION-SCALING.md": "
|
| 12 |
-
"README.md": "
|
| 13 |
"SENSITIVITY.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
|
| 14 |
"TASKS.md": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476",
|
| 15 |
"USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
|
|
@@ -22,6 +22,7 @@
|
|
| 22 |
"assets/decision-ranking.png": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6",
|
| 23 |
"assets/decision-ranking.svg": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62",
|
| 24 |
"metrics/benchmark.json": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56",
|
| 25 |
-
"metrics/evaluation-provenance.json": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5"
|
|
|
|
| 26 |
}
|
| 27 |
}
|
|
|
|
| 8 |
".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
|
| 9 |
"DIAGNOSTICS.md": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73",
|
| 10 |
"EVALUATION.md": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e",
|
| 11 |
+
"QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
|
| 12 |
+
"README.md": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524",
|
| 13 |
"SENSITIVITY.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
|
| 14 |
"TASKS.md": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476",
|
| 15 |
"USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
|
|
|
|
| 22 |
"assets/decision-ranking.png": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6",
|
| 23 |
"assets/decision-ranking.svg": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62",
|
| 24 |
"metrics/benchmark.json": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56",
|
| 25 |
+
"metrics/evaluation-provenance.json": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5",
|
| 26 |
+
"metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
|
| 27 |
}
|
| 28 |
}
|
QUESTION-SCALING.md
CHANGED
|
@@ -15,6 +15,6 @@ Nox · distinct Choice questions with **499 input tokens per question**. Only th
|
|
| 15 |
|
| 16 |
Six independently loaded process blocks supply 30 measured requests per point after warmup. End-to-end Python request latency includes rendering, tokenization, inference, output assembly and final synchronization. Model loading and network are excluded. Requests are sequential, with no cross-request prefix cache.
|
| 17 |
|
| 18 |
-
|
| 19 |
|
| 20 |
These fixed short-input Choice measurements do not establish concurrent HTTP throughput, long-context scaling, other question-type performance or a cross-hardware speed ranking.
|
|
|
|
| 15 |
|
| 16 |
Six independently loaded process blocks supply 30 measured requests per point after warmup. End-to-end Python request latency includes rendering, tokenization, inference, output assembly and final synchronization. Model loading and network are excluded. Requests are sequential, with no cross-request prefix cache.
|
| 17 |
|
| 18 |
+
These measurements precede the Choice null-description update. Every option in this workload supplies a description, so the update does not change these rendered inputs. Current Nox fills a null description with its option ID; this curve does not measure that case. The measured runtime previously passed an offline Hub-download proof and full 3,160-answer regression. [Measured revision and runtime scope](metrics/question-scaling.json).
|
| 19 |
|
| 20 |
These fixed short-input Choice measurements do not establish concurrent HTTP throughput, long-context scaling, other question-type performance or a cross-hardware speed ranking.
|
README.md
CHANGED
|
@@ -64,7 +64,7 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading *
|
|
| 64 |
|
| 65 |

|
| 66 |
|
| 67 |
-
Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. [p50, p95 and
|
| 68 |
|
| 69 |
## Use Nox-4B
|
| 70 |
|
|
|
|
| 64 |
|
| 65 |

|
| 66 |
|
| 67 |
+
Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](QUESTION-SCALING.md).
|
| 68 |
|
| 69 |
## Use Nox-4B
|
| 70 |
|
metrics/question-scaling.json
CHANGED
|
@@ -109,5 +109,17 @@
|
|
| 109 |
},
|
| 110 |
"timing_summary_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
|
| 111 |
"API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 112 |
-
"scope": "Only optimized path subsequently published, no old-release series. Fixed-length Python requests, excludes network and loading."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 113 |
}
|
|
|
|
| 109 |
},
|
| 110 |
"timing_summary_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
|
| 111 |
"API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 112 |
+
"scope": "Only optimized path subsequently published, no old-release series. Fixed-length Python requests, excludes network and loading.",
|
| 113 |
+
"measurement_applicability": {
|
| 114 |
+
"measured_API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 115 |
+
"current_API_sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e",
|
| 116 |
+
"measured_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 117 |
+
"current_bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 118 |
+
"current_API_byte_identical_to_measured": false,
|
| 119 |
+
"measured_criteria_all_have_non_null_descriptions": true,
|
| 120 |
+
"runtime_change": "Only Choice description=None is replaced with that option key; explicit descriptions use the same rendering path.",
|
| 121 |
+
"same_measured_points_retained": true,
|
| 122 |
+
"current_null_description_latency_not_measured_by_this_curve": true,
|
| 123 |
+
"no_HTTP_throughput_claim": true
|
| 124 |
+
}
|
| 125 |
}
|
release-manifest.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"status": "qualified-runtime-and-current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 5 |
"readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
|
| 6 |
-
"model_card_sha256": "
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
|
@@ -41,8 +41,8 @@
|
|
| 41 |
},
|
| 42 |
{
|
| 43 |
"file": "MATERIALS.json",
|
| 44 |
-
"bytes":
|
| 45 |
-
"sha256": "
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
@@ -56,8 +56,8 @@
|
|
| 56 |
},
|
| 57 |
{
|
| 58 |
"file": "QUESTION-SCALING.md",
|
| 59 |
-
"bytes":
|
| 60 |
-
"sha256": "
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"file": "QWEN-LICENSE",
|
|
@@ -66,8 +66,8 @@
|
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"file": "README.md",
|
| 69 |
-
"bytes":
|
| 70 |
-
"sha256": "
|
| 71 |
},
|
| 72 |
{
|
| 73 |
"file": "RUNTIME-RELEASE.json",
|
|
@@ -411,8 +411,8 @@
|
|
| 411 |
},
|
| 412 |
{
|
| 413 |
"file": "metrics/question-scaling.json",
|
| 414 |
-
"bytes":
|
| 415 |
-
"sha256": "
|
| 416 |
},
|
| 417 |
{
|
| 418 |
"file": "metrics/semantic-consistency.json",
|
|
@@ -577,11 +577,12 @@
|
|
| 577 |
"MATERIALS.json": "copy",
|
| 578 |
"metrics/semantic-consistency.json": "copy"
|
| 579 |
},
|
| 580 |
-
"scope": "
|
| 581 |
"release_tag": "v1.3.2",
|
| 582 |
-
"change_kind": "
|
| 583 |
-
"previous_main_revision": "
|
| 584 |
-
"previous_release_manifest_sha256": "
|
| 585 |
"presentation_amendment_sha256": "038f01d857b372a8236d0ca634e44d6d27095b5937218410b5050d5b9105a4bf",
|
| 586 |
-
"statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc"
|
|
|
|
| 587 |
}
|
|
|
|
| 3 |
"status": "qualified-runtime-and-current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 5 |
"readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
|
| 6 |
+
"model_card_sha256": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524",
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
|
|
|
| 41 |
},
|
| 42 |
{
|
| 43 |
"file": "MATERIALS.json",
|
| 44 |
+
"bytes": 2064,
|
| 45 |
+
"sha256": "7c54e652ef5cd333f43171d4732d337e7d102ae6b2f0702231415585e9d60686"
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
|
|
| 56 |
},
|
| 57 |
{
|
| 58 |
"file": "QUESTION-SCALING.md",
|
| 59 |
+
"bytes": 1408,
|
| 60 |
+
"sha256": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7"
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"file": "QWEN-LICENSE",
|
|
|
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"file": "README.md",
|
| 69 |
+
"bytes": 6169,
|
| 70 |
+
"sha256": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524"
|
| 71 |
},
|
| 72 |
{
|
| 73 |
"file": "RUNTIME-RELEASE.json",
|
|
|
|
| 411 |
},
|
| 412 |
{
|
| 413 |
"file": "metrics/question-scaling.json",
|
| 414 |
+
"bytes": 4331,
|
| 415 |
+
"sha256": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
|
| 416 |
},
|
| 417 |
{
|
| 418 |
"file": "metrics/semantic-consistency.json",
|
|
|
|
| 577 |
"MATERIALS.json": "copy",
|
| 578 |
"metrics/semantic-consistency.json": "copy"
|
| 579 |
},
|
| 580 |
+
"scope": "Existing question-scaling measurement documentation only; weights, runtime, temperature, tokenizer and quality comparison unchanged.",
|
| 581 |
"release_tag": "v1.3.2",
|
| 582 |
+
"change_kind": "latency-measurement-documentation-only",
|
| 583 |
+
"previous_main_revision": "39f0f0e74ef0ec55dc729437c719547e750d02ff",
|
| 584 |
+
"previous_release_manifest_sha256": "23972247abb0ab6d54b29139db0f735b6d0599ba15011375f7e3a4893e27076c",
|
| 585 |
"presentation_amendment_sha256": "038f01d857b372a8236d0ca634e44d6d27095b5937218410b5050d5b9105a4bf",
|
| 586 |
+
"statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc",
|
| 587 |
+
"latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
|
| 588 |
}
|