Xunzhuo commited on
Commit
65a7eb4
·
verified ·
1 Parent(s): 39f0f0e

Correct question-scaling measurement documentation

Browse files
MATERIALS.json CHANGED
@@ -8,8 +8,8 @@
8
  ".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
9
  "DIAGNOSTICS.md": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73",
10
  "EVALUATION.md": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e",
11
- "QUESTION-SCALING.md": "898f3554434284ff506dbc63c34859750f07b33dcf71f5b6d4ec7a798a56eb28",
12
- "README.md": "3ba1ce73b4d3c5a6c33a36aa6d22cf26f39315383889d6ca50955991476c5b53",
13
  "SENSITIVITY.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
14
  "TASKS.md": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476",
15
  "USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
@@ -22,6 +22,7 @@
22
  "assets/decision-ranking.png": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6",
23
  "assets/decision-ranking.svg": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62",
24
  "metrics/benchmark.json": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56",
25
- "metrics/evaluation-provenance.json": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5"
 
26
  }
27
  }
 
8
  ".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
9
  "DIAGNOSTICS.md": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73",
10
  "EVALUATION.md": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e",
11
+ "QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
12
+ "README.md": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524",
13
  "SENSITIVITY.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
14
  "TASKS.md": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476",
15
  "USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
 
22
  "assets/decision-ranking.png": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6",
23
  "assets/decision-ranking.svg": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62",
24
  "metrics/benchmark.json": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56",
25
+ "metrics/evaluation-provenance.json": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5",
26
+ "metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
27
  }
28
  }
QUESTION-SCALING.md CHANGED
@@ -15,6 +15,6 @@ Nox · distinct Choice questions with **499 input tokens per question**. Only th
15
 
16
  Six independently loaded process blocks supply 30 measured requests per point after warmup. End-to-end Python request latency includes rendering, tokenization, inference, output assembly and final synchronization. Model loading and network are excluded. Requests are sequential, with no cross-request prefix cache.
17
 
18
- The measured implementation is bound byte-for-byte to the current published API, verified by an offline Hub-download proof and full 3,160-answer regression. This documentation update changes no inference code or weights. [Exact measurements and immutable runtime identity](metrics/question-scaling.json).
19
 
20
  These fixed short-input Choice measurements do not establish concurrent HTTP throughput, long-context scaling, other question-type performance or a cross-hardware speed ranking.
 
15
 
16
  Six independently loaded process blocks supply 30 measured requests per point after warmup. End-to-end Python request latency includes rendering, tokenization, inference, output assembly and final synchronization. Model loading and network are excluded. Requests are sequential, with no cross-request prefix cache.
17
 
18
+ These measurements precede the Choice null-description update. Every option in this workload supplies a description, so the update does not change these rendered inputs. Current Nox fills a null description with its option ID; this curve does not measure that case. The measured runtime previously passed an offline Hub-download proof and full 3,160-answer regression. [Measured revision and runtime scope](metrics/question-scaling.json).
19
 
20
  These fixed short-input Choice measurements do not establish concurrent HTTP throughput, long-context scaling, other question-type performance or a cross-hardware speed ranking.
README.md CHANGED
@@ -64,7 +64,7 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading *
64
 
65
  ![Question-count latency](assets/decision-question-scaling.png)
66
 
67
- Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. [p50, p95 and memory](QUESTION-SCALING.md).
68
 
69
  ## Use Nox-4B
70
 
 
64
 
65
  ![Question-count latency](assets/decision-question-scaling.png)
66
 
67
+ Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](QUESTION-SCALING.md).
68
 
69
  ## Use Nox-4B
70
 
metrics/question-scaling.json CHANGED
@@ -109,5 +109,17 @@
109
  },
110
  "timing_summary_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
111
  "API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
112
- "scope": "Only optimized path subsequently published, no old-release series. Fixed-length Python requests, excludes network and loading."
 
 
 
 
 
 
 
 
 
 
 
 
113
  }
 
109
  },
110
  "timing_summary_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
111
  "API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
112
+ "scope": "Only optimized path subsequently published, no old-release series. Fixed-length Python requests, excludes network and loading.",
113
+ "measurement_applicability": {
114
+ "measured_API_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
115
+ "current_API_sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e",
116
+ "measured_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
117
+ "current_bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
118
+ "current_API_byte_identical_to_measured": false,
119
+ "measured_criteria_all_have_non_null_descriptions": true,
120
+ "runtime_change": "Only Choice description=None is replaced with that option key; explicit descriptions use the same rendering path.",
121
+ "same_measured_points_retained": true,
122
+ "current_null_description_latency_not_measured_by_this_curve": true,
123
+ "no_HTTP_throughput_claim": true
124
+ }
125
  }
release-manifest.json CHANGED
@@ -3,7 +3,7 @@
3
  "status": "qualified-runtime-and-current-documents-assembled",
4
  "bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
5
  "readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
6
- "model_card_sha256": "3ba1ce73b4d3c5a6c33a36aa6d22cf26f39315383889d6ca50955991476c5b53",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
8
  "assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
9
  "original_bundle_manifest_preserved": false,
@@ -41,8 +41,8 @@
41
  },
42
  {
43
  "file": "MATERIALS.json",
44
- "bytes": 1959,
45
- "sha256": "0c4049929d043f9fd72f101741187eb5dd5861e2eca4b2cce02fc8193abfc414"
46
  },
47
  {
48
  "file": "NORMALIZATION_RUNTIME.md",
@@ -56,8 +56,8 @@
56
  },
57
  {
58
  "file": "QUESTION-SCALING.md",
59
- "bytes": 1274,
60
- "sha256": "898f3554434284ff506dbc63c34859750f07b33dcf71f5b6d4ec7a798a56eb28"
61
  },
62
  {
63
  "file": "QWEN-LICENSE",
@@ -66,8 +66,8 @@
66
  },
67
  {
68
  "file": "README.md",
69
- "bytes": 6069,
70
- "sha256": "3ba1ce73b4d3c5a6c33a36aa6d22cf26f39315383889d6ca50955991476c5b53"
71
  },
72
  {
73
  "file": "RUNTIME-RELEASE.json",
@@ -411,8 +411,8 @@
411
  },
412
  {
413
  "file": "metrics/question-scaling.json",
414
- "bytes": 3484,
415
- "sha256": "9407f7ebf86581cdea21e704cdbebb9ced0a485eb980925f48233ebea0f830dc"
416
  },
417
  {
418
  "file": "metrics/semantic-consistency.json",
@@ -577,11 +577,12 @@
577
  "MATERIALS.json": "copy",
578
  "metrics/semantic-consistency.json": "copy"
579
  },
580
- "scope": "Latest fifteen-model comparison and official TypeSafe SDK examples; weights, runtime, tokenizer and temperature unchanged.",
581
  "release_tag": "v1.3.2",
582
- "change_kind": "current-documents-and-official-sdk-examples-only",
583
- "previous_main_revision": "e2f752a6d4609cfdab505299127cfdc1e759c1bc",
584
- "previous_release_manifest_sha256": "abb225c9f9cfe831fb31890b263ca906041f757ead14628cca302a5d56fd84d3",
585
  "presentation_amendment_sha256": "038f01d857b372a8236d0ca634e44d6d27095b5937218410b5050d5b9105a4bf",
586
- "statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc"
 
587
  }
 
3
  "status": "qualified-runtime-and-current-documents-assembled",
4
  "bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
5
  "readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
6
+ "model_card_sha256": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
8
  "assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
9
  "original_bundle_manifest_preserved": false,
 
41
  },
42
  {
43
  "file": "MATERIALS.json",
44
+ "bytes": 2064,
45
+ "sha256": "7c54e652ef5cd333f43171d4732d337e7d102ae6b2f0702231415585e9d60686"
46
  },
47
  {
48
  "file": "NORMALIZATION_RUNTIME.md",
 
56
  },
57
  {
58
  "file": "QUESTION-SCALING.md",
59
+ "bytes": 1408,
60
+ "sha256": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7"
61
  },
62
  {
63
  "file": "QWEN-LICENSE",
 
66
  },
67
  {
68
  "file": "README.md",
69
+ "bytes": 6169,
70
+ "sha256": "ce2fc2e7277097dd4a3c859b3eac2b65185ebc1e7010bf11fe1c0ff68b477524"
71
  },
72
  {
73
  "file": "RUNTIME-RELEASE.json",
 
411
  },
412
  {
413
  "file": "metrics/question-scaling.json",
414
+ "bytes": 4331,
415
+ "sha256": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
416
  },
417
  {
418
  "file": "metrics/semantic-consistency.json",
 
577
  "MATERIALS.json": "copy",
578
  "metrics/semantic-consistency.json": "copy"
579
  },
580
+ "scope": "Existing question-scaling measurement documentation only; weights, runtime, temperature, tokenizer and quality comparison unchanged.",
581
  "release_tag": "v1.3.2",
582
+ "change_kind": "latency-measurement-documentation-only",
583
+ "previous_main_revision": "39f0f0e74ef0ec55dc729437c719547e750d02ff",
584
+ "previous_release_manifest_sha256": "23972247abb0ab6d54b29139db0f735b6d0599ba15011375f7e3a4893e27076c",
585
  "presentation_amendment_sha256": "038f01d857b372a8236d0ca634e44d6d27095b5937218410b5050d5b9105a4bf",
586
+ "statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc",
587
+ "latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
588
  }