Xunzhuo commited on
Commit
7cae429
·
verified ·
1 Parent(s): d29e9a7

Release measured v1.1 quality improvement

Browse files
Files changed (46) hide show
  1. ATTRIBUTIONS.md +11 -0
  2. EVALUATION.md +73 -113
  3. FIGURE-NOTICES.md +0 -10
  4. NORMALIZATION_RUNTIME.md +9 -0
  5. QUESTION-SCALING.md +4 -2
  6. READING-SUPPLEMENT.md +32 -0
  7. README.md +29 -14
  8. RUNTIME.md +5 -1
  9. RUNTIME_BINDING.json +54 -0
  10. SOURCE_BUNDLE_MANIFEST.json +129 -0
  11. TIMING.md +0 -164
  12. USAGE.md +4 -0
  13. assets/decision-capabilities.pdf +2 -2
  14. assets/decision-capabilities.png +2 -2
  15. assets/decision-capabilities.svg +0 -0
  16. assets/decision-quality.pdf +0 -0
  17. assets/decision-quality.png +2 -2
  18. assets/decision-quality.svg +245 -253
  19. backbone/model.safetensors +1 -1
  20. bundle-manifest.json +109 -21
  21. code/decision_api.py +27 -0
  22. code/profile_guard.py +82 -0
  23. code/runtime_profile.py +36 -0
  24. decision_head.safetensors +1 -1
  25. metrics/asset-hashes.json +0 -17
  26. metrics/natural-confirmation.json +877 -0
  27. metrics/quality-aggregate.json +0 -0
  28. metrics/timing-aggregate.json +0 -0
  29. model-card-example.json +55 -11
  30. peer-projection-provenance.json +36 -0
  31. pyproject.toml +1 -1
  32. quality-metrics.json +0 -0
  33. release-manifest.json +159 -90
  34. runtime-build-provenance.json +3 -2
  35. runtime-profile/l2norm_fwd_kernel.json +646 -0
  36. runtime-profile/profile.json +822 -0
  37. runtime.json +16 -14
  38. src/decision/example.py +3 -0
  39. src/decision/model.py +13 -2
  40. src/decision_local.egg-info/PKG-INFO +7 -0
  41. src/decision_local.egg-info/SOURCES.txt +11 -0
  42. src/decision_local.egg-info/dependency_links.txt +1 -0
  43. src/decision_local.egg-info/entry_points.txt +2 -0
  44. src/decision_local.egg-info/requires.txt +3 -0
  45. src/decision_local.egg-info/top_level.txt +1 -0
  46. temperature.json +414 -16
ATTRIBUTIONS.md CHANGED
@@ -12,3 +12,14 @@ Intent utterances retain their source labels; training converts them into varied
12
  AG News and DBpedia-14 are used only in evaluation and excluded from custom training. No raw evaluation text is included in a model bundle. Exclusion from custom training does not establish absence from base-model pretraining.
13
 
14
  Sol inherits 200 primary backbone updates, 100 decision-head warmup updates, 800 bucket-mixed Stage2 updates and 200 Stage3 updates. Nox inherits 748 primary backbone updates, 100 decision-head warmup updates, 200 Stage2 updates and 800 Stage3 updates. Head-only warmup does not update the backbone. These are different training histories, not a controlled size-only experiment. Discarded training branches are not part of either released checkpoint. Stage3 contains 47,000 newly generated training rows and 22,000 replay rows; a separate 1,000 generated rows form internal validation.
 
 
 
 
 
 
 
 
 
 
 
 
12
  AG News and DBpedia-14 are used only in evaluation and excluded from custom training. No raw evaluation text is included in a model bundle. Exclusion from custom training does not establish absence from base-model pretraining.
13
 
14
  Sol inherits 200 primary backbone updates, 100 decision-head warmup updates, 800 bucket-mixed Stage2 updates and 200 Stage3 updates. Nox inherits 748 primary backbone updates, 100 decision-head warmup updates, 200 Stage2 updates and 800 Stage3 updates. Head-only warmup does not update the backbone. These are different training histories, not a controlled size-only experiment. Discarded training branches are not part of either released checkpoint. Stage3 contains 47,000 newly generated training rows and 22,000 replay rows; a separate 1,000 generated rows form internal validation.
15
+
16
+ ## Subsequent decision adaptation
17
+
18
+ The subsequent registered curriculum adds MultiNLI human-labeled evidence judgments to programmatically verified decision tasks and replay data. The scheduled MultiNLI training subset contains 12,000 rows from the government, slate, telephone and travel genres; fiction is excluded. Original entailment/neutral/contradiction annotations are preserved. Selection and probability calibration use separate premise groups. Scheduled rows are not a claim that every selected intermediate checkpoint has seen the complete schedule.
19
+
20
+ MultiNLI is by Adina Williams, Nikita Nangia and Samuel R. Bowman, *A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference* (NAACL 2018). The pinned dataset card describes the majority of its material under the Open American National Corpus's permissive terms, with separate terms for fiction. This update uses the stated non-fiction subset and retains this attribution; no source corpus text is redistributed in the model package.
21
+
22
+ - [Pinned MultiNLI dataset and license description](https://huggingface.co/datasets/nyu-mll/multi_nli/blob/da70db2af9d09693783c3320c4249840212ee221/README.md).
23
+ - [MultiNLI paper](https://aclanthology.org/N18-1101/).
24
+
25
+ MASSIVE and SLURP remain excluded from custom training, checkpoint selection and calibration. MASSIVE is used only for evaluation. Official Jev outputs are never used as training labels. The initial-release training histories above remain historical; the subsequent selected model and evaluation provenance identify the actual update.
EVALUATION.md CHANGED
@@ -1,147 +1,107 @@
1
- # Decision 1.0 decoder evaluation
2
 
3
- Measured on 21 September 2026. These results describe the first released Sol and Nox checkpoints, compared on the same frozen requests with official Jev, released open decision models and their untuned Qwen parents. They are a bounded evaluation, not proof of universal superiority or recovery of Jev's internal architecture.
4
 
5
- ## What the main score means
 
 
 
 
 
 
 
 
6
 
7
- The main score is **native decision accuracy, averaged equally across ten task families**. The core has **880 questions, 432 semantic groups, English and Chinese, and 2–14 candidates**: 752 Choice, 64 Noul and 64 Score questions. It combines 640 constructed decision tasks, 128 official AG News test examples and 112 official DBpedia-14 test examples. The natural-intent slice has only 16 underlying groups and 64 views. No benchmark-specific adaptation was performed on these evaluation rows.
8
 
9
- The same core examples and family weights apply to every model. Choice and Score use the selected/max-probability category; Noul uses P(true) ≥ 0.5. Score accuracy is ordinal-bin accuracy, not accuracy of a rounded expected scalar. The Choice/Noul/Score columns pool examples of each type and therefore do not average to the ten-family overall score. Missing or invalid decisions count as incorrect. Different accepted subsets are not silently substituted.
10
 
11
- Confidence intervals use 10,000 percentile bootstrap resamples of semantic groups **within each family**, followed by the equal-weight family mean. Related translations and option variants remain grouped. Paired differences reuse the same sampled groups for both models. These are not simultaneous confidence intervals for every family.
12
-
13
- | Model | Overall accuracy ↑ | 95% CI | Choice ↑ | Noul ↑ | Score ↑ |
14
- | --- | ---: | ---: | ---: | ---: | ---: |
15
- | Nox · 4B | 79.32 | 75.74–82.98 | 76.06 | 90.62 | 87.50 |
16
- | Jev 1.13.0 | 79.10 | 76.17–82.05 | 72.61 | 100.00 | 100.00 |
17
- | Qwen3.5 · 4B, untuned | 69.89 | 66.23–73.40 | 68.48 | 62.50 | 90.62 |
18
- | Sol · 2B | 66.25 | 61.94–70.64 | 71.41 | 42.19 | 43.75 |
19
- | Decider · 2B | 64.01 | 59.88–68.11 | 60.77 | 71.88 | 84.38 |
20
- | Qwen3.5 · 2B, untuned | 57.12 | 53.45–60.65 | 56.38 | 37.50 | 81.25 |
21
- | Laya · EN/ML routed | 57.01 | 53.16–60.88 | 64.10 | 43.75 | 18.75 |
22
- | Laya · English | 56.54 | 52.60–60.49 | 63.70 | 43.75 | 18.75 |
23
- | Laya · Multilingual | 47.25 | 43.44–51.12 | 53.72 | 39.06 | 12.50 |
24
-
25
- All entries are accuracy percentages. “Untuned” means the exact **post-trained Qwen3.5 parent before our decision adaptation**, not a Qwen `-Base` checkpoint. Its frozen adapter scores A–Z with the pretrained LM head, with thinking disabled; no random classifier is used. The tested adapter supports at most 26 candidates. These are adapter-specific zero-additional-training results, not an upper bound on everything the parent language model can do. Comparing the release with this baseline measures the whole adaptation package (new head plus supervised backbone updates), not a pure head-only or backbone-only ablation.
26
-
27
- Laya routed uses its published input-based English/multilingual routing policy, fixed before scoring, with automatic task detection disabled. The two individual Laya checkpoints are shown separately. The official Jev API reported version 1.13.0; hosted internals and server-side truncation are not observable. We do not infer an encoder/decoder attention mask from response behavior.
28
-
29
- ## Every family
30
-
31
- | Family | Questions | Nox · 4B | Jev 1.13.0 | Qwen3.5 · 4B, untuned | Sol · 2B | Decider · 2B | Qwen3.5 · 2B, untuned | Laya · EN/ML routed |
32
- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
33
- | Natural intents | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 85.94 |
34
- | AG News | 128 | 82.03 | 85.16 | 84.38 | 82.81 | 86.72 | 80.47 | 91.41 |
35
- | DBpedia-14 | 112 | 95.54 | 96.43 | 97.32 | 93.75 | 98.21 | 92.86 | 83.93 |
36
- | Boolean constraints | 64 | 90.62 | 100.00 | 62.50 | 42.19 | 71.88 | 37.50 | 43.75 |
37
- | Ordered rubrics | 64 | 87.50 | 100.00 | 90.62 | 43.75 | 84.38 | 81.25 | 18.75 |
38
- | Relational composition | 96 | 41.67 | 56.25 | 51.04 | 45.83 | 54.17 | 48.96 | 25.00 |
39
- | Scoped evidence | 96 | 72.92 | 89.58 | 51.04 | 54.17 | 47.92 | 36.46 | 37.50 |
40
- | State tracking | 96 | 35.42 | 33.33 | 28.12 | 18.75 | 29.17 | 25.00 | 23.96 |
41
- | Unknown rejection | 64 | 87.50 | 100.00 | 60.94 | 81.25 | 59.38 | 62.50 | 64.06 |
42
- | Option carriers | 96 | 100.00 | 30.21 | 72.92 | 100.00 | 8.33 | 9.38 | 95.83 |
43
-
44
- The option-carrier family deliberately stresses binding evidence and arbitrary candidate representations. Its weight is one tenth of the headline score; this is **not** an estimate of its frequency in user workloads. AG News and DBpedia-14 favor several reference models. Relational composition and state tracking remain difficult for both releases. A high aggregate must not be read as an all-family win. The unknown-rejection slice tests whether a known value is inside or outside the supplied menu; it does not establish recognition of unknown real-world facts.
45
-
46
- ## Paired differences and release scope
47
 
48
- | Candidate | Reference | Difference (pp) | Paired 95% CI (pp) |
49
- | --- | ---: | ---: | ---: |
50
- | Nox · 4B | Jev 1.13.0 | +0.22 | -3.58 to +4.05 |
51
- | Nox · 4B | Laya · EN/ML routed | +22.31 | +17.50 to +27.08 |
52
- | Nox · 4B | Decider · 2B | +15.31 | +11.28 to +19.51 |
53
- | Nox · 4B | Qwen3.5 · 4B, untuned | +9.43 | +4.82 to +14.15 |
54
- | Sol · 2B | Jev 1.13.0 | -12.85 | -17.71 to -8.00 |
55
- | Sol · 2B | Laya · EN/ML routed | +9.24 | +4.20 to +14.22 |
56
- | Sol · 2B | Decider · 2B | +2.24 | -1.83 to +6.39 |
57
- | Sol · 2B | Qwen3.5 · 2B, untuned | +9.13 | +4.45 to +13.90 |
58
 
59
- Nox has a higher average than Laya and Decider on this panel. Its mean is close to Jev's, but the interval does not establish equivalence and its calibration and several task families remain behind. Sol's point estimate exceeds Laya and Decider, but its paired interval against Decider includes zero; it is materially below Jev overall.
60
 
61
- The original preregistered v2 policy required stricter per-family/calibration conditions; **neither candidate passed that policy**. After seeing these results, the project owner authorized first releases based on average superiority over Laya and Decider, independently for each model. This publication decision did not change the examples, weights, candidate selection, temperatures or reported measurements. The original failed gates remain in [the aggregate record](metrics/quality-aggregate.json). These now-observed panels become regression tests; later generalization claims require new independent evaluations.
62
 
63
- ## Probabilities and typed decisions
 
 
 
 
 
 
 
 
64
 
65
- Brier below is the equal-family mean of the multiclass squared-probability error (sum across classes, not divided by K). NLL uses natural logarithms without epsilon clipping; a returned zero probability for the correct class yields infinity under the reported-distribution metric. For hosted APIs with rounded outputs, this does not prove that the unobserved internal probability is exactly zero. RPS averages squared cumulative errors over K−1 ordinal boundaries. Native Score MAE divides the absolute expected-index error by K−1. All core targets are hard labels, so this release does not establish quality on soft business targets.
66
 
67
- | Model | Macro Brier ↓ | Choice NLL ↓ | False recall ↑ | True recall ↑ | Noul balanced ↑ | Score RPS ↓ | Score MAE ↓ |
68
- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
69
- | Nox · 4B | 0.3577 | 2.0021 | 85.00 | 100.00 | 92.50 | 0.0291 | 0.0302 |
70
- | Jev 1.13.0 | 0.2481 | infinity | 100.00 | 100.00 | 100.00 | 0.0000 | 0.0000 |
71
- | Qwen3.5 · 4B, untuned | 0.4271 | 0.8927 | 40.00 | 100.00 | 70.00 | 0.0214 | 0.0484 |
72
- | Sol · 2B | 0.5187 | 1.4094 | 10.00 | 95.83 | 52.92 | 0.1946 | 0.2638 |
73
- | Decider · 2B | 0.4425 | 0.8561 | 55.00 | 100.00 | 77.50 | 0.0730 | 0.1744 |
74
- | Qwen3.5 · 2B, untuned | 0.5724 | 1.1525 | 0.00 | 100.00 | 50.00 | 0.0560 | 0.1033 |
75
- | Laya · EN/ML routed | 0.5710 | infinity | 20.00 | 83.33 | 51.67 | 0.2003 | 0.3122 |
76
- | Laya · English | 0.5708 | infinity | 20.00 | 83.33 | 51.67 | 0.2003 | 0.3122 |
77
- | Laya · Multilingual | 0.6314 | 1.0724 | 7.50 | 91.67 | 49.58 | 0.2817 | 0.4104 |
78
 
79
- Noul has 40 false and 24 true cases. In particular, Sol's false recall is only 10% here: a syntactically valid yes/no answer is not a correctness guarantee. Do not use its raw confidence as an automatic escalation threshold without application-specific validation.
 
 
 
 
 
 
 
 
80
 
81
- Both checkpoints keep the **single temperature fitted on development data before final evaluation**: Sol T=0.7033302993804421; Nox T=0.5332910931023557. On this final panel, the shipped temperatures worsen family-macro Brier relative to raw T=1 (Sol 0.5187 vs 0.4857; Nox 0.3577 vs 0.3253). They were not switched after seeing final results. Jev's corresponding Brier is 0.2481. The aggregate preserves raw and shipped proper scores separately. “Confidence” in the public Choice/Score schema is (K·max(p)−1)/(K−1), a concentration statistic, not a proven calibrated probability of correctness or a recovered Jev formula.
82
 
83
- An analytic uniform-probability control and its fixed-tie versus expected-random accuracy are included in the aggregate. A global training-class prior is not meaningful for request-specific arbitrary candidate IDs and is omitted with that reason.
84
 
85
- ## Input support and separate native probe
86
 
87
- Core probabilities are valid on all 880 questions for every primary model. Sol, Nox, Decider and the untuned Qwen adapters reported no core truncations. Each Laya variant reported truncation on 112 core questions; this is a comparison of the released interfaces on identical requests, not a claim that every implementation consumed identical token content. Jev does not expose enough information to verify server-side truncation.
88
 
89
- The independent native probe has **68 requests containing 220 questions**. It includes structured inputs, multiple questions, lengths and candidate counts up to 255. It is reported separately and does not contribute to the main ranking. Validity and semantic correctness are different measurements; unsupported requests remain in the all-requested denominator.
90
 
91
- | Model | Accepted requests / 68 | Valid questions / 220 | Correct questions / 220 |
92
- | --- | ---: | ---: | ---: |
93
- | Nox · 4B | 68 | 220 | 162 |
94
- | Jev 1.13.0 | 68 | 220 | 203 |
95
- | Qwen3.5 · 4B, untuned | 44 | 196 | 112 |
96
- | Sol · 2B | 68 | 220 | 116 |
97
- | Decider · 2B | 68 | 220 | 111 |
98
- | Qwen3.5 · 2B, untuned | 44 | 196 | 86 |
99
- | Laya · EN/ML routed | 56 | 208 | 88 |
100
- | Laya · English | 56 | 208 | 88 |
101
- | Laya · Multilingual | 62 | 214 | 58 |
102
 
103
- Sol and Nox both accepted all questions without truncation. Each scored 23/42 on the large-candidate probe and 9/9 on the length probe. Nox scored 162/220 and Sol 116/220 overall, versus Jev's 203/220. This is a substantial remaining native-capability gap despite Nox's similar core macro accuracy. The untuned Qwen adapter rejected candidate sets beyond A–Z explicitly. Laya's accepted subsets differ and are not interchangeable paired comparisons.
104
 
105
- The Decision engine supports Choice 2–255, Noul false/true, and Score 2–10 ordered descriptions. Score returns an expected **index**; arbitrary supplied numeric values are not implemented in this release. Each complete question—including state, instructions, all candidates and special tokens—must fit 16,384 tokens. Overflow rejects the complete call before inference; it is never silently truncated. This is a verified input budget, not a claim of uniformly strong reasoning across the full context. Questions execute independently in batches of eight with no cross-question state-prefix cache.
106
 
107
- ## Additional open references
108
 
109
- These models used the same core and the matched FLA runtime where relevant. They are additional measured references, not additional release gates or evidence of significant wins against every model.
110
 
111
- | Open reference | Macro accuracy (%) | 95% CI |
112
- | --- | ---: | ---: |
113
- | kev-0.8b | 60.70 | 56.88–64.40 |
114
- | kev-4b | 71.28 | 67.41–75.10 |
115
- | kev-9b | 76.36 | 72.94–79.81 |
116
- | nimble-9b | 78.85 | 75.65–82.02 |
117
- | llm2jev-2b | 62.60 | 59.08–66.08 |
118
- | llm2jev-4b | 69.91 | 66.36–73.41 |
119
 
120
- ## Training, selection and provenance
121
 
122
- The text-only backbones are Qwen/Qwen3.5-2B at `15852e8c16360a2fea060d615a32b45270f8a8fc` and Qwen/Qwen3.5-4B at `851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a`. Both are post-trained parents. The shared candidate head is trained with the text backbone. Selected checkpoints were frozen before final scoring: Sol at Stage3 step 200 and Nox at Stage3 step 800, selected among the scheduled checkpoints using production batch-size-eight development accuracy. They have unequal inherited training histories; the comparison does not isolate model size as the sole cause of differences.
123
 
124
- Stage3 uses 69,000 training rows: 47,000 new generated rows plus 22,000 replay rows; a separate 1,000 generated rows are internal validation. Training mixes programmatically verified decision tasks with licensed BANKING77 and CLINC150 training data, using varied instructions, structured views and candidate permutations. Reserved intent labels/domains are excluded from adaptation. AG News and DBpedia-14 are evaluation-only. The Stage3 overlap audit found no exact or token-Jaccard ≥0.5 matches against the final core; short overlaps with earlier development intent examples are a separate development limitation. Upstream pretraining exposure remains unknown. No official Jev predictions were used as training labels. Source licenses and exact training lineage are in [ATTRIBUTIONS.md](ATTRIBUTIONS.md).
125
 
126
- The actual architecture and numerical inference source are shipped with each model. Standalone export/reload with network access disabled reproduced all 256 development logits bit-exactly; emitted probabilities differed by at most 2.384×10⁻⁷ and selected decisions did not change. Real packaged Choice/Noul/Score inference also passed in the publicly rebuildable ROCm runtime; see [RUNTIME.md](RUNTIME.md). That small runtime equivalence probe is distinct from a complete benchmark rerun.
 
 
 
 
 
 
 
 
127
 
128
- The Laya checkpoint is `convaiinnovations/laya` at `1c5edc17a7acd8701df6fc341c0d179f1c62c982` (English root and multilingual subfolder). Its native router follows source commit `42626c348753fbb17572a813127df2278a1ec527`. Decider is `Mapika/decider-2b` at `b37f7e1ba3fbc9238004cf531fabbee2619973fd`; its released numerical interface is retained. Full aggregate metrics and immutable source/prediction fingerprints are in [metrics/quality-aggregate.json](metrics/quality-aggregate.json).
129
 
130
- ## Efficiency
 
 
 
131
 
132
- The completed quiet benchmark uses one AMD gfx942 accelerator with 261824 MiB visible memory. It includes all six local primary models, independently randomized model order in three blocks, 30 repetitions per supported cell, and separate Choice and Noul/Score panels. All 36 quality/runtime identity checks match. The 114 complete-input cells provide 3,420 measured request times; truncated and unsupported inputs are explicit exclusions rather than short-input substitutions.
133
 
134
- Latency includes rendering, tokenization, device transfers, model forward and answer assembly, but excludes loading, diagnostics, warmup, network and post-response shape checks. No other training/inference GPU process ran during measurement. CPU affinity and clocks were not locked, and telemetry counters on otherwise idle devices occasionally reported 1–3%; this is an empirical quiet request benchmark, not a noise-free hardware microbenchmark. After a transient process was caught by preflight before job 26, only the remaining jobs resumed; no timing job was repeated.
135
 
136
- | Model | Actual tokens | p50 / p95 (ms) | Peak allocated GiB |
137
- | --- | ---: | ---: | ---: |
138
- | Nox · 4B | 309 | 27.37 / 28.95 | 8.06 |
139
- | Qwen3.5 · 4B, untuned | 296 | 27.47 / 28.76 | 8.04 |
140
- | Sol · 2B | 309 | 21.04 / 21.41 | 3.61 |
141
- | Decider · 2B | 250 | 23.35 / 23.91 | 3.62 |
142
- | Qwen3.5 · 2B, untuned | 296 | 20.51 / 20.71 | 3.60 |
143
- | Laya · EN/ML routed | 228 | 11.54 / 12.21 | 3.56 |
144
 
145
- This table uses the same short English state, one question and four options. Native token counts vary with the prompt/template. Laya uses its English branch. Sol is faster than Decider here, but Decider is faster on multiquestion Choice and K=255; neither Decision release establishes universal speed superiority. Nox trades higher quality on the measured panel for more memory and latency. Hosted Jev wall time includes service and network cost and is not included in this local timing comparison.
146
 
147
- See [TIMING.md](TIMING.md) for every length/question-count/candidate-count/type cell, questions/s and peak-memory definitions, and [metrics/timing-aggregate.json](metrics/timing-aggregate.json) for machine-readable measurements and runtime pairing. The full request benchmark and the small publicly rebuilt runtime parity test are separate evidence.
 
1
+ # Sol: measured quality update
2
 
3
+ The primary score averages 20 task-family accuracies equally: one half the original 880-question panel and one half the independently constructed 880-question confirmation panel. The original panel was already observed during development. The added panel was excluded from custom training, selection and calibration; candidate weights, prompt and temperature were fixed before its inference. Once unsealed for the first candidate, it remains an observed regression panel. A later release does not make it fresh again.
4
 
5
+ | Model | Overall accuracy ↑ | Choice ↑ | Noul ↑ | Score ↑ |
6
+ | --- | ---: | ---: | ---: | ---: |
7
+ | Jev · 1.13.0 | **72.74** | **71.77** | **80.36** | **68.06** |
8
+ | Nox · 4B · v1.1 | 67.05 | 70.55 | 60.27 | 55.56 |
9
+ | Sol · 2B · v1.1 | 60.14 | 66.67 | 48.66 | 40.28 |
10
+ | Qwen3.5 · 4B · untuned | 56.61 | 60.06 | 53.57 | 52.08 |
11
+ | Decider · 2B | 55.30 | 57.26 | 58.93 | 50.00 |
12
+ | Qwen3.5 · 2B · untuned | 48.06 | 50.36 | 45.09 | 47.22 |
13
+ | Laya · EN/ML | 47.38 | 53.02 | 52.23 | 19.44 |
14
 
15
+ All values are percentages. Choice, Noul and Score columns pool correct/total questions of that type across both panels; their mean is not the overall family-macro score. Score accuracy tests the highest-probability rubric level; expected-value error is reported separately. Missing or invalid predictions count as incorrect. Confidence intervals use 10,000 paired cluster-bootstrap resamples. The two language views of each MASSIVE source utterance share draws within menu strata; synthetic counterfactual blocks remain intact. The original and confirmation panels are resampled independently, then combined with equal weight.
16
 
17
+ ## Change from the released weights
18
 
19
+ | Panel | Accuracy change (points) | Paired 95% interval |
20
+ | --- | ---: | ---: |
21
+ | Old | +7.07 | [+3.89, +10.33] |
22
+ | Fresh | +2.04 | [-0.62, +4.58] |
23
+ | Joint | +4.56 | [+2.47, +6.62] |
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
 
25
+ Release eligibility follows the prospectively fixed positive joint point improvement plus complete, finite, valid candidate execution. It does not require every family to improve or the confidence interval to exclude zero. An eligible update is not evidence of universal superiority.
 
 
 
 
 
 
 
 
 
26
 
27
+ ## Native API coverage
28
 
29
+ Each separate native panel has 68 requests and 220 requested answers, including repeated packing and isolation probes. These answers are descriptive diagnostics, not 220 independent questions and not part of the primary score. Unsupported requests are reported with the full denominator rather than relabeled as wrong semantic predictions.
30
 
31
+ | Model | Original: valid / 220 | Original: correct / 220 | New: valid / 220 | New: correct / 220 |
32
+ | --- | ---: | ---: | ---: | ---: |
33
+ | Nox · 4B · v1.1 | 220 | 200 | 220 | 138 |
34
+ | Sol · 2B · v1.1 | 220 | 138 | 220 | 127 |
35
+ | Jev · 1.13.0 | 220 | 203 | 220 | 168 |
36
+ | Laya · EN/ML | 208 | 88 | 208 | 98 |
37
+ | Decider · 2B | 220 | 111 | 220 | 96 |
38
+ | Qwen3.5 · 2B · untuned | 196 | 86 | 196 | 71 |
39
+ | Qwen3.5 · 4B · untuned | 196 | 112 | 196 | 104 |
40
 
41
+ ## Scope and interpretation
42
 
43
+ All comparators receive the same frozen requests through documented native or fixed adaptation interfaces. The untuned baselines are post-trained Qwen3.5 parents before decision adaptation, evaluated with a fixed LM-head decision adapter. Laya uses fixed English/multilingual routing. Jev is the official API; server-side truncation and internal model architecture cannot be independently inspected.
 
 
 
 
 
 
 
 
 
 
44
 
45
+ | Model | Original reported truncations | New reported truncations |
46
+ | --- | ---: | ---: |
47
+ | Nox · 4B · v1.1 | 0 | 0 |
48
+ | Sol · 2B · v1.1 | 0 | 0 |
49
+ | Jev · 1.13.0 | 0 | 0 |
50
+ | Laya · EN/ML | 112 | 40 |
51
+ | Decider · 2B | 0 | 0 |
52
+ | Qwen3.5 · 2B · untuned | 0 | 0 |
53
+ | Qwen3.5 · 4B · untuned | 0 | 0 |
54
 
55
+ Zero reported truncations for a remote service is not proof of complete server-side input processing. The Decision candidate additionally passes explicit complete-token and overflow checks.
56
 
57
+ The confirmation panel includes unseen MASSIVE English/Chinese intents and eight independently generated decision families: temporal exclusion, constraint assignment, transaction recovery, conflicting-rule reasoning, multiset reconciliation, record identity, capacity constraints and service-loss scoring. Training, selection and calibration exclude MASSIVE and SLURP. Exact overlap was zero; three moderate lexical near-pairs with training were disclosed and retained. Gold labels were not supplied by Jev.
58
 
59
+ MASSIVE uses the [official 1.1 archive](https://amazon-massive-nlu-dataset.s3.amazonaws.com/amazon-massive-dataset-1.1.tar.gz), CC BY 4.0. Exposure during base-model pretraining is unknown. Synthetic English/Chinese views share symbolic JSON state; they do not establish natural multilingual document comprehension.
60
 
61
+ The [complete aggregate measurements](quality-metrics.json) retain every family, raw and calibrated probability diagnostics, native coverage, failure counts, Brier/NLL and ordinal metrics. Proper probability scores are conditional on valid probability coverage; zero probability on the true answer produces explicit infinite NLL rather than clipping. Per-family regressions remain visible in the capability matrix.
62
 
63
+ The question-scaling figure retains measurements from v1.0 weights and is labeled accordingly. The architecture and answer semantics are unchanged. Sol v1.1 additionally binds an explicit FLA normalization profile after portable reload exposed a real autotuning difference. Its selected weights, prompt and temperature are unchanged by this runtime binding. The older curve is not a timing claim for these weights or this profile.
64
 
65
+ Evaluation policy SHA-256: `359de2b6b81be91a66c6c538b3655bed06f4728357328cd318b1a765e64565cd`. Scoring receipt SHA-256: `bca099653d5b92cf282a5a82c19998c65dae0ac89d6fd039123b60b3680e85b3`.
 
 
 
 
 
 
 
 
 
 
66
 
67
+ ## Peer revision and aggregate provenance
68
 
69
+ The displayed Nox peer is the already published [Nox v1.1](https://huggingface.co/llm-semantic-router/Decision-1.0-Nox/tree/5bee3061a7abf675813963e4e05267cb1a28f92f). Its 67.05% overall score, family counts, native coverage and absolute confidence intervals come from its separately completed evaluation on the exact same frozen inputs, family weights and bootstrap implementation. This replaces the historical Nox v1.0 peer only in this public comparison projection. No new Sol-versus-Nox paired significance claim is made.
70
 
71
+ Sol's qualification remains its original comparison against Sol v1.0: 55.58% to 60.14%, +4.56 points, paired 95% interval [+2.47,+6.62]. Its original-panel change is +7.07 points; its later-panel change is +2.04 points with interval [−0.63,+4.58]. The global first exposure occurred for Nox before Sol's selected checkpoint was frozen, so both panels are observed regression evidence for this Sol release.
72
 
73
+ [Projection provenance](peer-projection-provenance.json) binds both source reports and Nox's verified publication. Sol evaluation receipt SHA-256: `bca099653d5b92cf282a5a82c19998c65dae0ac89d6fd039123b60b3680e85b3`. Nox evaluation receipt SHA-256: `927c92121c2df534f2ba12523e466ecebd8bd8936f949c7792de3e747c92a47c`. The original qualification, receipt and statistics remain unchanged.
74
 
 
 
 
 
 
 
 
 
75
 
76
+ ## Reading supplied evidence: observed V4 supplement
77
 
78
+ Sol v1.1 scores **73.44%**, up **4.06 percentage points** from Sol v1.0; paired 95% CI **[+1.10, +7.03]pp**. This is observed regression evidence, supplementary to the unchanged V3 publication qualification. Reading remains a gap against Jev, Decider and the untuned 4B LM-head baseline.
79
 
80
+ The fixed score is **50% BoolQ + 25% Belebele English + 25% Belebele Chinese**. Each family has 160 questions; BoolQ contains 80 false and 80 true labels. All 480 requested examples remain in the denominator. Bold marks column leaders. Weighted accuracy differs from raw correct/480.
81
 
82
+ | Model | BoolQ % (correct) | Belebele EN % (correct) | Belebele ZH % (correct) | Weighted % | Weighted 95% CI | Raw correct |
83
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
84
+ | Jev · 1.13.0 | **92.50 (148/160)** | **96.88 (155/160)** | **96.25 (154/160)** | **94.53** | [92.01, 96.76] | 457/480 |
85
+ | Decider · 2B | 91.88 (147/160) | 93.12 (149/160) | 91.25 (146/160) | 92.03 | [89.17, 94.65] | 442/480 |
86
+ | Nox · 4B · v1.1 | 87.50 (140/160) | 71.25 (114/160) | 67.50 (108/160) | 78.44 | [73.99, 82.65] | 362/480 |
87
+ | Sol · 2B · v1.1 | 79.38 (127/160) | 68.12 (109/160) | 66.88 (107/160) | 73.44 | [68.69, 77.96] | 343/480 |
88
+ | Qwen3.5 · 4B · untuned | 83.75 (134/160) | 93.12 (149/160) | 91.25 (146/160) | 87.97 | [84.53, 91.25] | 429/480 |
89
+ | Qwen3.5 · 2B · untuned | 65.00 (104/160) | 83.75 (134/160) | 81.25 (130/160) | 73.75 | [69.18, 78.23] | 368/480 |
90
+ | Laya · EN/ML | 69.38 (111/160) | 38.12 (61/160) | 28.12 (45/160) | 51.25 | [46.60, 55.84] | 217/480 |
91
 
92
+ ## Sol version change
93
 
94
+ | Version | BoolQ | Belebele EN | Belebele ZH | Natural Choice mean | Weighted |
95
+ | --- | ---: | ---: | ---: | ---: | ---: |
96
+ | Sol v1.0 | 119/160 · 74.38% | 106/160 · 66.25% | 100/160 · 62.50% | 64.38% | 69.38% |
97
+ | Sol v1.1 | 127/160 · 79.38% | 109/160 · 68.12% | 107/160 · 66.88% | 67.50% | 73.44% |
98
 
99
+ The exact weighted delta is **13/320**. Calibration is the frozen production temperature; no parameters or prompts were changed using this panel. This Sol run used the default public `from_pretrained` entry point with its automatically bound, validated normalization profile. Nox is the already published **v1.1** peer, from the identical frozen panel and statistical implementation; its separate source receipt is linked in the aggregate evidence. Untuned Qwen models use their original vocabulary heads and fixed LM-head adapters, not random decision heads.
100
 
101
+ All seven models returned **480/480 finite, valid answers**. The six local models retained all inputs, with zero truncations; Jev's internal retention is unknown. These concurrent-workload measurements establish quality only and provide no latency comparison.
102
 
103
+ Intervals use 10,000 paired passage-cluster bootstrap draws. BoolQ has 160 passage parents; Belebele has 144 shared passage parents. All questions and both language translations belonging to each Belebele passage are resampled together. Identical draws are shared across models. The machine-readable [aggregate](metrics/natural-confirmation.json) includes family intervals, exact weighted fractions, Noul/Choice probability metrics, coverage and source hashes. Zero gold probability produces infinite NLL; no epsilon clipping is used.
 
 
 
 
 
 
 
104
 
105
+ [BoolQ validation](https://huggingface.co/datasets/google/boolq/tree/35b264d03638db9f4ce671b711558bf7ff0f80d5) (CC BY-SA 3.0) and [Belebele test](https://huggingface.co/datasets/facebook/belebele/tree/7899cdfa4e1e0d733fd77c848e2c273cb1d32be2) (CC BY-SA 4.0) were excluded from custom TRAIN/SELECT/CAL. Original human labels and parallel alignment were independently checked; they were not newly adjudicated. The lexical audit found no exact or character-5-gram Jaccard ≥0.5 overlaps across 51 earlier data/panel files. This does not prove semantic independence or exclude unknown upstream-pretraining exposure. BoolQ lacks article-title metadata, so its clustering is by passage, not article.
106
 
107
+ The permanent first-exposure ledger predates this Sol candidate freeze: this is **observed regression**, not a newly unseen confirmation result for Sol. This panel covers evidence yes/no and four-choice comprehension; it does not measure ordinal Score or establish universal decision capability. It adds no release gate and leaves the existing qualification receipt unchanged.
FIGURE-NOTICES.md DELETED
@@ -1,10 +0,0 @@
1
- # Figure provenance
2
-
3
- The architecture figures describe the shipped Decision implementation. Backbone operator details were checked against the actual Qwen3.5 implementation in Transformers 5.17.0 and the pinned parent configurations. They do not claim to reveal Jev's unpublished architecture.
4
-
5
- - [Qwen3.5-2B configuration](https://huggingface.co/Qwen/Qwen3.5-2B/blob/15852e8c16360a2fea060d615a32b45270f8a8fc/config.json)
6
- - [Qwen3.5-4B configuration](https://huggingface.co/Qwen/Qwen3.5-4B/blob/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a/config.json)
7
- - [Transformers source and Apache 2.0 license](https://github.com/huggingface/transformers)
8
- - [Decision family visual guide](https://gist.github.com/Xunzhuo/4020f574e3d38e5e5eae00063bdf9dce)
9
-
10
- The rank and capability figures are generated from the same aggregate evidence linked in EVALUATION.md. The supplied Decision mark is included unchanged; chart backgrounds are white. SVG and PDF preserve the chart/architecture geometry and typography as vectors; the branding mark is a raster asset.
 
 
 
 
 
 
 
 
 
 
 
NORMALIZATION_RUNTIME.md ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ # Validated normalization runtime
2
+
3
+ This Sol bundle automatically loads an explicit FLA normalization launch profile before importing FLA. The validated environment is the recorded ROCm runtime on gfx942 GPUs. No private compilation cache or manual environment setting is required. Use a fresh Python process; repeated loading of this same profile is allowed, while mixing profiled and unprofiled models in one process is rejected.
4
+
5
+ The profile fixes only the FLA l2norm forward launch settings for BF16 input/output, FP32 reciprocal norms, head dimension 128, and normalization buckets 1–32. This covers the production batch size of at most 8 and complete-question limit of 16,384 tokens with 16 key heads. Unmatched shapes/dtypes and alternate normalization kernels fail closed. Other FLA kernels keep the recorded runtime behavior; this is not a claim that every FLA kernel is deterministic.
6
+
7
+ Buckets previously observed during production preserve their chosen configurations. Previously unobserved buckets use the prospectively specified configuration in the profile. Boundary tests establish finite execution and dispatch coverage, not numerical equivalence to an unobserved previous automatic choice. Weights, tokenizer, prompt, readout and calibration are unchanged from the selected candidate.
8
+
9
+ This runtime has its own numerical validation. Earlier release latency measurements must not be treated as measurements of this profile.
QUESTION-SCALING.md CHANGED
@@ -1,4 +1,6 @@
1
- # Fixed-question latency scaling
 
 
2
 
3
  A fixed English state (819 characters), identical question text and a four-choice menu are repeated Q times in one API request. Native rendered rows, including state and question, contain 309 tokens for Sol/Nox, 228 for Laya and 250 for Decider. Tokenizers and templates differ. Every model uses the same physical AMD gfx942 GPU and matches its quality/runtime fingerprint. Native question packing is preserved, with no cross-request cache or concurrent-client load.
4
 
@@ -8,7 +10,7 @@ Q1/Q8/Q32 come from the original GPU-quiet round; Q2/Q4/Q16 are an additive roun
8
 
9
  Base-model appendix retains the original measured Q1/Q8/Q32 points only.
10
 
11
- The machine exposes 261824 MiB of device memory. Parameters, native precision, shipped temperature and model/runtime fingerprints are unchanged from the quality benchmark. No throughput saturation claim is made. p95 is an empirical percentile, not a confidence interval.
12
 
13
  Full aggregate evidence: [question-scaling.json](metrics/question-scaling.json).
14
 
 
1
+ # Fixed-question latency scaling — v1.0
2
+
3
+ These measurements use the initial Sol/Nox v1.0 weights. They are retained as versioned evidence and do not measure the updated weights.
4
 
5
  A fixed English state (819 characters), identical question text and a four-choice menu are repeated Q times in one API request. Native rendered rows, including state and question, contain 309 tokens for Sol/Nox, 228 for Laya and 250 for Decider. Tokenizers and templates differ. Every model uses the same physical AMD gfx942 GPU and matches its quality/runtime fingerprint. Native question packing is preserved, with no cross-request cache or concurrent-client load.
6
 
 
10
 
11
  Base-model appendix retains the original measured Q1/Q8/Q32 points only.
12
 
13
+ The machine exposes 261824 MiB of device memory. Parameters, native precision, shipped temperature and model/runtime fingerprints are those of the original v1.0 quality benchmark. No throughput saturation claim is made. p95 is an empirical percentile, not a confidence interval.
14
 
15
  Full aggregate evidence: [question-scaling.json](metrics/question-scaling.json).
16
 
READING-SUPPLEMENT.md ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Reading supplied evidence: observed V4 supplement
2
+
3
+ Sol v1.1 scores **73.44%**, up **4.06 percentage points** from Sol v1.0; paired 95% CI **[+1.10, +7.03]pp**. This is observed regression evidence, supplementary to the unchanged V3 publication qualification. Reading remains a gap against Jev, Decider and the untuned 4B LM-head baseline.
4
+
5
+ The fixed score is **50% BoolQ + 25% Belebele English + 25% Belebele Chinese**. Each family has 160 questions; BoolQ contains 80 false and 80 true labels. All 480 requested examples remain in the denominator. Bold marks column leaders. Weighted accuracy differs from raw correct/480.
6
+
7
+ | Model | BoolQ % (correct) | Belebele EN % (correct) | Belebele ZH % (correct) | Weighted % | Weighted 95% CI | Raw correct |
8
+ | --- | ---: | ---: | ---: | ---: | ---: | ---: |
9
+ | Jev · 1.13.0 | **92.50 (148/160)** | **96.88 (155/160)** | **96.25 (154/160)** | **94.53** | [92.01, 96.76] | 457/480 |
10
+ | Decider · 2B | 91.88 (147/160) | 93.12 (149/160) | 91.25 (146/160) | 92.03 | [89.17, 94.65] | 442/480 |
11
+ | Nox · 4B · v1.1 | 87.50 (140/160) | 71.25 (114/160) | 67.50 (108/160) | 78.44 | [73.99, 82.65] | 362/480 |
12
+ | Sol · 2B · v1.1 | 79.38 (127/160) | 68.12 (109/160) | 66.88 (107/160) | 73.44 | [68.69, 77.96] | 343/480 |
13
+ | Qwen3.5 · 4B · untuned | 83.75 (134/160) | 93.12 (149/160) | 91.25 (146/160) | 87.97 | [84.53, 91.25] | 429/480 |
14
+ | Qwen3.5 · 2B · untuned | 65.00 (104/160) | 83.75 (134/160) | 81.25 (130/160) | 73.75 | [69.18, 78.23] | 368/480 |
15
+ | Laya · EN/ML | 69.38 (111/160) | 38.12 (61/160) | 28.12 (45/160) | 51.25 | [46.60, 55.84] | 217/480 |
16
+
17
+ ## Sol version change
18
+
19
+ | Version | BoolQ | Belebele EN | Belebele ZH | Natural Choice mean | Weighted |
20
+ | --- | ---: | ---: | ---: | ---: | ---: |
21
+ | Sol v1.0 | 119/160 · 74.38% | 106/160 · 66.25% | 100/160 · 62.50% | 64.38% | 69.38% |
22
+ | Sol v1.1 | 127/160 · 79.38% | 109/160 · 68.12% | 107/160 · 66.88% | 67.50% | 73.44% |
23
+
24
+ The exact weighted delta is **13/320**. Calibration is the frozen production temperature; no parameters or prompts were changed using this panel. This Sol run used the default public `from_pretrained` entry point with its automatically bound, validated normalization profile. Nox is the already published **v1.1** peer, from the identical frozen panel and statistical implementation; its separate source receipt is linked in the aggregate evidence. Untuned Qwen models use their original vocabulary heads and fixed LM-head adapters, not random decision heads.
25
+
26
+ All seven models returned **480/480 finite, valid answers**. The six local models retained all inputs, with zero truncations; Jev's internal retention is unknown. These concurrent-workload measurements establish quality only and provide no latency comparison.
27
+
28
+ Intervals use 10,000 paired passage-cluster bootstrap draws. BoolQ has 160 passage parents; Belebele has 144 shared passage parents. All questions and both language translations belonging to each Belebele passage are resampled together. Identical draws are shared across models. The machine-readable [aggregate](metrics/natural-confirmation.json) includes family intervals, exact weighted fractions, Noul/Choice probability metrics, coverage and source hashes. Zero gold probability produces infinite NLL; no epsilon clipping is used.
29
+
30
+ [BoolQ validation](https://huggingface.co/datasets/google/boolq/tree/35b264d03638db9f4ce671b711558bf7ff0f80d5) (CC BY-SA 3.0) and [Belebele test](https://huggingface.co/datasets/facebook/belebele/tree/7899cdfa4e1e0d733fd77c848e2c273cb1d32be2) (CC BY-SA 4.0) were excluded from custom TRAIN/SELECT/CAL. Original human labels and parallel alignment were independently checked; they were not newly adjudicated. The lexical audit found no exact or character-5-gram Jaccard ≥0.5 overlaps across 51 earlier data/panel files. This does not prove semantic independence or exclude unknown upstream-pretraining exposure. BoolQ lacks article-title metadata, so its clustering is by passage, not article.
31
+
32
+ The permanent first-exposure ledger predates this Sol candidate freeze: this is **observed regression**, not a newly unseen confirmation result for Sol. This panel covers evidence yes/no and four-choice comprehension; it does not measure ordinal Score or establish universal decision capability. It adds no release gate and leaves the existing qualification receipt unchanged.
README.md CHANGED
@@ -18,6 +18,7 @@ tags:
18
  datasets:
19
  - PolyAI/banking77
20
  - clinc/clinc_oos
 
21
  ---
22
 
23
  ![Decision 1.0 — Your move.](assets/decision-family-header.png)
@@ -46,35 +47,49 @@ Your question names and candidate IDs are preserved. Labels are defined at runti
46
 
47
  ## Measured capability
48
 
49
- **66.25% overall**, +9.13 points over its untuned parent. Its mean exceeds Laya and Decider; the paired interval against Decider includes zero.
50
 
51
  | Model | Overall accuracy ↑ | Choice ↑ | Noul ↑ | Score ↑ |
52
  | --- | ---: | ---: | ---: | ---: |
53
- | Nox · 4B | **79.32** | **76.06** | 90.62 | 87.50 |
54
- | Jev 1.13.0 | 79.10 | 72.61 | **100.00** | **100.00** |
55
- | Qwen3.5 · 4B, untuned | 69.89 | 68.48 | 62.50 | 90.62 |
56
- | Sol · 2B | 66.25 | 71.41 | 42.19 | 43.75 |
57
- | Decider · 2B | 64.01 | 60.77 | 71.88 | 84.38 |
58
- | Qwen3.5 · 2B, untuned | 57.12 | 56.38 | 37.50 | 81.25 |
59
- | Laya · EN/ML | 57.01 | 64.10 | 43.75 | 18.75 |
60
 
61
- Accuracy (%). Overall averages **10 families / 880 questions** equally; type columns pool their questions. Same requests for every model. Untuned means the post-trained parent before decision adaptation. Laya's native interface truncated 112 inputs. [Methods and uncertainty](EVALUATION.md).
62
 
63
  ![Decision quality ranking](assets/decision-quality.png)
64
 
65
- ![Capability comparison across ten families](assets/decision-capabilities.png)
66
 
67
- False judgments, rubric scoring and state tracking remain weak; Jev is substantially stronger overall.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
68
 
69
  ## More questions, measured
70
 
71
  ![Latency as questions per request increase](assets/decision-question-scaling.png)
72
 
73
- Same state, same question length, four choices; only the number of questions changes. Each model keeps its native interface (Sol: 309 tokens/question). Median of 30 requests per point on one AMD gfx942 GPU; local Python request time includes tokenization and inference, excluding loading and network. [p95, drift checks and full methods](QUESTION-SCALING.md).
74
 
75
  ## Try it
76
 
77
- Download the fixed release with `hf download llm-semantic-router/Decision-1.0-Sol --revision v1.0 --local-dir decision-model`, then follow the [ROCm setup](RUNTIME.md). Inside that container, with the model mounted at `/model`:
78
 
79
  ```python
80
  from decision import DecisionModel
@@ -86,7 +101,7 @@ print(model.decide(**REQUEST)["answers"])
86
 
87
  [Actual request and measured output](model-card-example.json) · [Install and API guide](USAGE.md)
88
 
89
- The complete state, question and candidates must fit 16,384 tokens. Overflow is rejected. Score returns an expected ordinal index. Runtime: AMD gfx942 validated; CPU/MPS unsupported; NVIDIA unqualified.
90
 
91
  ## Architecture
92
 
 
18
  datasets:
19
  - PolyAI/banking77
20
  - clinc/clinc_oos
21
+ - nyu-mll/multi_nli
22
  ---
23
 
24
  ![Decision 1.0 — Your move.](assets/decision-family-header.png)
 
47
 
48
  ## Measured capability
49
 
50
+ **60.14% overall**, up 4.56 points from Sol v1.0 across 20 equally weighted capabilities.
51
 
52
  | Model | Overall accuracy ↑ | Choice ↑ | Noul ↑ | Score ↑ |
53
  | --- | ---: | ---: | ---: | ---: |
54
+ | Jev · 1.13.0 | **72.74** | **71.77** | **80.36** | **68.06** |
55
+ | Nox · 4B · v1.1 | 67.05 | 70.55 | 60.27 | 55.56 |
56
+ | Sol · 2B · v1.1 | 60.14 | 66.67 | 48.66 | 40.28 |
57
+ | Qwen3.5 · 4B · untuned | 56.61 | 60.06 | 53.57 | 52.08 |
58
+ | Decider · 2B | 55.30 | 57.26 | 58.93 | 50.00 |
59
+ | Qwen3.5 · 2B · untuned | 48.06 | 50.36 | 45.09 | 47.22 |
60
+ | Laya · EN/ML | 47.38 | 53.02 | 52.23 | 19.44 |
61
 
62
+ Accuracy (%). Overall averages **20 task families / 1,760 questions** equally; type columns pool their questions. Both Decision peers are v1.1. Laya's native interface reported 152 truncated inputs. [Methods, coverage and uncertainty](EVALUATION.md).
63
 
64
  ![Decision quality ranking](assets/decision-quality.png)
65
 
66
+ ![Capability comparison across twenty families](assets/decision-capabilities.png)
67
 
68
+ Sol improves overall, while evidence judgments, rubric scoring and state reasoning remain uneven. See the evaluation for observed-panel limitations.
69
+
70
+ ## Reading supplied evidence
71
+
72
+ | Model | BoolQ ↑ | Natural Choice ↑ | Weighted ↑ |
73
+ | --- | ---: | ---: | ---: |
74
+ | Jev · 1.13.0 | **92.50** | **96.56** | **94.53** |
75
+ | Decider · 2B | 91.88 | 92.19 | 92.03 |
76
+ | Nox · 4B · v1.1 | 87.50 | 69.38 | 78.44 |
77
+ | Sol · 2B · v1.1 | 79.38 | 67.50 | 73.44 |
78
+ | Qwen3.5 · 4B · untuned | 83.75 | 92.19 | 87.97 |
79
+ | Qwen3.5 · 2B · untuned | 65.00 | 82.50 | 73.75 |
80
+ | Laya · EN/ML | 69.38 | 33.12 | 51.25 |
81
+
82
+ Accuracy (%), **480 human-annotated questions**. Natural Choice averages English and Chinese Belebele; weighted = 50% BoolQ + 50% Natural Choice. All local inputs were retained; Jev's internal retention is unknown. Public pretraining exposure is unknown. [Counts, uncertainty and limitations](READING-SUPPLEMENT.md).
83
 
84
  ## More questions, measured
85
 
86
  ![Latency as questions per request increase](assets/decision-question-scaling.png)
87
 
88
+ **Measured with v1.0 weights and runtime.** Same state, same question length, four choices; only the number of questions changes. Each model keeps its native interface (Sol: 309 tokens/question). Median of 30 requests per point on one AMD gfx942 GPU; local Python request time includes tokenization and inference, excluding loading and network. [p95, drift checks and full methods](QUESTION-SCALING.md).
89
 
90
  ## Try it
91
 
92
+ Download the fixed release with `hf download llm-semantic-router/Decision-1.0-Sol --revision v1.1 --local-dir decision-model`, then follow the [ROCm setup](RUNTIME.md). Inside that container, with the model mounted at `/model`:
93
 
94
  ```python
95
  from decision import DecisionModel
 
101
 
102
  [Actual request and measured output](model-card-example.json) · [Install and API guide](USAGE.md)
103
 
104
+ The complete state, question and candidates must fit 16,384 tokens. Overflow is rejected. Score returns an expected ordinal index. Runtime: AMD gfx942 validated; CPU/MPS unsupported; NVIDIA unqualified. This release automatically binds its validated normalization profile; use a fresh Python process. [Runtime profile](NORMALIZATION_RUNTIME.md).
105
 
106
  ## Architecture
107
 
RUNTIME.md CHANGED
@@ -2,7 +2,9 @@
2
 
3
  The package has a public, digest-pinned installation path. `Dockerfile.runtime` starts from `vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339`, adds two hash-checked FLA wheels, and installs this repository's loading wrapper. It keeps the base image's ROCm PyTorch and Triton builds. The vLLM server is not used by Decision inference.
4
 
5
- **Validation boundary:** the public registry manifest, base-image ancestry, package metadata and critical PyTorch binary hashes have been checked. The recipe built successfully and passed CPU imports and real AMD ROCm gfx942 GPU GPU inference for both released bundles. On the packaged three-question Choice/Noul/Score example, its complete responses matched the qualified research runtime exactly; the wrapper matched the direct engine and rejected an oversized complete input. Evidence is in `runtime-build-provenance.json`. This example establishes a working public installation path; it is not a full rerun of the quality or timing benchmark. Published benchmark results use the qualified runtime in each bundle's `runtime.json`.
 
 
6
 
7
  ## Build and run
8
 
@@ -50,3 +52,5 @@ The official [FLA installation guide](https://github.com/fla-org/flash-linear-at
50
  The public [PyTorch ROCm 7.2 wheel index](https://download.pytorch.org/whl/rocm7.2/torch/) contains ordinary release wheels such as `2.12.0+rocm7.2`. They are a different artifact from the qualified `2.12.0+git6bbd260` build and are not interchangeable evidence. The latter's [source commit is public](https://github.com/pytorch/pytorch/commit/6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5), but a source commit alone does not reproduce compiler flags, linked libraries and binary behavior.
51
 
52
  An alternative runtime must recheck the installed versions, actual FLA Gated DeltaNet dispatch, BF16 backbone plus FP32 head, fixed prompt rendering, batch size eight, and model output agreement. Runtime changes can shift probabilities near a decision boundary even with identical weights. The supplied engine uses FLA Gated DeltaNet, reference PyTorch causal convolution and SDPA; selecting a different kernel is a new runtime configuration.
 
 
 
2
 
3
  The package has a public, digest-pinned installation path. `Dockerfile.runtime` starts from `vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339`, adds two hash-checked FLA wheels, and installs this repository's loading wrapper. It keeps the base image's ROCm PyTorch and Triton builds. The vLLM server is not used by Decision inference.
4
 
5
+ **Validation boundary:** the public registry manifest, base-image ancestry, package metadata and critical PyTorch binary hashes have been checked. The recipe built successfully and passed CPU imports and real AMD ROCm gfx942 GPU inference for both initial v1.0 bundles. On the packaged three-question Choice/Noul/Score example, its complete responses matched the qualified research runtime exactly; the wrapper matched the direct engine and rejected an oversized complete input. The initial v1.0 build evidence is in `runtime-build-provenance.json`. Updated weights retain the same installation recipe and are separately checked during release; their actual request and output are in `model-card-example.json`. This example establishes a working public installation path; it is not a full rerun of the quality or timing benchmark. Published benchmark results use the qualified runtime in each bundle's `runtime.json`.
6
+
7
+ The Sol v1.1 default entrypoint additionally binds its bundled normalization profile before importing FLA. It has passed a fresh-cache, network-disabled reload on all 1,000 calibration questions, plus the three-type example and overflow rejection. Use a fresh Python process; see [normalization runtime](NORMALIZATION_RUNTIME.md). The historical image-build evidence below identifies its original v1.0 wrappers and bundles; it does not substitute for v1.1 validation.
8
 
9
  ## Build and run
10
 
 
52
  The public [PyTorch ROCm 7.2 wheel index](https://download.pytorch.org/whl/rocm7.2/torch/) contains ordinary release wheels such as `2.12.0+rocm7.2`. They are a different artifact from the qualified `2.12.0+git6bbd260` build and are not interchangeable evidence. The latter's [source commit is public](https://github.com/pytorch/pytorch/commit/6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5), but a source commit alone does not reproduce compiler flags, linked libraries and binary behavior.
53
 
54
  An alternative runtime must recheck the installed versions, actual FLA Gated DeltaNet dispatch, BF16 backbone plus FP32 head, fixed prompt rendering, batch size eight, and model output agreement. Runtime changes can shift probabilities near a decision boundary even with identical weights. The supplied engine uses FLA Gated DeltaNet, reference PyTorch causal convolution and SDPA; selecting a different kernel is a new runtime configuration.
55
+
56
+ For this profiled Sol release, `allow_unvalidated_runtime=True` does not bypass the normalization profile version, GPU architecture or supported-shape checks.
RUNTIME_BINDING.json ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source_bundle_manifest_sha256": "9242b7fd4ce226f638108810ec5156dca6f270cce5a16efcf5ef091180ad75be",
3
+ "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
4
+ "unchanged_files": [
5
+ {
6
+ "file": "backbone/config.json",
7
+ "bytes": 1789,
8
+ "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05"
9
+ },
10
+ {
11
+ "file": "backbone/model.safetensors",
12
+ "bytes": 3763685328,
13
+ "sha256": "168f7f90bf410130da92fd5270305286d53d5cbb68be02f1c611c40cb9243fe2"
14
+ },
15
+ {
16
+ "file": "code/decision_model.py",
17
+ "bytes": 10114,
18
+ "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
19
+ },
20
+ {
21
+ "file": "code/predict.py",
22
+ "bytes": 3164,
23
+ "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
24
+ },
25
+ {
26
+ "file": "decision_config.json",
27
+ "bytes": 753,
28
+ "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a"
29
+ },
30
+ {
31
+ "file": "decision_head.safetensors",
32
+ "bytes": 8424272,
33
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
34
+ },
35
+ {
36
+ "file": "temperature.json",
37
+ "bytes": 13733,
38
+ "sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44"
39
+ },
40
+ {
41
+ "file": "tokenizer.json",
42
+ "bytes": 19989325,
43
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
44
+ },
45
+ {
46
+ "file": "tokenizer_config.json",
47
+ "bytes": 1123,
48
+ "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
49
+ }
50
+ ],
51
+ "training_selection_calibration_unchanged": true,
52
+ "recomputed_or_recalibrated": false,
53
+ "private_compiled_cache_required": false
54
+ }
SOURCE_BUNDLE_MANIFEST.json ADDED
@@ -0,0 +1,129 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "research-pointer-bundle-v1",
3
+ "status": "candidate-export-awaiting-independent-reload-and-quality-gates",
4
+ "files": [
5
+ {
6
+ "file": "backbone/config.json",
7
+ "bytes": 1789,
8
+ "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05"
9
+ },
10
+ {
11
+ "file": "backbone/model.safetensors",
12
+ "bytes": 3763685328,
13
+ "sha256": "168f7f90bf410130da92fd5270305286d53d5cbb68be02f1c611c40cb9243fe2"
14
+ },
15
+ {
16
+ "file": "chat_template.jinja",
17
+ "bytes": 7755,
18
+ "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80"
19
+ },
20
+ {
21
+ "file": "code/decision_api.py",
22
+ "bytes": 6724,
23
+ "sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21"
24
+ },
25
+ {
26
+ "file": "code/decision_model.py",
27
+ "bytes": 10114,
28
+ "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
29
+ },
30
+ {
31
+ "file": "code/predict.py",
32
+ "bytes": 3164,
33
+ "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
34
+ },
35
+ {
36
+ "file": "decision_config.json",
37
+ "bytes": 753,
38
+ "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a"
39
+ },
40
+ {
41
+ "file": "decision_head.safetensors",
42
+ "bytes": 8424272,
43
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
44
+ },
45
+ {
46
+ "file": "runtime.json",
47
+ "bytes": 378,
48
+ "sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9"
49
+ },
50
+ {
51
+ "file": "temperature.json",
52
+ "bytes": 13733,
53
+ "sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44"
54
+ },
55
+ {
56
+ "file": "tokenizer.json",
57
+ "bytes": 19989325,
58
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
59
+ },
60
+ {
61
+ "file": "tokenizer_config.json",
62
+ "bytes": 1123,
63
+ "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
64
+ }
65
+ ],
66
+ "tensors": [
67
+ {
68
+ "file": "backbone/model.safetensors",
69
+ "elements": 1881825088,
70
+ "elements_by_dtype": {
71
+ "BF16": 1881825088
72
+ }
73
+ },
74
+ {
75
+ "file": "decision_head.safetensors",
76
+ "elements": 2105856,
77
+ "elements_by_dtype": {
78
+ "F32": 2105856
79
+ }
80
+ }
81
+ ],
82
+ "source_checkpoint_files": [
83
+ {
84
+ "file": "backbone/config.json",
85
+ "bytes": 1788,
86
+ "sha256": "071e97d8291168ba712237744c9a60e733acce554e6164322d22550dc96de19b"
87
+ },
88
+ {
89
+ "file": "backbone/model-00001-of-00002.safetensors",
90
+ "bytes": 3999855496,
91
+ "sha256": "93059b3982827486760a708d4f10aa6ae61de94ce90607b080021529bd1562ab"
92
+ },
93
+ {
94
+ "file": "backbone/model-00002-of-00002.safetensors",
95
+ "bytes": 3527479552,
96
+ "sha256": "2ef40a6f2cd250aa0c08a6ced808d1aa025bff3dd6720857651c8f8943cf5cc9"
97
+ },
98
+ {
99
+ "file": "decision_config.json",
100
+ "bytes": 1119,
101
+ "sha256": "1e6d926e1593ba243e58448c3d3cca39c6731faa55e5acec369832e4efab074d"
102
+ },
103
+ {
104
+ "file": "decision_head.safetensors",
105
+ "bytes": 8424272,
106
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
107
+ },
108
+ {
109
+ "file": "tokenizer.json",
110
+ "bytes": 19989325,
111
+ "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
112
+ },
113
+ {
114
+ "file": "tokenizer_config.json",
115
+ "bytes": 1123,
116
+ "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
117
+ }
118
+ ],
119
+ "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
120
+ "source_api_code_sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21",
121
+ "dev_sha256": "5ba3bd7527da831d77cec8c6a2f260a71e10a57742e356510c719f8f30d61a70",
122
+ "production_predictions_sha256": "b9e2cf2c466e6f9422e627905625400d683c6ce0cecb015c16e5784b0c3c577c",
123
+ "temperature_sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44",
124
+ "runtime_sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9",
125
+ "production_batch_size": 8,
126
+ "input_length_limit": 16384,
127
+ "original_checkpoint_name": "checkpoint-001563",
128
+ "no_publication_performed": true
129
+ }
TIMING.md DELETED
@@ -1,164 +0,0 @@
1
- # Local request efficiency
2
-
3
- One AMD gfx942 accelerator, 261824 MiB visible device memory; the same physical device for every model.
4
-
5
- Synchronized local API request latency including input rendering, tokenization, transfers, forward passes, and answer assembly. Excludes model loading, eligibility diagnostics, warmup, network, and post-response shape validation.
6
-
7
- Three outer blocks with independently randomized model order. Each model is reloaded per panel/block, with three warmups per case followed by ten timed repetitions. Case order is shuffled within each block. 30 timings per eligible cell.
8
-
9
- No other training or inference GPU process during measurement; each job has two idle-process preflight checks. ROCm telemetry is sampled during execution. CPU affinity, host services, and clocks are not locked; this is a GPU-quiet reproducible request benchmark, not a noise-free hardware microbenchmark.
10
-
11
- One API request at a time; Q1/Q8/Q32 are questions within one request, not concurrent client requests. Native adapters retain their question packing/batch strategy. No cross-request cache.
12
-
13
- Qwen/Decider backbone BF16, Sol/Nox decision head FP32; Laya preserves upstream FP32 parameters with native BF16 autocast. GDN-capable Qwen paths use FLA 0.5.2; Laya architecture has no GDN.
14
-
15
- The timing panel contains English states and questions. Routed Laya selects its English branch here; these speeds do not characterize its multilingual branch.
16
-
17
- p50/p95 are empirical percentiles of 30 observations, not confidence intervals. Questions/s equals 1000 times total questions divided by total measured milliseconds; it is not maximum concurrent-serving capacity.
18
-
19
- Peak PyTorch allocated memory, including resident model parameters, in GiB. Excludes allocator-reserved memory, non-PyTorch allocations, and device-driver overhead.
20
-
21
- All models receive identical byte-level states and question specifications. Token counts differ by native tokenizer/template. Truncated/unsupported cells are explicitly excluded; no short-input substitution. These repeated timing prompts do not measure semantic accuracy.
22
-
23
- Every timing model_runtime_key is identical to that model’s reported final-quality key. No reference-quality/optimized-speed mixture.
24
-
25
- After 25 successful jobs, preflight detected a transient ROCm process before starting job 26. The process exited; an explicit resume executed only the remaining 11 jobs after idle checks. No timing job was repeated.
26
-
27
- No comparable local GPU timing is available for hosted Jev. Existing API wall times are a different workload/network measure and are not in the local latency ranking.
28
-
29
- ## choice
30
-
31
- | Model | Case | Q / K | Tokens per row | p50 / p95 ms | Questions/s | Peak GiB |
32
- |---|---|---:|---:|---:|---:|---:|
33
- | Sol 2B | short-q1 | 1 / 4 | 309–309 | 21.04 / 21.41 | 47.6 | 3.61 |
34
- | Sol 2B | medium-q1 | 1 / 4 | 2243–2243 | 35.88 / 36.10 | 27.9 | 3.76 |
35
- | Sol 2B | long-q1 | 1 / 4 | 8863–8863 | 134.31 / 135.52 | 7.4 | 4.33 |
36
- | Sol 2B | short-q8 | 8 / 4 | 309–309 | 37.29 / 37.51 | 214.3 | 3.78 |
37
- | Sol 2B | medium-q8 | 8 / 4 | 2243–2243 | 216.20 / 216.66 | 37.0 | 5.00 |
38
- | Sol 2B | short-q32 | 32 / 4 | 309–309 | 147.50 / 147.91 | 216.8 | 3.78 |
39
- | Sol 2B | short-k255 | 1 / 255 | 7870–7870 | 118.33 / 119.12 | 8.4 | 4.24 |
40
- | Nox 4B | short-q1 | 1 / 4 | 309–309 | 27.37 / 28.95 | 36.0 | 8.06 |
41
- | Nox 4B | medium-q1 | 1 / 4 | 2243–2243 | 68.72 / 78.11 | 14.3 | 8.32 |
42
- | Nox 4B | long-q1 | 1 / 4 | 8863–8863 | 278.96 / 296.37 | 3.6 | 9.25 |
43
- | Nox 4B | short-q8 | 8 / 4 | 309–309 | 69.25 / 69.84 | 115.4 | 8.35 |
44
- | Nox 4B | medium-q8 | 8 / 4 | 2243–2243 | 442.80 / 445.90 | 18.0 | 10.43 |
45
- | Nox 4B | short-q32 | 32 / 4 | 309–309 | 275.65 / 278.26 | 115.8 | 8.35 |
46
- | Nox 4B | short-k255 | 1 / 255 | 7870–7870 | 248.31 / 249.58 | 4.0 | 9.10 |
47
- | Laya routed | short-q1 | 1 / 4 | 228–228 | 11.54 / 12.21 | 85.8 | 3.56 |
48
- | Laya routed | medium-q1 | 1 / 4 | — | input_truncated | — | — |
49
- | Laya routed | long-q1 | 1 / 4 | — | input_truncated | — | — |
50
- | Laya routed | short-q8 | 8 / 4 | 228–228 | 14.56 / 15.02 | 548.5 | 3.61 |
51
- | Laya routed | medium-q8 | 8 / 4 | — | input_truncated | — | — |
52
- | Laya routed | short-q32 | 32 / 4 | 228–228 | 37.89 / 38.30 | 843.8 | 3.78 |
53
- | Laya routed | short-k255 | 1 / 255 | — | unsupported | — | — |
54
- | Decider 2B | short-q1 | 1 / 4 | 250–250 | 23.35 / 23.91 | 42.9 | 3.62 |
55
- | Decider 2B | medium-q1 | 1 / 4 | 2184–2184 | 35.24 / 35.41 | 28.4 | 3.80 |
56
- | Decider 2B | long-q1 | 1 / 4 | 8804–8804 | 131.47 / 131.99 | 7.6 | 4.42 |
57
- | Decider 2B | short-q8 | 8 / 4 | 250–250 | 31.59 / 31.77 | 253.4 | 3.90 |
58
- | Decider 2B | medium-q8 | 8 / 4 | 2184–2184 | 209.86 / 210.64 | 38.1 | 5.28 |
59
- | Decider 2B | short-q32 | 32 / 4 | 250–250 | 96.67 / 97.02 | 330.9 | 4.87 |
60
- | Decider 2B | short-k255 | 1 / 255 | 5556–5556 | 78.83 / 79.27 | 12.7 | 4.10 |
61
- | Qwen3.5 2B LM-head | short-q1 | 1 / 4 | 296–296 | 20.51 / 20.71 | 49.2 | 3.60 |
62
- | Qwen3.5 2B LM-head | medium-q1 | 1 / 4 | 2230–2230 | 35.22 / 35.53 | 28.4 | 3.74 |
63
- | Qwen3.5 2B LM-head | long-q1 | 1 / 4 | 8850–8850 | 129.02 / 131.25 | 7.7 | 4.21 |
64
- | Qwen3.5 2B LM-head | short-q8 | 8 / 4 | 296–296 | 37.32 / 37.61 | 214.5 | 3.75 |
65
- | Qwen3.5 2B LM-head | medium-q8 | 8 / 4 | 2230–2230 | 202.06 / 202.75 | 39.6 | 4.86 |
66
- | Qwen3.5 2B LM-head | short-q32 | 32 / 4 | 296–296 | 146.88 / 147.16 | 218.3 | 3.75 |
67
- | Qwen3.5 2B LM-head | short-k255 | 1 / 255 | — | unsupported | — | — |
68
- | Qwen3.5 4B LM-head | short-q1 | 1 / 4 | 296–296 | 27.47 / 28.76 | 36.3 | 8.04 |
69
- | Qwen3.5 4B LM-head | medium-q1 | 1 / 4 | 2230–2230 | 66.97 / 67.66 | 14.9 | 8.29 |
70
- | Qwen3.5 4B LM-head | long-q1 | 1 / 4 | 8850–8850 | 257.47 / 258.85 | 3.9 | 9.12 |
71
- | Qwen3.5 4B LM-head | short-q8 | 8 / 4 | 296–296 | 69.03 / 69.46 | 115.8 | 8.31 |
72
- | Qwen3.5 4B LM-head | medium-q8 | 8 / 4 | 2230–2230 | 414.85 / 415.72 | 19.3 | 10.25 |
73
- | Qwen3.5 4B LM-head | short-q32 | 32 / 4 | 296–296 | 274.69 / 275.19 | 116.5 | 8.31 |
74
- | Qwen3.5 4B LM-head | short-k255 | 1 / 255 | — | unsupported | — | — |
75
-
76
- ## noul-score
77
-
78
- | Model | Case | Q / K | Tokens per row | p50 / p95 ms | Questions/s | Peak GiB |
79
- |---|---|---:|---:|---:|---:|---:|
80
- | Sol 2B | noul-short-q1 | 1 / 2 | 262–262 | 20.84 / 21.38 | 47.8 | 3.61 |
81
- | Sol 2B | noul-medium-q1 | 1 / 2 | 2196–2196 | 34.86 / 35.12 | 28.7 | 3.76 |
82
- | Sol 2B | noul-long-q1 | 1 / 2 | 8816–8816 | 131.69 / 132.53 | 7.6 | 4.33 |
83
- | Sol 2B | noul-short-q8 | 8 / 2 | 262–262 | 33.20 / 33.91 | 239.9 | 3.76 |
84
- | Sol 2B | noul-medium-q8 | 8 / 2 | 2196–2196 | 207.14 / 208.02 | 38.6 | 4.96 |
85
- | Sol 2B | noul-short-q32 | 32 / 2 | 262–262 | 131.43 / 132.38 | 243.5 | 3.77 |
86
- | Sol 2B | score-short-q1 | 1 / 4 | 307–307 | 20.91 / 21.50 | 47.6 | 3.61 |
87
- | Sol 2B | score-medium-q1 | 1 / 4 | 2241–2241 | 35.80 / 36.03 | 27.9 | 3.76 |
88
- | Sol 2B | score-long-q1 | 1 / 4 | 8861–8861 | 134.45 / 135.55 | 7.4 | 4.33 |
89
- | Sol 2B | score-short-q8 | 8 / 4 | 307–307 | 37.37 / 37.68 | 214.1 | 3.78 |
90
- | Sol 2B | score-medium-q8 | 8 / 4 | 2241–2241 | 215.72 / 216.75 | 37.1 | 5.00 |
91
- | Sol 2B | score-short-q32 | 32 / 4 | 307–307 | 146.54 / 147.00 | 218.4 | 3.78 |
92
- | Sol 2B | score-short-q1-levels2 | 1 / 2 | 274–274 | 20.99 / 22.65 | 47.2 | 3.61 |
93
- | Sol 2B | score-short-q1-levels10 | 1 / 10 | 490–490 | 21.10 / 21.56 | 47.3 | 3.63 |
94
- | Nox 4B | noul-short-q1 | 1 / 2 | 262–262 | 27.44 / 29.16 | 35.8 | 8.05 |
95
- | Nox 4B | noul-medium-q1 | 1 / 2 | 2196–2196 | 67.56 / 67.75 | 14.8 | 8.31 |
96
- | Nox 4B | noul-long-q1 | 1 / 2 | 8816–8816 | 273.90 / 275.65 | 3.6 | 9.25 |
97
- | Nox 4B | noul-short-q8 | 8 / 2 | 262–262 | 64.61 / 64.87 | 124.2 | 8.32 |
98
- | Nox 4B | noul-medium-q8 | 8 / 2 | 2196–2196 | 427.53 / 429.06 | 18.7 | 10.36 |
99
- | Nox 4B | noul-short-q32 | 32 / 2 | 262–262 | 256.45 / 257.09 | 124.8 | 8.32 |
100
- | Nox 4B | score-short-q1 | 1 / 4 | 307–307 | 27.41 / 29.51 | 35.7 | 8.06 |
101
- | Nox 4B | score-medium-q1 | 1 / 4 | 2241–2241 | 68.76 / 69.12 | 14.5 | 8.32 |
102
- | Nox 4B | score-long-q1 | 1 / 4 | 8861–8861 | 279.43 / 280.17 | 3.6 | 9.25 |
103
- | Nox 4B | score-short-q8 | 8 / 4 | 307–307 | 69.56 / 69.74 | 115.4 | 8.35 |
104
- | Nox 4B | score-medium-q8 | 8 / 4 | 2241–2241 | 442.15 / 443.60 | 18.1 | 10.43 |
105
- | Nox 4B | score-short-q32 | 32 / 4 | 307–307 | 277.89 / 278.51 | 115.1 | 8.35 |
106
- | Nox 4B | score-short-q1-levels2 | 1 / 2 | 274–274 | 27.76 / 28.89 | 35.8 | 8.05 |
107
- | Nox 4B | score-short-q1-levels10 | 1 / 10 | 490–490 | 27.75 / 29.39 | 35.4 | 8.08 |
108
- | Laya routed | noul-short-q1 | 1 / 2 | 202–202 | 12.05 / 12.24 | 84.2 | 3.56 |
109
- | Laya routed | noul-medium-q1 | 1 / 2 | — | input_truncated | — | — |
110
- | Laya routed | noul-long-q1 | 1 / 2 | — | input_truncated | — | — |
111
- | Laya routed | noul-short-q8 | 8 / 2 | 202–202 | 13.83 / 14.32 | 577.7 | 3.60 |
112
- | Laya routed | noul-medium-q8 | 8 / 2 | — | input_truncated | — | — |
113
- | Laya routed | noul-short-q32 | 32 / 2 | 202–202 | 34.38 / 34.68 | 929.9 | 3.75 |
114
- | Laya routed | score-short-q1 | 1 / 4 | 231–231 | 12.17 / 12.59 | 83.3 | 3.56 |
115
- | Laya routed | score-medium-q1 | 1 / 4 | — | input_truncated | — | — |
116
- | Laya routed | score-long-q1 | 1 / 4 | — | input_truncated | — | — |
117
- | Laya routed | score-short-q8 | 8 / 4 | 231–231 | 16.92 / 17.22 | 471.6 | 3.61 |
118
- | Laya routed | score-medium-q8 | 8 / 4 | — | input_truncated | — | — |
119
- | Laya routed | score-short-q32 | 32 / 4 | 231–231 | 38.64 / 38.95 | 827.9 | 3.79 |
120
- | Laya routed | score-short-q1-levels2 | 1 / 2 | 216–216 | 11.73 / 11.98 | 86.4 | 3.56 |
121
- | Laya routed | score-short-q1-levels10 | 1 / 10 | 328–328 | 11.91 / 12.54 | 84.4 | 3.56 |
122
- | Decider 2B | noul-short-q1 | 1 / 2 | 204–204 | 23.29 / 23.83 | 42.9 | 3.62 |
123
- | Decider 2B | noul-medium-q1 | 1 / 2 | 2138–2138 | 34.91 / 35.29 | 28.6 | 3.79 |
124
- | Decider 2B | noul-long-q1 | 1 / 2 | 8758–8758 | 132.39 / 133.57 | 7.5 | 4.41 |
125
- | Decider 2B | noul-short-q8 | 8 / 2 | 204–204 | 31.21 / 31.46 | 256.5 | 3.90 |
126
- | Decider 2B | noul-medium-q8 | 8 / 2 | 2138–2138 | 204.89 / 205.29 | 39.1 | 5.24 |
127
- | Decider 2B | noul-short-q32 | 32 / 2 | 204–204 | 95.38 / 95.82 | 335.8 | 4.87 |
128
- | Decider 2B | score-short-q1 | 1 / 4 | 222–225 | 24.42 / 24.66 | 41.2 | 3.74 |
129
- | Decider 2B | score-medium-q1 | 1 / 4 | 2156–2159 | 107.51 / 108.72 | 9.3 | 4.41 |
130
- | Decider 2B | score-long-q1 | 1 / 4 | 8776–8779 | 469.64 / 472.21 | 2.1 | 6.93 |
131
- | Decider 2B | score-short-q8 | 8 / 4 | 222–225 | 96.03 / 96.75 | 83.2 | 4.87 |
132
- | Decider 2B | score-medium-q8 | 8 / 4 | 2156–2159 | 773.01 / 774.95 | 10.4 | 9.79 |
133
- | Decider 2B | score-short-q32 | 32 / 4 | 222–225 | 353.70 / 355.63 | 90.4 | 8.71 |
134
- | Decider 2B | score-short-q1-levels2 | 1 / 2 | 234–234 | 23.60 / 24.24 | 42.3 | 3.66 |
135
- | Decider 2B | score-short-q1-levels10 | 1 / 10 | 234–234 | 37.79 / 38.16 | 26.4 | 3.98 |
136
- | Qwen3.5 2B LM-head | noul-short-q1 | 1 / 2 | 263–263 | 19.87 / 20.91 | 49.8 | 3.60 |
137
- | Qwen3.5 2B LM-head | noul-medium-q1 | 1 / 2 | 2197–2197 | 34.91 / 35.11 | 28.6 | 3.74 |
138
- | Qwen3.5 2B LM-head | noul-long-q1 | 1 / 2 | 8817–8817 | 127.50 / 128.67 | 7.8 | 4.21 |
139
- | Qwen3.5 2B LM-head | noul-short-q8 | 8 / 2 | 263–263 | 34.26 / 34.58 | 233.3 | 3.73 |
140
- | Qwen3.5 2B LM-head | noul-medium-q8 | 8 / 2 | 2197–2197 | 199.54 / 200.27 | 40.1 | 4.83 |
141
- | Qwen3.5 2B LM-head | noul-short-q32 | 32 / 2 | 263–263 | 134.40 / 136.62 | 237.4 | 3.74 |
142
- | Qwen3.5 2B LM-head | score-short-q1 | 1 / 4 | 303–303 | 19.90 / 20.24 | 50.2 | 3.60 |
143
- | Qwen3.5 2B LM-head | score-medium-q1 | 1 / 4 | 2237–2237 | 35.52 / 35.90 | 28.1 | 3.74 |
144
- | Qwen3.5 2B LM-head | score-long-q1 | 1 / 4 | 8857–8857 | 128.28 / 129.84 | 7.8 | 4.21 |
145
- | Qwen3.5 2B LM-head | score-short-q8 | 8 / 4 | 303–303 | 37.81 / 37.98 | 211.6 | 3.75 |
146
- | Qwen3.5 2B LM-head | score-medium-q8 | 8 / 4 | 2237–2237 | 202.06 / 202.79 | 39.6 | 4.86 |
147
- | Qwen3.5 2B LM-head | score-short-q32 | 32 / 4 | 303–303 | 148.68 / 149.07 | 215.8 | 3.76 |
148
- | Qwen3.5 2B LM-head | score-short-q1-levels2 | 1 / 2 | 286–286 | 19.93 / 20.44 | 50.0 | 3.60 |
149
- | Qwen3.5 2B LM-head | score-short-q1-levels10 | 1 / 10 | 438–438 | 20.13 / 20.45 | 49.6 | 3.61 |
150
- | Qwen3.5 4B LM-head | noul-short-q1 | 1 / 2 | 263–263 | 26.61 / 26.89 | 37.6 | 8.04 |
151
- | Qwen3.5 4B LM-head | noul-medium-q1 | 1 / 2 | 2197–2197 | 65.78 / 65.96 | 15.2 | 8.28 |
152
- | Qwen3.5 4B LM-head | noul-long-q1 | 1 / 2 | 8817–8817 | 256.56 / 258.56 | 3.9 | 9.11 |
153
- | Qwen3.5 4B LM-head | noul-short-q8 | 8 / 2 | 263–263 | 64.86 / 64.99 | 123.3 | 8.27 |
154
- | Qwen3.5 4B LM-head | noul-medium-q8 | 8 / 2 | 2197–2197 | 410.51 / 411.79 | 19.5 | 10.22 |
155
- | Qwen3.5 4B LM-head | noul-short-q32 | 32 / 2 | 263–263 | 257.86 / 258.60 | 124.4 | 8.28 |
156
- | Qwen3.5 4B LM-head | score-short-q1 | 1 / 4 | 303–303 | 26.60 / 26.74 | 37.6 | 8.04 |
157
- | Qwen3.5 4B LM-head | score-medium-q1 | 1 / 4 | 2237–2237 | 66.41 / 80.27 | 14.7 | 8.29 |
158
- | Qwen3.5 4B LM-head | score-long-q1 | 1 / 4 | 8857–8857 | 257.84 / 258.14 | 3.9 | 9.12 |
159
- | Qwen3.5 4B LM-head | score-short-q8 | 8 / 4 | 303–303 | 69.96 / 70.42 | 114.3 | 8.31 |
160
- | Qwen3.5 4B LM-head | score-medium-q8 | 8 / 4 | 2237–2237 | 416.17 / 446.59 | 19.0 | 10.26 |
161
- | Qwen3.5 4B LM-head | score-short-q32 | 32 / 4 | 303–303 | 278.88 / 280.80 | 114.7 | 8.31 |
162
- | Qwen3.5 4B LM-head | score-short-q1-levels2 | 1 / 2 | 286–286 | 26.65 / 30.12 | 36.4 | 8.04 |
163
- | Qwen3.5 4B LM-head | score-short-q1-levels10 | 1 / 10 | 438–438 | 26.82 / 27.25 | 37.2 | 8.06 |
164
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
USAGE.md CHANGED
@@ -64,3 +64,7 @@ The bound is **16,384 tokens per complete question**, including its state, instr
64
  The qualified device is an AMD ROCm GPU accessed as `cuda:0` in PyTorch. Backbone parameters remain BF16, the candidate head FP32, with FLA Gated DeltaNet and SDPA. CPU and MPS inference are not implemented by the current engine and are rejected explicitly. Other GPU/runtime combinations require validation. The loader checks manifest hashes and the recorded runtime before loading. `allow_unvalidated_runtime=True` downgrades runtime differences to an explicit warning for experiments; it does not silently substitute a claimed validated environment or bypass unsupported CPU execution.
65
 
66
  This package is a loading wrapper, not a trainer, HTTP service or new model selection policy. No prediction logs or private local paths are required by the bundle.
 
 
 
 
 
64
  The qualified device is an AMD ROCm GPU accessed as `cuda:0` in PyTorch. Backbone parameters remain BF16, the candidate head FP32, with FLA Gated DeltaNet and SDPA. CPU and MPS inference are not implemented by the current engine and are rejected explicitly. Other GPU/runtime combinations require validation. The loader checks manifest hashes and the recorded runtime before loading. `allow_unvalidated_runtime=True` downgrades runtime differences to an explicit warning for experiments; it does not silently substitute a claimed validated environment or bypass unsupported CPU execution.
65
 
66
  This package is a loading wrapper, not a trainer, HTTP service or new model selection policy. No prediction logs or private local paths are required by the bundle.
67
+
68
+ For the Sol v1.1 bundle, the validated normalization profile loads automatically. Use one model profile per fresh Python process. See [normalization runtime](NORMALIZATION_RUNTIME.md) for the supported shape and hardware contract.
69
+
70
+ For this profiled Sol release, `allow_unvalidated_runtime=True` does not bypass the normalization profile version, GPU architecture or supported-shape checks.
assets/decision-capabilities.pdf CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1314aae927ea0aa598e62a80138e2ea9bfd8e851bc6faa91504d5aa691426e24
3
- size 45170
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:41f1b6e93b41c7f2057ad9c0743ddf0fc643e0c48ba0de58d8a3b58dbc897f1e
3
+ size 51635
assets/decision-capabilities.png CHANGED

Git LFS Details

  • SHA256: ed88c09a355c8583ac4d7b31bd4ff93b7be8725fde436f3d76e1754ae92a5962
  • Pointer size: 131 Bytes
  • Size of remote file: 276 kB

Git LFS Details

  • SHA256: 81a551e4b353bf2934d722f6f6038972bb8b4cd68ffd22ed546506a8303daebe
  • Pointer size: 131 Bytes
  • Size of remote file: 446 kB
assets/decision-capabilities.svg CHANGED
assets/decision-quality.pdf CHANGED
Binary files a/assets/decision-quality.pdf and b/assets/decision-quality.pdf differ
 
assets/decision-quality.png CHANGED

Git LFS Details

  • SHA256: dbf0e32d3e7846f94b8d6fd224eaf54677cd1ecb140f1671fb1b80f2a6d25935
  • Pointer size: 131 Bytes
  • Size of remote file: 176 kB

Git LFS Details

  • SHA256: 2176d6ffb40424f9121989ec5a63e51789f39e79d086ba0a27735aa37938df01
  • Pointer size: 131 Bytes
  • Size of remote file: 171 kB
assets/decision-quality.svg CHANGED
backbone/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cab4b9b9dfd490781acf489b78b6d3faa1d56a1be8fab5d8baf6c464e4c0ea1d
3
  size 3763685328
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:168f7f90bf410130da92fd5270305286d53d5cbb68be02f1c611c40cb9243fe2
3
  size 3763685328
bundle-manifest.json CHANGED
@@ -1,7 +1,22 @@
1
  {
2
  "format": "research-pointer-bundle-v1",
3
- "status": "candidate-export-awaiting-independent-reload-and-quality-gates",
4
  "files": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  {
6
  "file": "backbone/config.json",
7
  "bytes": 1789,
@@ -10,7 +25,7 @@
10
  {
11
  "file": "backbone/model.safetensors",
12
  "bytes": 3763685328,
13
- "sha256": "cab4b9b9dfd490781acf489b78b6d3faa1d56a1be8fab5d8baf6c464e4c0ea1d"
14
  },
15
  {
16
  "file": "chat_template.jinja",
@@ -19,8 +34,8 @@
19
  },
20
  {
21
  "file": "code/decision_api.py",
22
- "bytes": 6724,
23
- "sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21"
24
  },
25
  {
26
  "file": "code/decision_model.py",
@@ -32,6 +47,16 @@
32
  "bytes": 3164,
33
  "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
34
  },
 
 
 
 
 
 
 
 
 
 
35
  {
36
  "file": "decision_config.json",
37
  "bytes": 753,
@@ -40,17 +65,77 @@
40
  {
41
  "file": "decision_head.safetensors",
42
  "bytes": 8424272,
43
- "sha256": "308bc01085a2bcee89a6668cb06ca4b771e01338040689527127bdefd3cc7415"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
44
  },
45
  {
46
  "file": "runtime.json",
47
- "bytes": 952,
48
- "sha256": "7ed527476826756d6076f23d0eaa8b4b53ff34b9f8bd57ae8e65b50e4d28cfa6"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
49
  },
50
  {
51
  "file": "temperature.json",
52
- "bytes": 667,
53
- "sha256": "159d308a6b970c425b8938b6c50c4fff6a10bb40ca82b119e43386a855d609ab"
54
  },
55
  {
56
  "file": "tokenizer.json",
@@ -88,22 +173,22 @@
88
  {
89
  "file": "backbone/model-00001-of-00002.safetensors",
90
  "bytes": 3999855496,
91
- "sha256": "0cdb50da983e4d803d0a853bb6b1ad03634a118f734cfe5ecc8439090bd0e8dd"
92
  },
93
  {
94
  "file": "backbone/model-00002-of-00002.safetensors",
95
  "bytes": 3527479552,
96
- "sha256": "9147e23c767af4994fa55ef1208fe27764eb3043e31c54585f509206bf053f7d"
97
  },
98
  {
99
  "file": "decision_config.json",
100
- "bytes": 1109,
101
- "sha256": "b68c7406a7dd5da3f6558d6404aef3aee81f1168a5fc56a009fe918056628202"
102
  },
103
  {
104
  "file": "decision_head.safetensors",
105
  "bytes": 8424272,
106
- "sha256": "308bc01085a2bcee89a6668cb06ca4b771e01338040689527127bdefd3cc7415"
107
  },
108
  {
109
  "file": "tokenizer.json",
@@ -117,13 +202,16 @@
117
  }
118
  ],
119
  "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
120
- "source_api_code_sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21",
121
- "dev_sha256": "a38f9be168553d5a82a91991aca86aaa602e1951957bac7c475abcec8dcc232d",
122
- "production_predictions_sha256": "38df5ace521ee8ec7decc752ab7a49e5e7b2cda751c7d7d6a4fdf00ad8af4202",
123
- "temperature_sha256": "159d308a6b970c425b8938b6c50c4fff6a10bb40ca82b119e43386a855d609ab",
124
- "runtime_sha256": "7ed527476826756d6076f23d0eaa8b4b53ff34b9f8bd57ae8e65b50e4d28cfa6",
125
  "production_batch_size": 8,
126
  "input_length_limit": 16384,
127
- "original_checkpoint_name": "checkpoint-000200",
128
- "no_publication_performed": true
 
 
 
129
  }
 
1
  {
2
  "format": "research-pointer-bundle-v1",
3
+ "status": "profile-bound-candidate-awaiting-default-public-entrypoint-offline-proof",
4
  "files": [
5
+ {
6
+ "file": "NORMALIZATION_RUNTIME.md",
7
+ "bytes": 1441,
8
+ "sha256": "9938bbe9ffdb890490c1db08903659edde6d7c5a5b9ce0487949e7a4dc2d38f6"
9
+ },
10
+ {
11
+ "file": "RUNTIME_BINDING.json",
12
+ "bytes": 1764,
13
+ "sha256": "c050b5a43d9ddf98e07924878a4e72cfb1c4746131a3045c31557c262791f922"
14
+ },
15
+ {
16
+ "file": "SOURCE_BUNDLE_MANIFEST.json",
17
+ "bytes": 4206,
18
+ "sha256": "9242b7fd4ce226f638108810ec5156dca6f270cce5a16efcf5ef091180ad75be"
19
+ },
20
  {
21
  "file": "backbone/config.json",
22
  "bytes": 1789,
 
25
  {
26
  "file": "backbone/model.safetensors",
27
  "bytes": 3763685328,
28
+ "sha256": "168f7f90bf410130da92fd5270305286d53d5cbb68be02f1c611c40cb9243fe2"
29
  },
30
  {
31
  "file": "chat_template.jinja",
 
34
  },
35
  {
36
  "file": "code/decision_api.py",
37
+ "bytes": 8251,
38
+ "sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40"
39
  },
40
  {
41
  "file": "code/decision_model.py",
 
47
  "bytes": 3164,
48
  "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
49
  },
50
+ {
51
+ "file": "code/profile_guard.py",
52
+ "bytes": 6131,
53
+ "sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452"
54
+ },
55
+ {
56
+ "file": "code/runtime_profile.py",
57
+ "bytes": 2857,
58
+ "sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda"
59
+ },
60
  {
61
  "file": "decision_config.json",
62
  "bytes": 753,
 
65
  {
66
  "file": "decision_head.safetensors",
67
  "bytes": 8424272,
68
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
69
+ },
70
+ {
71
+ "file": "pyproject.toml",
72
+ "bytes": 436,
73
+ "sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946"
74
+ },
75
+ {
76
+ "file": "runtime-profile/l2norm_fwd_kernel.json",
77
+ "bytes": 13210,
78
+ "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc"
79
+ },
80
+ {
81
+ "file": "runtime-profile/profile.json",
82
+ "bytes": 18101,
83
+ "sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f"
84
  },
85
  {
86
  "file": "runtime.json",
87
+ "bytes": 1133,
88
+ "sha256": "9fe0a2bd626a8b974c85e4925cdc79dca320c9116c6d0196c12a87b9aeae8d2a"
89
+ },
90
+ {
91
+ "file": "src/decision/__init__.py",
92
+ "bytes": 167,
93
+ "sha256": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313"
94
+ },
95
+ {
96
+ "file": "src/decision/example.py",
97
+ "bytes": 3581,
98
+ "sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6"
99
+ },
100
+ {
101
+ "file": "src/decision/model.py",
102
+ "bytes": 9415,
103
+ "sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0"
104
+ },
105
+ {
106
+ "file": "src/decision_local.egg-info/PKG-INFO",
107
+ "bytes": 225,
108
+ "sha256": "cc5266721b2e02c5c963b59856d979d619f7d9d687b60bda037c7bad2f22cd1d"
109
+ },
110
+ {
111
+ "file": "src/decision_local.egg-info/SOURCES.txt",
112
+ "bytes": 361,
113
+ "sha256": "741f317b6e9b27453e8f2fbeb76c0d1da326e89de7978fe87ce906ed1e2a68a9"
114
+ },
115
+ {
116
+ "file": "src/decision_local.egg-info/dependency_links.txt",
117
+ "bytes": 1,
118
+ "sha256": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b"
119
+ },
120
+ {
121
+ "file": "src/decision_local.egg-info/entry_points.txt",
122
+ "bytes": 59,
123
+ "sha256": "5400ff8d993d37849bc01b702e4674b8d9c4122af514101ace0bcd691eab39fb"
124
+ },
125
+ {
126
+ "file": "src/decision_local.egg-info/requires.txt",
127
+ "bytes": 31,
128
+ "sha256": "b8bf334329a333c3bba85388fbf41377b17305ee71b402c431c1af7445eadc08"
129
+ },
130
+ {
131
+ "file": "src/decision_local.egg-info/top_level.txt",
132
+ "bytes": 9,
133
+ "sha256": "834608781b00d5df58e3ae55a6b15c201aeebacc26ec99a1035fd75b83e16760"
134
  },
135
  {
136
  "file": "temperature.json",
137
+ "bytes": 13733,
138
+ "sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44"
139
  },
140
  {
141
  "file": "tokenizer.json",
 
173
  {
174
  "file": "backbone/model-00001-of-00002.safetensors",
175
  "bytes": 3999855496,
176
+ "sha256": "93059b3982827486760a708d4f10aa6ae61de94ce90607b080021529bd1562ab"
177
  },
178
  {
179
  "file": "backbone/model-00002-of-00002.safetensors",
180
  "bytes": 3527479552,
181
+ "sha256": "2ef40a6f2cd250aa0c08a6ced808d1aa025bff3dd6720857651c8f8943cf5cc9"
182
  },
183
  {
184
  "file": "decision_config.json",
185
+ "bytes": 1119,
186
+ "sha256": "1e6d926e1593ba243e58448c3d3cca39c6731faa55e5acec369832e4efab074d"
187
  },
188
  {
189
  "file": "decision_head.safetensors",
190
  "bytes": 8424272,
191
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
192
  },
193
  {
194
  "file": "tokenizer.json",
 
202
  }
203
  ],
204
  "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
205
+ "source_api_code_sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40",
206
+ "dev_sha256": "5ba3bd7527da831d77cec8c6a2f260a71e10a57742e356510c719f8f30d61a70",
207
+ "production_predictions_sha256": "b9e2cf2c466e6f9422e627905625400d683c6ce0cecb015c16e5784b0c3c577c",
208
+ "temperature_sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44",
209
+ "runtime_sha256": "9fe0a2bd626a8b974c85e4925cdc79dca320c9116c6d0196c12a87b9aeae8d2a",
210
  "production_batch_size": 8,
211
  "input_length_limit": 16384,
212
+ "original_checkpoint_name": "checkpoint-001563",
213
+ "no_publication_performed": true,
214
+ "source_bundle_manifest_sha256": "9242b7fd4ce226f638108810ec5156dca6f270cce5a16efcf5ef091180ad75be",
215
+ "normalization_profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
216
+ "public_wrapper_included": true
217
  }
code/decision_api.py CHANGED
@@ -10,6 +10,32 @@ import math
10
  from pathlib import Path
11
 
12
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  def question_row(state, name, question):
14
  kind=question.get('type')
15
  if kind not in {'choice','noul','score'}:raise ValueError('Unknown question type')
@@ -61,6 +87,7 @@ def typed_answer(row, probabilities):
61
  class DecisionEngine:
62
  def __init__(self, checkpoint, model_code, *, device='cuda:0', max_length=16384,
63
  batch_size=8, temperatures=None, model_name='local-decision-research'):
 
64
  import torch
65
  path=Path(model_code)/'decision_model.py'
66
  spec=importlib.util.spec_from_file_location('research_decision_runtime',path)
 
10
  from pathlib import Path
11
 
12
 
13
+ def prepare_runtime_profile(checkpoint, device='cuda:0'):
14
+ # This profile is verified before any dependency import can choose kernels.
15
+ import hashlib, json, sys
16
+ root=Path(checkpoint);runtime=json.loads((root/'runtime.json').read_text())
17
+ spec=runtime.get('normalization_profile')
18
+ if spec is None:
19
+ if '_decision_process_normalization_profile_v1' in sys.modules:
20
+ raise RuntimeError('Use separate processes for profiled and unprofiled models')
21
+ return None
22
+ import torch
23
+ target=torch.device(device)
24
+ if target.type!='cuda' or not torch.cuda.is_available():
25
+ raise RuntimeError('The bound profile requires a ROCm CUDA device')
26
+ arch=getattr(torch.cuda.get_device_properties(target),'gcnArchName','').split(':')[0]
27
+ if arch!=spec['validated_arch']:
28
+ raise RuntimeError('Target GPU architecture does not match the bound profile: '+arch)
29
+ relative=Path(spec['loader_file'])
30
+ if relative.is_absolute() or '..' in relative.parts:raise ValueError('Unsafe profile loader path')
31
+ path=root/relative
32
+ if hashlib.sha256(path.read_bytes()).hexdigest()!=spec['loader_sha256']:
33
+ raise ValueError('Bound runtime profile loader changed')
34
+ definition=importlib.util.spec_from_file_location('decision_bundle_runtime_profile',path)
35
+ module=importlib.util.module_from_spec(definition);definition.loader.exec_module(module)
36
+ return module.ensure_profile(root)
37
+
38
+
39
  def question_row(state, name, question):
40
  kind=question.get('type')
41
  if kind not in {'choice','noul','score'}:raise ValueError('Unknown question type')
 
87
  class DecisionEngine:
88
  def __init__(self, checkpoint, model_code, *, device='cuda:0', max_length=16384,
89
  batch_size=8, temperatures=None, model_name='local-decision-research'):
90
+ self.normalization_profile=prepare_runtime_profile(checkpoint, device=device)
91
  import torch
92
  path=Path(model_code)/'decision_model.py'
93
  spec=importlib.util.spec_from_file_location('research_decision_runtime',path)
code/profile_guard.py ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Fail-closed official FLA strict-config setup for one isolated process.
2
+
3
+ No model/prompt/head changes. The guard prevents FLA STRICT's ordinary missing
4
+ key fallback and records actual configured calls. This is a diagnostic module,
5
+ not an installed change to the published wrapper or dependency environment.
6
+ """
7
+ import hashlib,importlib,json,os,sys
8
+ from pathlib import Path
9
+
10
+ def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
11
+ def serialized(key):return json.dumps(key,separators=(',',':'),sort_keys=True)
12
+ def config_fields(config):
13
+ if isinstance(config,dict):return {k:config.get(k) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
14
+ return {k:getattr(config,k,None) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
15
+ def validate_profile(path,expected_sha):
16
+ path=Path(path).resolve()
17
+ if sha(path)!=expected_sha:raise ValueError('Profile hash changed')
18
+ profile=json.loads(path.read_text())
19
+ if profile['format']!='decision-fla-l2norm-profile-v1' or profile['cache_mode']!='strict':raise ValueError('Profile format/mode unsupported')
20
+ if len(profile['files'])!=1 or profile['files'][0]['file']!='l2norm_fwd_kernel.json':raise ValueError('Unexpected profile file set')
21
+ f=path.parent/'l2norm_fwd_kernel.json'
22
+ if sha(f)!=profile['files'][0]['sha256']:raise ValueError('Explicit kernel config changed')
23
+ data=json.loads(f.read_text());entries={}
24
+ if data.get('default_config') is not None:raise ValueError('Implicit fallback defaults forbidden')
25
+ for h,item in data['autotune_entries'].items():
26
+ key=item['autotune_key'];encoded=serialized(key)
27
+ if hashlib.md5(encoded.encode()).hexdigest()!=h or encoded in entries:raise ValueError('Invalid/duplicate key')
28
+ if len(key)!=5 or key[0]!=128 or type(key[1]) is not int or not 1<=key[1]<=32 or key[2:]!=['torch.bfloat16','torch.bfloat16','torch.float32']:raise ValueError('Unsupported numerical key')
29
+ c=item['config']
30
+ if c['kwargs'].keys()!={'BT'} or c['kwargs']['BT'] not in [8,16,32] or c['num_warps'] not in [1,2,4,8,16] or c['num_stages']!=3 or c['num_ctas']!=1 or any(c.get(x) is not None for x in ['maxnreg','pre_hook','ir_override']):raise ValueError('Unexpected launch configuration')
31
+ entries[encoded]=c
32
+ if {json.loads(k)[1] for k in entries}!=set(range(1,33)):raise ValueError('Incomplete legal NB coverage')
33
+ return path,profile,entries
34
+
35
+ def attach_guard(kernel,cache_module,entries,telemetry):
36
+ original=kernel.run
37
+ def guarded(*args,**kwargs):
38
+ if cache_module.FLA_CACHE_MODE is not cache_module.FlaCacheMode.STRICT:raise RuntimeError('FLA strict mode changed')
39
+ key=cache_module.AutotuneKey.build(kernel.arg_names,kernel.keys,args,kwargs);encoded=serialized(list(key.autotune_key))
40
+ if encoded not in entries:raise RuntimeError('Uncontracted FLA l2norm key: '+encoded)
41
+ expected=entries[encoded];loaded=cache_module.load_cached_config(kernel.kernel_name,key)
42
+ if config_fields(loaded)!=config_fields(expected):raise RuntimeError('FLA exact config lookup mismatch')
43
+ if key.autotune_key in kernel.cache and config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('A conflicting in-process kernel cache exists')
44
+ # Explicit official configuration load guarantees that the following original
45
+ # run finds this exact cache entry and cannot perform timing-based autotune.
46
+ kernel.maybe_load_cached_config(key)
47
+ if key.autotune_key not in kernel.cache or config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Official strict config did not load')
48
+ result=original(*args,**kwargs)
49
+ if config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Kernel config changed during call')
50
+ telemetry['calls']+=1;telemetry['keys'][encoded]=telemetry['keys'].get(encoded,0)+1
51
+ return result
52
+ kernel.run=guarded
53
+ return original
54
+
55
+ def install(profile_path,expected_sha):
56
+ path,profile,entries=validate_profile(profile_path,expected_sha)
57
+ if any(n=='fla' or n.startswith('fla.') for n in sys.modules):raise RuntimeError('Install profile before importing FLA; use a fresh isolated process')
58
+ for name,wanted in {'FLA_CACHE_MODE':'strict','FLA_CONFIG_DIR':str(path.parent)}.items():
59
+ actual=os.environ.get(name)
60
+ if actual is not None and actual!=wanted:raise RuntimeError('Conflicting '+name)
61
+ os.environ[name]=wanted
62
+ import torch,triton,fla
63
+ actual={'torch':str(torch.__version__),'hip':torch.version.hip,'triton':triton.__version__,'fla':fla.__version__}
64
+ for name,value in actual.items():
65
+ if value!=profile['runtime'][name]:raise RuntimeError('Runtime mismatch: '+name)
66
+ if not torch.cuda.is_available():raise RuntimeError('Profile is only qualified for the specified ROCm GPU')
67
+ arch=torch.cuda.get_device_properties(0).gcnArchName.split(':')[0]
68
+ if arch!=profile['runtime']['gpu_arch']:raise RuntimeError('Unsupported GPU architecture '+arch)
69
+ module=importlib.import_module('fla.modules.l2norm');cache_module=importlib.import_module('fla.ops.utils.cache');root=Path(fla.__file__).parent
70
+ for name,value in profile['fla_source_sha256'].items():
71
+ if sha(root/name)!=value:raise RuntimeError('Pinned FLA source changed: '+name)
72
+ kernel=module.l2norm_fwd_kernel
73
+ if kernel.kernel_name!='l2norm_fwd_kernel' or kernel.keys!=['D','NB'] or kernel.cache:raise RuntimeError('Kernel identity or fresh-cache precondition failed')
74
+ telemetry={'profile_sha256':expected_sha,'status':'installed','calls':0,'keys':{},'strict_guard':True,'unknown_keys':'raise','autotune_fallback_permitted':False,'runtime':actual,'gpu_arch':arch,'process_scope':'one explicitly profiled Sol model; other model loading in this process is not supported'}
75
+ attach_guard(kernel,cache_module,entries,telemetry)
76
+ # The validated inference path only uses the vectorized D128 forward kernel.
77
+ # Other dimensions/backward must not silently enter a different autotuner.
78
+ for name in ['l2norm_fwd_kernel1','l2norm_bwd_kernel','l2norm_bwd_kernel1']:
79
+ other=getattr(module,name)
80
+ def reject(*args,_name=name,**kwargs):raise RuntimeError('Uncontracted normalization kernel: '+_name)
81
+ other.run=reject
82
+ return telemetry
code/runtime_profile.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Bundle-local automatic launch-profile binding, before importing FLA.
2
+
3
+ One profile is active per Python process. Repeated loading of the same verified
4
+ profile is allowed; mixing with an unprofiled/different-profile model is not.
5
+ """
6
+ import hashlib,importlib.util,json,os,sys,types
7
+ from pathlib import Path
8
+ STATE='_decision_process_normalization_profile_v1'
9
+ def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
10
+ def safe(root,relative):
11
+ p=Path(relative)
12
+ if p.is_absolute() or '..' in p.parts:raise ValueError('Unsafe runtime-profile path')
13
+ return root/p
14
+
15
+ def ensure_profile(bundle):
16
+ bundle=Path(bundle).resolve();runtime=json.loads((bundle/'runtime.json').read_text());spec=runtime.get('normalization_profile');active=sys.modules.get(STATE)
17
+ if spec is None:
18
+ if active is not None:raise RuntimeError('Load unprofiled and profiled Decision models in separate processes')
19
+ return None
20
+ if spec.get('kind')!='decision-fla-l2norm-profile-v1' or spec.get('validated_arch')!='gfx942':raise ValueError('Unknown normalization profile contract')
21
+ profile=safe(bundle,spec['profile_file']);guard=safe(bundle,spec['guard_file'])
22
+ if sha(profile)!=spec['profile_sha256'] or sha(guard)!=spec['guard_sha256']:raise ValueError('Bound runtime profile bytes changed')
23
+ if json.loads((bundle/'decision_config.json').read_text()).get('base_model')!='Qwen/Qwen3.5-2B':raise ValueError('This bundle profile is bound to the validated Sol family only')
24
+ if active is not None:
25
+ if active.profile_sha256!=spec['profile_sha256'] or active.guard_sha256!=spec['guard_sha256']:raise RuntimeError('Different Decision normalization profile is already active; use a separate process')
26
+ active.guard.validate_profile(profile,spec['profile_sha256'])
27
+ if os.environ.get('FLA_CACHE_MODE')!='strict' or os.environ.get('FLA_CONFIG_DIR')!=active.profile_dir:raise RuntimeError('Active FLA profile environment changed')
28
+ return {'profile_sha256':active.profile_sha256,'guard_sha256':active.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
29
+ module_spec=importlib.util.spec_from_file_location('decision_profile_guard_'+spec['guard_sha256'][:16],guard);module=importlib.util.module_from_spec(module_spec);module_spec.loader.exec_module(module)
30
+ telemetry=module.install(profile,spec['profile_sha256'])
31
+ state=types.ModuleType(STATE);state.profile_sha256=spec['profile_sha256'];state.guard_sha256=spec['guard_sha256'];state.profile_dir=str(profile.parent);state.guard=module;state.telemetry=telemetry;sys.modules[STATE]=state
32
+ return {'profile_sha256':state.profile_sha256,'guard_sha256':state.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
33
+
34
+ def active_telemetry():
35
+ state=sys.modules.get(STATE)
36
+ return None if state is None else dict(state.telemetry)
decision_head.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:308bc01085a2bcee89a6668cb06ca4b771e01338040689527127bdefd3cc7415
3
  size 8424272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d
3
  size 8424272
metrics/asset-hashes.json DELETED
@@ -1,17 +0,0 @@
1
- {
2
- "readout.png": "fb6b3d1cf149a1a3ccb584eb1c213df73c07bcd328732b4238928e2a2a4af085",
3
- "readout.pdf": "d489b0d1814e37495a8612c12fabd3bf4a6e73d11c7e356bf1ce0cedded8a745",
4
- "decision-family-header.png": "213511289ce8df038d938ac470e803c427ed57f0f85cc397dd4d79964b866541",
5
- "readout.svg": "701084bf7b0858a3adb247ae446df737b77d4b75004341dfd5f956b933f3d253",
6
- "architecture.svg": "215d9af6f94cc24847fc1d23a0b2a287b36afdc1e10ea4f5c82895b2fe103002",
7
- "decision-quality.png": "50af70f9120e1f13024a18b264d9471c0f441464f33188306e5f5e0dc0cc4955",
8
- "decision-capabilities.png": "ec86958b569c84c9ed8fcd8180095f88bbc3fe8f8ded70f685b08bd6f59f2ef6",
9
- "decision-capabilities.pdf": "8cd467055c05d5c7616e776174ed0e3e953e71672558ea346e347e8663b5e5ce",
10
- "decision-quality.pdf": "403a02af4ba6292d27fe2863dc568a026278002fbb58082befed37ac55f31ce4",
11
- "architecture-atlas.pdf": "d1582ac115aa6a40129b6221e15a197e33d20847e03d69396f02fdab82771376",
12
- "decision-mark.png": "d83cf7878c3839f3085feb8c02334217ad2fef0a615b805e1726b292552c46e8",
13
- "decision-capabilities.svg": "e3b03c292b420b039b972756ae40e93e73c0610672a9bca5d795736faeb7f697",
14
- "decision-quality.svg": "b430df49b3c74fa0cd8b7e39ab1a7f6768295feaa423213816fa6ce8f2fde8a2",
15
- "architecture.pdf": "ee87f997a0aa4343c2ec66038b7526361e0a3c0ae02b383f8928aac9dd9bd74c",
16
- "architecture.png": "8fcfa10b3941ee4aec1f5fba20a25e382a7d5de83074071c9ef68e627996de40"
17
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
metrics/natural-confirmation.json ADDED
@@ -0,0 +1,877 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "sol-v4-observed-reading-public-1",
3
+ "requested_examples": 480,
4
+ "families": {
5
+ "boolq": {
6
+ "requested": 160,
7
+ "false": 80,
8
+ "true": 80,
9
+ "weight": "1/2"
10
+ },
11
+ "belebele_en": {
12
+ "requested": 160,
13
+ "weight": "1/4"
14
+ },
15
+ "belebele_zh": {
16
+ "requested": 160,
17
+ "weight": "1/4"
18
+ }
19
+ },
20
+ "score_definition": "0.5 * BoolQ accuracy + 0.25 * Belebele English accuracy + 0.25 * Belebele Chinese accuracy; all 480 requested examples retained.",
21
+ "models": {
22
+ "jev-1.13.0": {
23
+ "label": "Jev \u00b7 1.13.0",
24
+ "weighted_accuracy": 0.9453125,
25
+ "weighted_accuracy_exact": {
26
+ "numerator": 121,
27
+ "denominator": 128
28
+ },
29
+ "weighted_accuracy_ci95": [
30
+ 0.9200550899983875,
31
+ 0.9676459580838324
32
+ ],
33
+ "families": {
34
+ "boolq": {
35
+ "correct": 148,
36
+ "requested": 160,
37
+ "accuracy": 0.925,
38
+ "ci95": [
39
+ 0.88125,
40
+ 0.9625
41
+ ]
42
+ },
43
+ "belebele_en": {
44
+ "correct": 155,
45
+ "requested": 160,
46
+ "accuracy": 0.96875,
47
+ "ci95": [
48
+ 0.9386503067484663,
49
+ 0.9936708860759493
50
+ ]
51
+ },
52
+ "belebele_zh": {
53
+ "correct": 154,
54
+ "requested": 160,
55
+ "accuracy": 0.9625,
56
+ "ci95": [
57
+ 0.9308176100628931,
58
+ 0.9877300613496932
59
+ ]
60
+ }
61
+ },
62
+ "unweighted": {
63
+ "correct": 457,
64
+ "requested": 480,
65
+ "accuracy": 0.9520833333333333
66
+ },
67
+ "coverage": {
68
+ "requested": 480,
69
+ "accepted": 480,
70
+ "finite_valid_outputs": 480,
71
+ "full_input_verified": null,
72
+ "truncated": null,
73
+ "retention": "Unknown inside the closed API."
74
+ },
75
+ "typed_probability_metrics": {
76
+ "noul": {
77
+ "n": 160,
78
+ "valid_probability_coverage": 160,
79
+ "missing_or_invalid": 0,
80
+ "accuracy": 0.925,
81
+ "brier": 0.10445375000000001,
82
+ "nll": 0.19516064889282234,
83
+ "zero_true_probability_count": 0,
84
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
85
+ "soft_target_coverage": 0,
86
+ "soft_target_nll": null,
87
+ "class_support": {
88
+ "0": 80,
89
+ "1": 80
90
+ },
91
+ "recall_false": 0.925,
92
+ "recall_true": 0.925,
93
+ "balanced_accuracy": 0.925,
94
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
95
+ },
96
+ "choice": {
97
+ "n": 320,
98
+ "valid_probability_coverage": 320,
99
+ "missing_or_invalid": 0,
100
+ "accuracy": 0.965625,
101
+ "brier": 0.040956875,
102
+ "nll": 0.07843730477267183,
103
+ "zero_true_probability_count": 0,
104
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
105
+ "soft_target_coverage": 0,
106
+ "soft_target_nll": null
107
+ }
108
+ },
109
+ "natural_choice": {
110
+ "correct": 309,
111
+ "requested": 320,
112
+ "accuracy": 0.965625
113
+ }
114
+ },
115
+ "decider": {
116
+ "label": "Decider \u00b7 2B",
117
+ "weighted_accuracy": 0.9203125,
118
+ "weighted_accuracy_exact": {
119
+ "numerator": 589,
120
+ "denominator": 640
121
+ },
122
+ "weighted_accuracy_ci95": [
123
+ 0.8916901676829269,
124
+ 0.9464989496201366
125
+ ],
126
+ "families": {
127
+ "boolq": {
128
+ "correct": 147,
129
+ "requested": 160,
130
+ "accuracy": 0.91875,
131
+ "ci95": [
132
+ 0.875,
133
+ 0.95625
134
+ ]
135
+ },
136
+ "belebele_en": {
137
+ "correct": 149,
138
+ "requested": 160,
139
+ "accuracy": 0.93125,
140
+ "ci95": [
141
+ 0.8902439024390244,
142
+ 0.967948717948718
143
+ ]
144
+ },
145
+ "belebele_zh": {
146
+ "correct": 146,
147
+ "requested": 160,
148
+ "accuracy": 0.9125,
149
+ "ci95": [
150
+ 0.8630864845938376,
151
+ 0.9559748427672956
152
+ ]
153
+ }
154
+ },
155
+ "unweighted": {
156
+ "correct": 442,
157
+ "requested": 480,
158
+ "accuracy": 0.9208333333333333
159
+ },
160
+ "coverage": {
161
+ "requested": 480,
162
+ "accepted": 480,
163
+ "finite_valid_outputs": 480,
164
+ "full_input_verified": 480,
165
+ "truncated": 0,
166
+ "retention": "Complete frozen inputs verified."
167
+ },
168
+ "typed_probability_metrics": {
169
+ "noul": {
170
+ "n": 160,
171
+ "valid_probability_coverage": 160,
172
+ "missing_or_invalid": 0,
173
+ "accuracy": 0.91875,
174
+ "brier": 0.14787474037499998,
175
+ "nll": 0.2631311328165121,
176
+ "zero_true_probability_count": 0,
177
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
178
+ "soft_target_coverage": 0,
179
+ "soft_target_nll": null,
180
+ "class_support": {
181
+ "0": 80,
182
+ "1": 80
183
+ },
184
+ "recall_false": 0.9375,
185
+ "recall_true": 0.9,
186
+ "balanced_accuracy": 0.91875,
187
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
188
+ },
189
+ "choice": {
190
+ "n": 320,
191
+ "valid_probability_coverage": 320,
192
+ "missing_or_invalid": 0,
193
+ "accuracy": 0.921875,
194
+ "brier": 0.12688092915001592,
195
+ "nll": 0.24728184456069316,
196
+ "zero_true_probability_count": 0,
197
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
198
+ "soft_target_coverage": 0,
199
+ "soft_target_nll": null
200
+ }
201
+ },
202
+ "natural_choice": {
203
+ "correct": 295,
204
+ "requested": 320,
205
+ "accuracy": 0.921875
206
+ }
207
+ },
208
+ "nox-v1.1": {
209
+ "label": "Nox \u00b7 4B \u00b7 v1.1",
210
+ "weighted_accuracy": 0.784375,
211
+ "weighted_accuracy_exact": {
212
+ "numerator": 251,
213
+ "denominator": 320
214
+ },
215
+ "weighted_accuracy_ci95": [
216
+ 0.7399293154761905,
217
+ 0.8265123552522746
218
+ ],
219
+ "families": {
220
+ "boolq": {
221
+ "correct": 140,
222
+ "requested": 160,
223
+ "accuracy": 0.875,
224
+ "ci95": [
225
+ 0.81875,
226
+ 0.925
227
+ ]
228
+ },
229
+ "belebele_en": {
230
+ "correct": 114,
231
+ "requested": 160,
232
+ "accuracy": 0.7125,
233
+ "ci95": [
234
+ 0.6380368098159509,
235
+ 0.782608695652174
236
+ ]
237
+ },
238
+ "belebele_zh": {
239
+ "correct": 108,
240
+ "requested": 160,
241
+ "accuracy": 0.675,
242
+ "ci95": [
243
+ 0.5975609756097561,
244
+ 0.7469135802469136
245
+ ]
246
+ }
247
+ },
248
+ "unweighted": {
249
+ "correct": 362,
250
+ "requested": 480,
251
+ "accuracy": 0.7541666666666667
252
+ },
253
+ "coverage": {
254
+ "requested": 480,
255
+ "accepted": 480,
256
+ "finite_valid_outputs": 480,
257
+ "full_input_verified": 480,
258
+ "truncated": 0,
259
+ "retention": "Complete frozen inputs verified."
260
+ },
261
+ "typed_probability_metrics": {
262
+ "noul": {
263
+ "n": 160,
264
+ "valid_probability_coverage": 160,
265
+ "missing_or_invalid": 0,
266
+ "accuracy": 0.875,
267
+ "brier": 0.21024758152305395,
268
+ "nll": 0.4175424597184884,
269
+ "zero_true_probability_count": 0,
270
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
271
+ "soft_target_coverage": 0,
272
+ "soft_target_nll": null,
273
+ "class_support": {
274
+ "0": 80,
275
+ "1": 80
276
+ },
277
+ "recall_false": 0.875,
278
+ "recall_true": 0.875,
279
+ "balanced_accuracy": 0.875,
280
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
281
+ },
282
+ "choice": {
283
+ "n": 320,
284
+ "valid_probability_coverage": 320,
285
+ "missing_or_invalid": 0,
286
+ "accuracy": 0.69375,
287
+ "brier": 0.4329685852043793,
288
+ "nll": 1.0936456725285504,
289
+ "zero_true_probability_count": 0,
290
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
291
+ "soft_target_coverage": 0,
292
+ "soft_target_nll": null
293
+ }
294
+ },
295
+ "natural_choice": {
296
+ "correct": 222,
297
+ "requested": 320,
298
+ "accuracy": 0.6937500000000001
299
+ }
300
+ },
301
+ "sol-v1.1": {
302
+ "label": "Sol \u00b7 2B \u00b7 v1.1",
303
+ "weighted_accuracy": 0.734375,
304
+ "weighted_accuracy_exact": {
305
+ "numerator": 47,
306
+ "denominator": 64
307
+ },
308
+ "weighted_accuracy_ci95": [
309
+ 0.6869478485838779,
310
+ 0.7795678401898735
311
+ ],
312
+ "families": {
313
+ "boolq": {
314
+ "correct": 127,
315
+ "requested": 160,
316
+ "accuracy": 0.79375,
317
+ "ci95": [
318
+ 0.73125,
319
+ 0.85625
320
+ ]
321
+ },
322
+ "belebele_en": {
323
+ "correct": 109,
324
+ "requested": 160,
325
+ "accuracy": 0.68125,
326
+ "ci95": [
327
+ 0.60625,
328
+ 0.7532467532467533
329
+ ]
330
+ },
331
+ "belebele_zh": {
332
+ "correct": 107,
333
+ "requested": 160,
334
+ "accuracy": 0.66875,
335
+ "ci95": [
336
+ 0.5925925925925926,
337
+ 0.7419354838709677
338
+ ]
339
+ }
340
+ },
341
+ "unweighted": {
342
+ "correct": 343,
343
+ "requested": 480,
344
+ "accuracy": 0.7145833333333333
345
+ },
346
+ "coverage": {
347
+ "requested": 480,
348
+ "accepted": 480,
349
+ "finite_valid_outputs": 480,
350
+ "full_input_verified": 480,
351
+ "truncated": 0,
352
+ "retention": "Complete frozen inputs verified."
353
+ },
354
+ "typed_probability_metrics": {
355
+ "noul": {
356
+ "n": 160,
357
+ "valid_probability_coverage": 160,
358
+ "missing_or_invalid": 0,
359
+ "accuracy": 0.79375,
360
+ "brier": 0.3233785546556422,
361
+ "nll": 0.6047157313227172,
362
+ "zero_true_probability_count": 0,
363
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
364
+ "soft_target_coverage": 0,
365
+ "soft_target_nll": null,
366
+ "class_support": {
367
+ "0": 80,
368
+ "1": 80
369
+ },
370
+ "recall_false": 0.7375,
371
+ "recall_true": 0.85,
372
+ "balanced_accuracy": 0.79375,
373
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
374
+ },
375
+ "choice": {
376
+ "n": 320,
377
+ "valid_probability_coverage": 320,
378
+ "missing_or_invalid": 0,
379
+ "accuracy": 0.675,
380
+ "brier": 0.4428554242484616,
381
+ "nll": 1.0762990015746408,
382
+ "zero_true_probability_count": 0,
383
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
384
+ "soft_target_coverage": 0,
385
+ "soft_target_nll": null
386
+ }
387
+ },
388
+ "natural_choice": {
389
+ "correct": 216,
390
+ "requested": 320,
391
+ "accuracy": 0.675
392
+ }
393
+ },
394
+ "qwen35-4b-lm-head": {
395
+ "label": "Qwen3.5 \u00b7 4B \u00b7 untuned",
396
+ "weighted_accuracy": 0.8796875,
397
+ "weighted_accuracy_exact": {
398
+ "numerator": 563,
399
+ "denominator": 640
400
+ },
401
+ "weighted_accuracy_ci95": [
402
+ 0.8453125000000001,
403
+ 0.9125
404
+ ],
405
+ "families": {
406
+ "boolq": {
407
+ "correct": 134,
408
+ "requested": 160,
409
+ "accuracy": 0.8375,
410
+ "ci95": [
411
+ 0.78125,
412
+ 0.89375
413
+ ]
414
+ },
415
+ "belebele_en": {
416
+ "correct": 149,
417
+ "requested": 160,
418
+ "accuracy": 0.93125,
419
+ "ci95": [
420
+ 0.89171974522293,
421
+ 0.967948717948718
422
+ ]
423
+ },
424
+ "belebele_zh": {
425
+ "correct": 146,
426
+ "requested": 160,
427
+ "accuracy": 0.9125,
428
+ "ci95": [
429
+ 0.8647011385199241,
430
+ 0.9559748427672956
431
+ ]
432
+ }
433
+ },
434
+ "unweighted": {
435
+ "correct": 429,
436
+ "requested": 480,
437
+ "accuracy": 0.89375
438
+ },
439
+ "coverage": {
440
+ "requested": 480,
441
+ "accepted": 480,
442
+ "finite_valid_outputs": 480,
443
+ "full_input_verified": 480,
444
+ "truncated": 0,
445
+ "retention": "Complete frozen inputs verified."
446
+ },
447
+ "typed_probability_metrics": {
448
+ "noul": {
449
+ "n": 160,
450
+ "valid_probability_coverage": 160,
451
+ "missing_or_invalid": 0,
452
+ "accuracy": 0.8375,
453
+ "brier": 0.2685250413432339,
454
+ "nll": 0.43323629871265307,
455
+ "zero_true_probability_count": 0,
456
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
457
+ "soft_target_coverage": 0,
458
+ "soft_target_nll": null,
459
+ "class_support": {
460
+ "0": 80,
461
+ "1": 80
462
+ },
463
+ "recall_false": 0.7875,
464
+ "recall_true": 0.8875,
465
+ "balanced_accuracy": 0.8374999999999999,
466
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
467
+ },
468
+ "choice": {
469
+ "n": 320,
470
+ "valid_probability_coverage": 320,
471
+ "missing_or_invalid": 0,
472
+ "accuracy": 0.921875,
473
+ "brier": 0.12427284941036536,
474
+ "nll": 0.26369763063759766,
475
+ "zero_true_probability_count": 0,
476
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
477
+ "soft_target_coverage": 0,
478
+ "soft_target_nll": null
479
+ }
480
+ },
481
+ "natural_choice": {
482
+ "correct": 295,
483
+ "requested": 320,
484
+ "accuracy": 0.921875
485
+ }
486
+ },
487
+ "qwen35-2b-lm-head": {
488
+ "label": "Qwen3.5 \u00b7 2B \u00b7 untuned",
489
+ "weighted_accuracy": 0.7375,
490
+ "weighted_accuracy_exact": {
491
+ "numerator": 59,
492
+ "denominator": 80
493
+ },
494
+ "weighted_accuracy_ci95": [
495
+ 0.6917801490514904,
496
+ 0.7823309748427674
497
+ ],
498
+ "families": {
499
+ "boolq": {
500
+ "correct": 104,
501
+ "requested": 160,
502
+ "accuracy": 0.65,
503
+ "ci95": [
504
+ 0.575,
505
+ 0.725
506
+ ]
507
+ },
508
+ "belebele_en": {
509
+ "correct": 134,
510
+ "requested": 160,
511
+ "accuracy": 0.8375,
512
+ "ci95": [
513
+ 0.7784810126582279,
514
+ 0.89171974522293
515
+ ]
516
+ },
517
+ "belebele_zh": {
518
+ "correct": 130,
519
+ "requested": 160,
520
+ "accuracy": 0.8125,
521
+ "ci95": [
522
+ 0.746986092518221,
523
+ 0.8727272727272727
524
+ ]
525
+ }
526
+ },
527
+ "unweighted": {
528
+ "correct": 368,
529
+ "requested": 480,
530
+ "accuracy": 0.7666666666666667
531
+ },
532
+ "coverage": {
533
+ "requested": 480,
534
+ "accepted": 480,
535
+ "finite_valid_outputs": 480,
536
+ "full_input_verified": 480,
537
+ "truncated": 0,
538
+ "retention": "Complete frozen inputs verified."
539
+ },
540
+ "typed_probability_metrics": {
541
+ "noul": {
542
+ "n": 160,
543
+ "valid_probability_coverage": 160,
544
+ "missing_or_invalid": 0,
545
+ "accuracy": 0.65,
546
+ "brier": 0.4013130395712821,
547
+ "nll": 0.5740625822372918,
548
+ "zero_true_probability_count": 0,
549
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
550
+ "soft_target_coverage": 0,
551
+ "soft_target_nll": null,
552
+ "class_support": {
553
+ "0": 80,
554
+ "1": 80
555
+ },
556
+ "recall_false": 0.9375,
557
+ "recall_true": 0.3625,
558
+ "balanced_accuracy": 0.65,
559
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
560
+ },
561
+ "choice": {
562
+ "n": 320,
563
+ "valid_probability_coverage": 320,
564
+ "missing_or_invalid": 0,
565
+ "accuracy": 0.825,
566
+ "brier": 0.23224826689575004,
567
+ "nll": 0.47036782610781563,
568
+ "zero_true_probability_count": 0,
569
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
570
+ "soft_target_coverage": 0,
571
+ "soft_target_nll": null
572
+ }
573
+ },
574
+ "natural_choice": {
575
+ "correct": 264,
576
+ "requested": 320,
577
+ "accuracy": 0.825
578
+ }
579
+ },
580
+ "laya-routed": {
581
+ "label": "Laya \u00b7 EN/ML",
582
+ "weighted_accuracy": 0.5125,
583
+ "weighted_accuracy_exact": {
584
+ "numerator": 41,
585
+ "denominator": 80
586
+ },
587
+ "weighted_accuracy_ci95": [
588
+ 0.465988068600994,
589
+ 0.5583870648734177
590
+ ],
591
+ "families": {
592
+ "boolq": {
593
+ "correct": 111,
594
+ "requested": 160,
595
+ "accuracy": 0.69375,
596
+ "ci95": [
597
+ 0.61875,
598
+ 0.7625
599
+ ]
600
+ },
601
+ "belebele_en": {
602
+ "correct": 61,
603
+ "requested": 160,
604
+ "accuracy": 0.38125,
605
+ "ci95": [
606
+ 0.30538922155688625,
607
+ 0.4591194968553459
608
+ ]
609
+ },
610
+ "belebele_zh": {
611
+ "correct": 45,
612
+ "requested": 160,
613
+ "accuracy": 0.28125,
614
+ "ci95": [
615
+ 0.2147239263803681,
616
+ 0.3515151515151515
617
+ ]
618
+ }
619
+ },
620
+ "unweighted": {
621
+ "correct": 217,
622
+ "requested": 480,
623
+ "accuracy": 0.45208333333333334
624
+ },
625
+ "coverage": {
626
+ "requested": 480,
627
+ "accepted": 480,
628
+ "finite_valid_outputs": 480,
629
+ "full_input_verified": 480,
630
+ "truncated": 0,
631
+ "retention": "Complete frozen inputs verified."
632
+ },
633
+ "typed_probability_metrics": {
634
+ "noul": {
635
+ "n": 160,
636
+ "valid_probability_coverage": 160,
637
+ "missing_or_invalid": 0,
638
+ "accuracy": 0.69375,
639
+ "brier": 0.41904120474999995,
640
+ "nll": 0.6464549663099783,
641
+ "zero_true_probability_count": 0,
642
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
643
+ "soft_target_coverage": 0,
644
+ "soft_target_nll": null,
645
+ "class_support": {
646
+ "0": 80,
647
+ "1": 80
648
+ },
649
+ "recall_false": 0.525,
650
+ "recall_true": 0.8625,
651
+ "balanced_accuracy": 0.6937500000000001,
652
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
653
+ },
654
+ "choice": {
655
+ "n": 320,
656
+ "valid_probability_coverage": 320,
657
+ "missing_or_invalid": 0,
658
+ "accuracy": 0.33125,
659
+ "brier": 0.8108916245962444,
660
+ "nll": 1.5869034433374543,
661
+ "zero_true_probability_count": 0,
662
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
663
+ "soft_target_coverage": 0,
664
+ "soft_target_nll": null
665
+ }
666
+ },
667
+ "natural_choice": {
668
+ "correct": 106,
669
+ "requested": 320,
670
+ "accuracy": 0.33125
671
+ }
672
+ }
673
+ },
674
+ "previous_sol_reference": {
675
+ "label": "Sol \u00b7 2B \u00b7 v1.0",
676
+ "weighted_accuracy": 0.69375,
677
+ "weighted_accuracy_exact": {
678
+ "numerator": 111,
679
+ "denominator": 160
680
+ },
681
+ "weighted_accuracy_ci95": [
682
+ 0.6437499999999999,
683
+ 0.7417484177215189
684
+ ],
685
+ "families": {
686
+ "boolq": {
687
+ "correct": 119,
688
+ "requested": 160,
689
+ "accuracy": 0.74375,
690
+ "ci95": [
691
+ 0.675,
692
+ 0.8125
693
+ ]
694
+ },
695
+ "belebele_en": {
696
+ "correct": 106,
697
+ "requested": 160,
698
+ "accuracy": 0.6625,
699
+ "ci95": [
700
+ 0.5870967741935483,
701
+ 0.7349397590361446
702
+ ]
703
+ },
704
+ "belebele_zh": {
705
+ "correct": 100,
706
+ "requested": 160,
707
+ "accuracy": 0.625,
708
+ "ci95": [
709
+ 0.546583850931677,
710
+ 0.7005988023952096
711
+ ]
712
+ }
713
+ },
714
+ "unweighted": {
715
+ "correct": 325,
716
+ "requested": 480,
717
+ "accuracy": 0.6770833333333334
718
+ },
719
+ "coverage": {
720
+ "requested": 480,
721
+ "accepted": 480,
722
+ "finite_valid_outputs": 480,
723
+ "full_input_verified": 480,
724
+ "truncated": 0,
725
+ "retention": "Complete frozen inputs verified."
726
+ },
727
+ "typed_probability_metrics": {
728
+ "noul": {
729
+ "n": 160,
730
+ "valid_probability_coverage": 160,
731
+ "missing_or_invalid": 0,
732
+ "accuracy": 0.74375,
733
+ "brier": 0.43895629019878823,
734
+ "nll": 0.9002462702649545,
735
+ "zero_true_probability_count": 0,
736
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
737
+ "soft_target_coverage": 0,
738
+ "soft_target_nll": null,
739
+ "class_support": {
740
+ "0": 80,
741
+ "1": 80
742
+ },
743
+ "recall_false": 0.575,
744
+ "recall_true": 0.9125,
745
+ "balanced_accuracy": 0.7437499999999999,
746
+ "recall_denominator": "All gold rows, including invalid/missing predictions as wrong"
747
+ },
748
+ "choice": {
749
+ "n": 320,
750
+ "valid_probability_coverage": 320,
751
+ "missing_or_invalid": 0,
752
+ "accuracy": 0.64375,
753
+ "brier": 0.5085778581327408,
754
+ "nll": 1.2810086045362705,
755
+ "zero_true_probability_count": 0,
756
+ "nll_policy": "Natural log; no epsilon clipping; infinity is explicit JSON string; proper scores conditional on valid coverage",
757
+ "soft_target_coverage": 0,
758
+ "soft_target_nll": null
759
+ }
760
+ }
761
+ },
762
+ "sol_update_comparison": {
763
+ "point_delta": 0.040625,
764
+ "exact_fraction": "13/320",
765
+ "paired_ci95": [
766
+ 0.010993975903614502,
767
+ 0.0703125
768
+ ]
769
+ },
770
+ "bootstrap": {
771
+ "resamples": 10000,
772
+ "seed": 202609215404,
773
+ "weights": {
774
+ "boolq": "1/2",
775
+ "belebele_en": "1/4",
776
+ "belebele_zh": "1/4"
777
+ },
778
+ "clusters": "BoolQ complete normalized passage; Belebele source link with all question IDs and both translations coupled. Identical random draws shared across models.",
779
+ "point_denominator": "All requested rows; invalid/unsupported count as incorrect.",
780
+ "readiness": "Descriptive confirmation only; no change to active V3 update policy."
781
+ },
782
+ "sources": [
783
+ {
784
+ "name": "BoolQ validation",
785
+ "revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5",
786
+ "license": "CC BY-SA 3.0",
787
+ "url": "https://huggingface.co/datasets/google/boolq/tree/35b264d03638db9f4ce671b711558bf7ff0f80d5"
788
+ },
789
+ {
790
+ "name": "Belebele English/Chinese test",
791
+ "revision": "7899cdfa4e1e0d733fd77c848e2c273cb1d32be2",
792
+ "license": "CC BY-SA 4.0",
793
+ "url": "https://huggingface.co/datasets/facebook/belebele/tree/7899cdfa4e1e0d733fd77c848e2c273cb1d32be2"
794
+ }
795
+ ],
796
+ "scope": "Evidence yes/no and four-choice reading comprehension; no ordinal Score or all-capability claim.",
797
+ "changes_release_gate": false,
798
+ "candidate_bundle_manifest_sha256": "009d4892300c0a98f6a9b61706cdb6cc7ce2f6883c25c120477b98fe7fe82d0f",
799
+ "candidate_normalization_profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
800
+ "nox_peer_bundle_manifest_sha256": "a8cfd42bc908fc93d5a2a20be527d09704b6e3f6c1217f27307c32cc2d822971",
801
+ "nox_peer_published_revision": "5bee3061a7abf675813963e4e05267cb1a28f92f",
802
+ "sol_v1_revision": "b6a4ce881ca625dcf19663a277d27577cfe4540a",
803
+ "independence": {
804
+ "excluded_from_custom_train_select_cal": true,
805
+ "original_human_annotations": true,
806
+ "new_human_adjudication": false,
807
+ "upstream_pretraining_exposure": "Unknown",
808
+ "first_global_unseal_at": "2026-09-21T16:40:29.779741+00:00",
809
+ "candidate_frozen_before_first_unseal": false,
810
+ "subsequent_use": "Observed regression for candidates not frozen before first exposure.",
811
+ "source_parent_clusters": {
812
+ "boolq_passages": 160,
813
+ "belebele_shared_passages": 144
814
+ },
815
+ "overlap_audit": "No exact or character-5-gram Jaccard >=0.5 overlap across 51 custom-data and earlier-panel files; this is not proof of semantic or upstream-pretraining independence.",
816
+ "boolq_grouping_limit": "Passage grouping only; this distribution lacks article titles, so article-level independence is not established.",
817
+ "candidate_exposure": {
818
+ "global_first_unseal": {
819
+ "policy_sha256": "259b907f3634471f7c077acbe6cb4eba6ebb85d24ba305c86ea6a679349491c3",
820
+ "first_global_unseal_at": "2026-09-21T16:40:29.779741+00:00",
821
+ "first_candidate": "Decision-1.0-Nox-stage4-selected",
822
+ "first_candidate_bundle_manifest_sha256": "a8cfd42bc908fc93d5a2a20be527d09704b6e3f6c1217f27307c32cc2d822971",
823
+ "first_candidate_dispatch_sha256": "c04b96e7beb312cfd2196299aaaef23c8bd5a0acccc01c7d83002da76830652a",
824
+ "meaning": "V4 natural confirmation is observed regression for all subsequent unfrozen candidates. This first exposure record is never replaced."
825
+ },
826
+ "this_evaluation_at": "2026-09-21T18:05:29.511389+00:00",
827
+ "candidate_frozen_at": "2026-09-21T17:53:15.626713+00:00",
828
+ "candidate_frozen_before_first_global_unseal": false,
829
+ "candidate_is_first_unseal_identity": false,
830
+ "confirmation_interpretation": "Observed regression for this candidate; do not relabel it as fresh confirmation.",
831
+ "historical_panel_key": "v4_natural_confirmation",
832
+ "historical_key_is_not_a_new_freshness_claim": true
833
+ }
834
+ },
835
+ "provenance": {
836
+ "sol_report": {
837
+ "aggregate": "bb5fab144cb57513bbb7ce22ce2c8af38a5a3d28709df362e230bfa2995454a9",
838
+ "statistics": "fa3fe24079d7221b19fa522b38da7d9b4c03687130749c46e77c029d0f7bf951",
839
+ "receipt": "968ab2a8c6228bd61444001d19529598c90f69f89b097b1269e132eb86b35878"
840
+ },
841
+ "nox_peer_report": {
842
+ "aggregate": "c55a218a533751cf6fd8e783182bb2edfa1731f1a7aa3446fbed4575d6291dcb",
843
+ "statistics": "23105cc6fbffb15b1fac390b64bf31b70403091b5f0d4a3de06a78a27b3c5474",
844
+ "receipt": "e3220736d2da9c87d7b08808fa8903af0ce1c90961f88e7a54a8348abca29794"
845
+ },
846
+ "nox_publication_receipt_sha256": "5e0a3b7ba2637c5689e225b850441d532816f41b9f9240cce73a3ec9b66c0ea5",
847
+ "policy_sha256": "259b907f3634471f7c077acbe6cb4eba6ebb85d24ba305c86ea6a679349491c3",
848
+ "suite_manifest_sha256": "969a92c4c5a2f395a3ba55d3a4aff28be060d89327c3b09de3e1f5188cf74830",
849
+ "prediction_sha256": {
850
+ "jev-1.13.0": {
851
+ "predictions_sha256": "d6583f903e974d70747fae2944014828954dc4d7344ceddcbb2cd1b06fbfc799"
852
+ },
853
+ "decider": {
854
+ "predictions_sha256": "626ef366ae471ec10dfb89ef2ff2f7b29a6c879976d3ff14fcbb2bafd7d9b045"
855
+ },
856
+ "nox-v1.1": {
857
+ "predictions_sha256": "dedd6747282d694f2e5c7a79ee4133225f01cb55b1c7d8fd6882d3dddce6c723",
858
+ "raw_T1_sha256": "08e3cd068b1ccbbffc216e53d6fb9deb64c5a66b2ce6a5b0cbe9aa2667304ea4"
859
+ },
860
+ "sol-v1.1": {
861
+ "predictions_sha256": "0c7156e7d8709175ecf352c8aa5e8511f1eb0669c965ea5311992efb3d479dc3",
862
+ "raw_T1_sha256": "c7d581453f2c0f51f89051bcd8574fc6333c39475b20fb7dcb82e757580f24bc"
863
+ },
864
+ "qwen35-4b-lm-head": {
865
+ "predictions_sha256": "7a6d252ccaa72bf54b94445190bbad023fb465a3eaf4cf657532e2cdd4f750f5"
866
+ },
867
+ "qwen35-2b-lm-head": {
868
+ "predictions_sha256": "93d84cdbe607691e2cef35ece3aa2dd5900f6410b0d72796b86d74c9156f3cdd"
869
+ },
870
+ "laya-routed": {
871
+ "predictions_sha256": "a10922609e2223ca0a818667bc1425d476aec58a4d881002e0e322596c1d0b8d"
872
+ }
873
+ }
874
+ },
875
+ "quality_only": true,
876
+ "timing_claims": false
877
+ }
metrics/quality-aggregate.json DELETED
The diff for this file is too large to render. See raw diff
 
metrics/timing-aggregate.json DELETED
The diff for this file is too large to render. See raw diff
 
model-card-example.json CHANGED
@@ -31,25 +31,25 @@
31
  "destination": {
32
  "type": "choice",
33
  "probabilities": {
34
- "billing": 0.9999959977119728,
35
- "technical": 4.00228802725166e-06
36
  },
37
- "confidence": 0.9999919954239456,
38
  "choice": "billing"
39
  },
40
  "refund_requested": {
41
  "type": "noul",
42
- "noul": 0.9998253639445287
43
  },
44
  "urgency": {
45
  "type": "score",
46
  "probabilities": {
47
- "0": 0.02944745914852826,
48
- "1": 0.9703999910699086,
49
- "2": 0.00015254978156315938
50
  },
51
- "confidence": 0.955599986604863,
52
- "score": 0.9707050906330349,
53
  "legend": {
54
  "0": "Routine information request with no payment problem or outage",
55
  "1": "A payment or billing problem, with no product outage",
@@ -64,6 +64,50 @@
64
  },
65
  "direct_engine_exact_response": true,
66
  "overflow_rejected": true,
67
- "bundle_manifest_sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c",
68
- "example_source_sha256": "d52c20b602bd1049aecda55244e71c27d8ce44bd1aa8b1941db62ae41d31ba12"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
69
  }
 
31
  "destination": {
32
  "type": "choice",
33
  "probabilities": {
34
+ "billing": 0.9998550772527804,
35
+ "technical": 0.00014492274721960988
36
  },
37
+ "confidence": 0.9997101545055609,
38
  "choice": "billing"
39
  },
40
  "refund_requested": {
41
  "type": "noul",
42
+ "noul": 0.9984572750397842
43
  },
44
  "urgency": {
45
  "type": "score",
46
  "probabilities": {
47
+ "0": 0.13640706014048662,
48
+ "1": 0.8623479461162635,
49
+ "2": 0.0012449937432499015
50
  },
51
+ "confidence": 0.7935219191743954,
52
+ "score": 0.8648379336027633,
53
  "legend": {
54
  "0": "Routine information request with no payment problem or outage",
55
  "1": "A payment or billing problem, with no product outage",
 
64
  },
65
  "direct_engine_exact_response": true,
66
  "overflow_rejected": true,
67
+ "overflow_message": "check: 40083 tokens exceeds max_length=16384; no truncation allowed",
68
+ "bundle_manifest_sha256": "009d4892300c0a98f6a9b61706cdb6cc7ce2f6883c25c120477b98fe7fe82d0f",
69
+ "runtime": {
70
+ "actual": {
71
+ "torch": "2.12.0+git6bbd260",
72
+ "hip": "7.2.53211",
73
+ "transformers": "5.17.0",
74
+ "fla": "0.5.2",
75
+ "tokenizers": "0.23.2",
76
+ "safetensors": "0.8.0",
77
+ "triton": "3.7.1",
78
+ "gated_delta": "fla.ops.gated_delta_rule.chunk"
79
+ },
80
+ "differences": {},
81
+ "matches_validated_runtime": true,
82
+ "normalization_profile": {
83
+ "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
84
+ "guard_sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452",
85
+ "validated_arch": "gfx942",
86
+ "automatic_bundle_binding": true,
87
+ "scope": "single profile per process"
88
+ }
89
+ },
90
+ "revision": null,
91
+ "device": "cuda:0",
92
+ "model_name": "Decision-1.0-Sol",
93
+ "example_source_sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6",
94
+ "normalization_telemetry": {
95
+ "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
96
+ "status": "installed",
97
+ "calls": 72,
98
+ "keys": {
99
+ "[128,1,\"torch.bfloat16\",\"torch.bfloat16\",\"torch.float32\"]": 72
100
+ },
101
+ "strict_guard": true,
102
+ "unknown_keys": "raise",
103
+ "autotune_fallback_permitted": false,
104
+ "runtime": {
105
+ "torch": "2.12.0+git6bbd260",
106
+ "hip": "7.2.53211",
107
+ "triton": "3.7.1",
108
+ "fla": "0.5.2"
109
+ },
110
+ "gpu_arch": "gfx942",
111
+ "process_scope": "one explicitly profiled Sol model; other model loading in this process is not supported"
112
+ }
113
  }
peer-projection-provenance.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "decision-public-peer-projection-v1",
3
+ "target_release": "Sol v1.1",
4
+ "peer_release": "Nox v1.1",
5
+ "peer_repository": "llm-semantic-router/Decision-1.0-Nox",
6
+ "peer_revision": "5bee3061a7abf675813963e4e05267cb1a28f92f",
7
+ "peer_bundle_manifest_sha256": "a8cfd42bc908fc93d5a2a20be527d09704b6e3f6c1217f27307c32cc2d822971",
8
+ "target_bundle_manifest_sha256": "009d4892300c0a98f6a9b61706cdb6cc7ce2f6883c25c120477b98fe7fe82d0f",
9
+ "target_report_sha256": {
10
+ "public": "d6712e27f1f0aed164404ad14dab4c920f024566d46dd42971fa385ce78b9c25",
11
+ "statistics": "af1eb2ea01eb096e7c41febb95dddeb76e76da0693fa40c3df25f39212dcca57",
12
+ "qualification": "0ac43b7803eedaf3c5c3cc450e9daddad5f9335b3ea82dc9401693a8cfc9e454",
13
+ "receipt": "bca099653d5b92cf282a5a82c19998c65dae0ac89d6fd039123b60b3680e85b3"
14
+ },
15
+ "peer_report_sha256": {
16
+ "public": "68adfdde02ec23a9947a1df8d44d3837bc66a23681a364bca8986f2644120b8b",
17
+ "statistics": "0f55bc193c0f7f4359dac177df1d03ef641ea4c09ca71e0bc7eec84452559f94",
18
+ "qualification": "469defd8f05af52eea4b7d2eec856484be694e82d6f7f4c930c71d8fc7028975",
19
+ "receipt": "927c92121c2df534f2ba12523e466ecebd8bd8936f949c7792de3e747c92a47c"
20
+ },
21
+ "peer_publication_receipt_sha256": "5e0a3b7ba2637c5689e225b850441d532816f41b9f9240cce73a3ec9b66c0ea5",
22
+ "common_policy_sha256": "359de2b6b81be91a66c6c538b3655bed06f4728357328cd318b1a765e64565cd",
23
+ "same_frozen_panels_families_baseline_predictions_and_statistical_sources_verified": true,
24
+ "peer_prediction_sha256": {
25
+ "old_core": "b706110fba0da5a7e00c122586094a3947077ba9142374207ecea56717acccb9",
26
+ "fresh_core": "a565bd5f362d31e51d631605bc3852700cca094c8047e0fec1672a3e4860a381",
27
+ "old_native": "c9304a3153f8dbe2598d2678b955c886b48276db44c4bdb6988ddaf81ec484e3",
28
+ "fresh_native": "0a969ae31845f3522651c32d7d269b81c9984bed352040a00e20bc7be097c330",
29
+ "old_core_raw_T1": "1c17319811808e45f7596f756cb07fd6f28b8e9e6187e9f55b9ebe11ae06ea31",
30
+ "fresh_core_raw_T1": "a370f801490f2d253dd96d40481be45ccf1338c5345e6ea2b008cf47e71db679"
31
+ },
32
+ "own_v1_0_qualification_unchanged": true,
33
+ "no_new_pairwise_Sol_Nox_test_claim": true,
34
+ "no_examples_or_predictions_read": true,
35
+ "exposure": "Observed regression for Sol; Nox first-exposure provenance preserved in its source receipt."
36
+ }
pyproject.toml CHANGED
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
 
5
  [project]
6
  name = "decision-local"
7
- version = "1.0.0"
8
  description = "Local typed inference for exported Decision decoder models"
9
  requires-python = ">=3.10"
10
  dependencies = []
 
4
 
5
  [project]
6
  name = "decision-local"
7
+ version = "1.1.0"
8
  description = "Local typed inference for exported Decision decoder models"
9
  requires-python = ">=3.10"
10
  dependencies = []
quality-metrics.json ADDED
The diff for this file is too large to render. See raw diff
 
release-manifest.json CHANGED
@@ -1,18 +1,18 @@
1
  {
2
  "format": "decision-public-release-v1",
3
- "status": "documentation-refresh",
4
- "bundle_manifest_sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c",
5
- "readiness_sha256": "46888ec7eef0c62fb477bbd2c1ca5368ab40c151221d3051debfcc9d6fcd8cc1",
6
- "model_card_sha256": "9eed30c689e947d913204fcfd839fb296d9a2e5921e9c6b67bd34791d2e6a09c",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Sol",
8
- "assembly_script_sha256": "0ca32612da99c16b9801149d69f375b157e51b6e426b0a74efa3efb6bcb9f792",
9
  "original_bundle_manifest_preserved": true,
10
  "files_exclude_this_manifest": true,
11
  "files": [
12
  {
13
  "file": "ATTRIBUTIONS.md",
14
- "bytes": 2610,
15
- "sha256": "47c35fdcde567fb511f31d1d7d0f2e2ba21daef0beb31217c025132aafe0d2e9"
16
  },
17
  {
18
  "file": "Dockerfile.runtime",
@@ -21,48 +21,58 @@
21
  },
22
  {
23
  "file": "EVALUATION.md",
24
- "bytes": 16973,
25
- "sha256": "1f4f7bc151e757da8e35db2bf690cd13839ab67de130655004400c2bd425ffb3"
26
- },
27
- {
28
- "file": "FIGURE-NOTICES.md",
29
- "bytes": 1041,
30
- "sha256": "d70794c70cc8811a74ef19e9dbeb7368d3ff89b71e0117bff40149dc9b72055d"
31
  },
32
  {
33
  "file": "LICENSE",
34
  "bytes": 11544,
35
  "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
36
  },
 
 
 
 
 
37
  {
38
  "file": "QUESTION-SCALING.md",
39
- "bytes": 3608,
40
- "sha256": "ad72203f319d9344ffa917b6d12ae34fd976c71c39fc715fa862dd23685bbbf1"
41
  },
42
  {
43
  "file": "QWEN-LICENSE",
44
  "bytes": 11544,
45
  "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
46
  },
 
 
 
 
 
47
  {
48
  "file": "README.md",
49
- "bytes": 4480,
50
- "sha256": "9eed30c689e947d913204fcfd839fb296d9a2e5921e9c6b67bd34791d2e6a09c"
51
  },
52
  {
53
  "file": "RUNTIME.md",
54
- "bytes": 5205,
55
- "sha256": "be35b2f8a9c977f38be9b5eb892fdc809db5167c327575c9311355aae3e781c7"
 
 
 
 
 
56
  },
57
  {
58
- "file": "TIMING.md",
59
- "bytes": 13432,
60
- "sha256": "bdf371d3e93056125840829105b7f5a91abce343889530c8f7cb339f6612f9a6"
61
  },
62
  {
63
  "file": "USAGE.md",
64
- "bytes": 5435,
65
- "sha256": "46ec1078d2bb8cb54a1850809a101f07a77bb54c18b92276cb89d40245dbbc8f"
66
  },
67
  {
68
  "file": "assets/architecture-atlas.pdf",
@@ -86,18 +96,18 @@
86
  },
87
  {
88
  "file": "assets/decision-capabilities.pdf",
89
- "bytes": 45170,
90
- "sha256": "1314aae927ea0aa598e62a80138e2ea9bfd8e851bc6faa91504d5aa691426e24"
91
  },
92
  {
93
  "file": "assets/decision-capabilities.png",
94
- "bytes": 276453,
95
- "sha256": "ed88c09a355c8583ac4d7b31bd4ff93b7be8725fde436f3d76e1754ae92a5962"
96
  },
97
  {
98
  "file": "assets/decision-capabilities.svg",
99
- "bytes": 69169,
100
- "sha256": "cae4bf2c1aa13e8d6dc94c435eed1fd8bde1e523fbfa83a0b7dde5ba37650022"
101
  },
102
  {
103
  "file": "assets/decision-family-header.png",
@@ -111,18 +121,18 @@
111
  },
112
  {
113
  "file": "assets/decision-quality.pdf",
114
- "bytes": 37484,
115
- "sha256": "aec3769c803f3b87668332e2a6d036cab1bd65b21c8ebbd919fbb79f1637290f"
116
  },
117
  {
118
  "file": "assets/decision-quality.png",
119
- "bytes": 175813,
120
- "sha256": "dbf0e32d3e7846f94b8d6fd224eaf54677cd1ecb140f1671fb1b80f2a6d25935"
121
  },
122
  {
123
  "file": "assets/decision-quality.svg",
124
- "bytes": 37293,
125
- "sha256": "e324b1446ac18beeefd6da1d3d4c2ec4db579b7d989390663c1ab5e43bf76841"
126
  },
127
  {
128
  "file": "assets/decision-question-scaling.pdf",
@@ -162,12 +172,12 @@
162
  {
163
  "file": "backbone/model.safetensors",
164
  "bytes": 3763685328,
165
- "sha256": "cab4b9b9dfd490781acf489b78b6d3faa1d56a1be8fab5d8baf6c464e4c0ea1d"
166
  },
167
  {
168
  "file": "bundle-manifest.json",
169
- "bytes": 4204,
170
- "sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c"
171
  },
172
  {
173
  "file": "chat_template.jinja",
@@ -176,8 +186,8 @@
176
  },
177
  {
178
  "file": "code/decision_api.py",
179
- "bytes": 6724,
180
- "sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21"
181
  },
182
  {
183
  "file": "code/decision_model.py",
@@ -189,6 +199,16 @@
189
  "bytes": 3164,
190
  "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
191
  },
 
 
 
 
 
 
 
 
 
 
192
  {
193
  "file": "decision_config.json",
194
  "bytes": 753,
@@ -197,17 +217,12 @@
197
  {
198
  "file": "decision_head.safetensors",
199
  "bytes": 8424272,
200
- "sha256": "308bc01085a2bcee89a6668cb06ca4b771e01338040689527127bdefd3cc7415"
201
- },
202
- {
203
- "file": "metrics/asset-hashes.json",
204
- "bytes": 1394,
205
- "sha256": "b4aec27dd4a4032c37ef013cda0cb72161b6c1de25161d61864a5f1971af952a"
206
  },
207
  {
208
- "file": "metrics/quality-aggregate.json",
209
- "bytes": 1214948,
210
- "sha256": "b2181b7636f0ddace6af33a45010bfb4e62fb32f987f4abe2d2a4d58b3a007e4"
211
  },
212
  {
213
  "file": "metrics/question-scaling.json",
@@ -215,30 +230,45 @@
215
  "sha256": "4708635c6f26cd0209dafb4e45af6921ed22091e1996bbeb5861c98214ff5886"
216
  },
217
  {
218
- "file": "metrics/timing-aggregate.json",
219
- "bytes": 148812,
220
- "sha256": "a0e5f1c94277b31a922ae1e1a708ce70a2f03b5e49996a688112fc1e9697615a"
221
  },
222
  {
223
- "file": "model-card-example.json",
224
- "bytes": 2271,
225
- "sha256": "77c4450389fe872d7e61ee4d02f0c89b3a0126913379255abc7f7f03ab5b8723"
226
  },
227
  {
228
  "file": "pyproject.toml",
229
  "bytes": 436,
230
- "sha256": "6a557fbe472027af103ca2cfc07e977caab3e257de930aa57cf9fea8cf90f8a0"
 
 
 
 
 
231
  },
232
  {
233
  "file": "runtime-build-provenance.json",
234
- "bytes": 2894,
235
- "sha256": "61d064b0db35d0ebd04a038e376896d2aab8b69e6525a8ec1f317f73b5284345"
236
  },
237
  {
238
  "file": "runtime-fla-requirements.lock",
239
  "bytes": 654,
240
  "sha256": "35e1fedff9ca49092a7ada277bbd794abe7e8474e14a678a7e4bcec6406d0eb5"
241
  },
 
 
 
 
 
 
 
 
 
 
242
  {
243
  "file": "runtime-provenance.json",
244
  "bytes": 5805,
@@ -246,8 +276,8 @@
246
  },
247
  {
248
  "file": "runtime.json",
249
- "bytes": 952,
250
- "sha256": "7ed527476826756d6076f23d0eaa8b4b53ff34b9f8bd57ae8e65b50e4d28cfa6"
251
  },
252
  {
253
  "file": "src/decision/__init__.py",
@@ -256,18 +286,48 @@
256
  },
257
  {
258
  "file": "src/decision/example.py",
259
- "bytes": 3381,
260
- "sha256": "d52c20b602bd1049aecda55244e71c27d8ce44bd1aa8b1941db62ae41d31ba12"
261
  },
262
  {
263
  "file": "src/decision/model.py",
264
- "bytes": 8790,
265
- "sha256": "ab428234d508a1245d3723c5a5bb0c796685abebd0447f77c7fbed9113954b21"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
266
  },
267
  {
268
  "file": "temperature.json",
269
- "bytes": 667,
270
- "sha256": "159d308a6b970c425b8938b6c50c4fff6a10bb40ca82b119e43386a855d609ab"
271
  },
272
  {
273
  "file": "tokenizer.json",
@@ -284,25 +344,31 @@
284
  "ATTRIBUTIONS.md": "copy",
285
  "Dockerfile.runtime": "copy",
286
  "EVALUATION.md": "copy",
287
- "FIGURE-NOTICES.md": "copy",
288
  "LICENSE": "copy",
 
 
289
  "QWEN-LICENSE": "copy",
290
- "README.md": "documentation-copy",
 
291
  "RUNTIME.md": "copy",
292
- "TIMING.md": "copy",
 
293
  "USAGE.md": "copy",
294
  "assets/architecture-atlas.pdf": "copy",
295
  "assets/architecture.pdf": "copy",
296
  "assets/architecture.png": "copy",
297
  "assets/architecture.svg": "copy",
298
- "assets/decision-capabilities.pdf": "documentation-copy",
299
- "assets/decision-capabilities.png": "documentation-copy",
300
- "assets/decision-capabilities.svg": "documentation-copy",
301
  "assets/decision-family-header.png": "copy",
302
  "assets/decision-mark.png": "copy",
303
- "assets/decision-quality.pdf": "documentation-copy",
304
- "assets/decision-quality.png": "documentation-copy",
305
- "assets/decision-quality.svg": "documentation-copy",
 
 
 
306
  "assets/readout.pdf": "copy",
307
  "assets/readout.png": "copy",
308
  "assets/readout.svg": "copy",
@@ -313,31 +379,34 @@
313
  "code/decision_api.py": "copy",
314
  "code/decision_model.py": "copy",
315
  "code/predict.py": "copy",
 
 
316
  "decision_config.json": "copy",
317
  "decision_head.safetensors": "hardlink",
318
- "metrics/asset-hashes.json": "copy",
319
- "metrics/quality-aggregate.json": "copy",
320
- "metrics/timing-aggregate.json": "copy",
321
  "model-card-example.json": "copy",
 
322
  "pyproject.toml": "copy",
 
323
  "runtime-build-provenance.json": "copy",
324
  "runtime-fla-requirements.lock": "copy",
 
 
325
  "runtime-provenance.json": "copy",
326
  "runtime.json": "copy",
327
  "src/decision/__init__.py": "copy",
328
  "src/decision/example.py": "copy",
329
  "src/decision/model.py": "copy",
 
 
 
 
 
 
330
  "temperature.json": "copy",
331
  "tokenizer.json": "copy",
332
- "tokenizer_config.json": "copy",
333
- "QUESTION-SCALING.md": "documentation-copy",
334
- "metrics/question-scaling.json": "documentation-copy",
335
- "assets/decision-question-scaling.pdf": "documentation-copy",
336
- "assets/decision-question-scaling.png": "documentation-copy",
337
- "assets/decision-question-scaling.svg": "documentation-copy"
338
  },
339
- "scope": "Inference weights, original numerical code, calibrated runtime metadata, wrapper, model card, license, attribution and approved assets only.",
340
- "previous_release_manifest_sha256": "996b5b0244bf7d75b7330f93560000701b1cc12bdd33fa50a09565d2eaee9d15",
341
- "documentation_refresh_script_sha256": "faf40a7508c73266e2337acac18d05d5fe1a119cee13768d41947b4369a44afc",
342
- "inference_files_unchanged": true
343
  }
 
1
  {
2
  "format": "decision-public-release-v1",
3
+ "status": "assembled-not-published",
4
+ "bundle_manifest_sha256": "009d4892300c0a98f6a9b61706cdb6cc7ce2f6883c25c120477b98fe7fe82d0f",
5
+ "readiness_sha256": "80c971663626c56275585c7e854e06b138e1459b28ef9d76293cb45d4fc11b68",
6
+ "model_card_sha256": "05cacb9febff3519c7ff3daad2ac2d872a06067019214c2d3cf8b1e466e26401",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Sol",
8
+ "assembly_script_sha256": "7831ef5952b8256e585d4efa91cac4ce87738f4eb93c84b1cf1e7250def1f430",
9
  "original_bundle_manifest_preserved": true,
10
  "files_exclude_this_manifest": true,
11
  "files": [
12
  {
13
  "file": "ATTRIBUTIONS.md",
14
+ "bytes": 4182,
15
+ "sha256": "76ff86ae499bdabb46ef75f974dfbc48eec794b8651bd9671780b55a1fc108e3"
16
  },
17
  {
18
  "file": "Dockerfile.runtime",
 
21
  },
22
  {
23
  "file": "EVALUATION.md",
24
+ "bytes": 11526,
25
+ "sha256": "3d5568d6a8e97556c33877145327036c2b43737064d692f0991aac13946ce8bf"
 
 
 
 
 
26
  },
27
  {
28
  "file": "LICENSE",
29
  "bytes": 11544,
30
  "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
31
  },
32
+ {
33
+ "file": "NORMALIZATION_RUNTIME.md",
34
+ "bytes": 1441,
35
+ "sha256": "9938bbe9ffdb890490c1db08903659edde6d7c5a5b9ce0487949e7a4dc2d38f6"
36
+ },
37
  {
38
  "file": "QUESTION-SCALING.md",
39
+ "bytes": 3763,
40
+ "sha256": "7dde01b31f9292116e715b501f78bf7577061f979ece9e95f7e3739ff2ead141"
41
  },
42
  {
43
  "file": "QWEN-LICENSE",
44
  "bytes": 11544,
45
  "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a"
46
  },
47
+ {
48
+ "file": "READING-SUPPLEMENT.md",
49
+ "bytes": 4359,
50
+ "sha256": "674a9efdbc28387f4526e1794890a0377db268b8fa89915601d3ce11a46d272a"
51
+ },
52
  {
53
  "file": "README.md",
54
+ "bytes": 5434,
55
+ "sha256": "05cacb9febff3519c7ff3daad2ac2d872a06067019214c2d3cf8b1e466e26401"
56
  },
57
  {
58
  "file": "RUNTIME.md",
59
+ "bytes": 6018,
60
+ "sha256": "4eb7f04296218bb859f223f27d000bdd4415bfbc0326a4e98358a66e67c73273"
61
+ },
62
+ {
63
+ "file": "RUNTIME_BINDING.json",
64
+ "bytes": 1764,
65
+ "sha256": "c050b5a43d9ddf98e07924878a4e72cfb1c4746131a3045c31557c262791f922"
66
  },
67
  {
68
+ "file": "SOURCE_BUNDLE_MANIFEST.json",
69
+ "bytes": 4206,
70
+ "sha256": "9242b7fd4ce226f638108810ec5156dca6f270cce5a16efcf5ef091180ad75be"
71
  },
72
  {
73
  "file": "USAGE.md",
74
+ "bytes": 5827,
75
+ "sha256": "819bf0195a511d149a15e415261307b75e0b9052e1aa828ada9b9b5d3ca9fe8c"
76
  },
77
  {
78
  "file": "assets/architecture-atlas.pdf",
 
96
  },
97
  {
98
  "file": "assets/decision-capabilities.pdf",
99
+ "bytes": 51635,
100
+ "sha256": "41f1b6e93b41c7f2057ad9c0743ddf0fc643e0c48ba0de58d8a3b58dbc897f1e"
101
  },
102
  {
103
  "file": "assets/decision-capabilities.png",
104
+ "bytes": 445740,
105
+ "sha256": "81a551e4b353bf2934d722f6f6038972bb8b4cd68ffd22ed546506a8303daebe"
106
  },
107
  {
108
  "file": "assets/decision-capabilities.svg",
109
+ "bytes": 114828,
110
+ "sha256": "5d7c39e232094f74fbbb10915ec9dc9c5a030159a2e7497dee4180b46251d643"
111
  },
112
  {
113
  "file": "assets/decision-family-header.png",
 
121
  },
122
  {
123
  "file": "assets/decision-quality.pdf",
124
+ "bytes": 37119,
125
+ "sha256": "c85f236e2d74ae5eb87018277bdac22ec8b4132c5fb10d1ea20231c593c33b41"
126
  },
127
  {
128
  "file": "assets/decision-quality.png",
129
+ "bytes": 170682,
130
+ "sha256": "2176d6ffb40424f9121989ec5a63e51789f39e79d086ba0a27735aa37938df01"
131
  },
132
  {
133
  "file": "assets/decision-quality.svg",
134
+ "bytes": 36923,
135
+ "sha256": "367cb8096c8441fce35259973198a4eb6fa79424531884b066cfc502634c3cec"
136
  },
137
  {
138
  "file": "assets/decision-question-scaling.pdf",
 
172
  {
173
  "file": "backbone/model.safetensors",
174
  "bytes": 3763685328,
175
+ "sha256": "168f7f90bf410130da92fd5270305286d53d5cbb68be02f1c611c40cb9243fe2"
176
  },
177
  {
178
  "file": "bundle-manifest.json",
179
+ "bytes": 7251,
180
+ "sha256": "009d4892300c0a98f6a9b61706cdb6cc7ce2f6883c25c120477b98fe7fe82d0f"
181
  },
182
  {
183
  "file": "chat_template.jinja",
 
186
  },
187
  {
188
  "file": "code/decision_api.py",
189
+ "bytes": 8251,
190
+ "sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40"
191
  },
192
  {
193
  "file": "code/decision_model.py",
 
199
  "bytes": 3164,
200
  "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
201
  },
202
+ {
203
+ "file": "code/profile_guard.py",
204
+ "bytes": 6131,
205
+ "sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452"
206
+ },
207
+ {
208
+ "file": "code/runtime_profile.py",
209
+ "bytes": 2857,
210
+ "sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda"
211
+ },
212
  {
213
  "file": "decision_config.json",
214
  "bytes": 753,
 
217
  {
218
  "file": "decision_head.safetensors",
219
  "bytes": 8424272,
220
+ "sha256": "924a43288d48804c22e7e2573898d694303879ab9bb819d1b20eadc2ab32f00d"
 
 
 
 
 
221
  },
222
  {
223
+ "file": "metrics/natural-confirmation.json",
224
+ "bytes": 27933,
225
+ "sha256": "64863075f92fd7e9df489fa37b8ddaedf29e2ee53da3da2a6009d1ee5625268e"
226
  },
227
  {
228
  "file": "metrics/question-scaling.json",
 
230
  "sha256": "4708635c6f26cd0209dafb4e45af6921ed22091e1996bbeb5861c98214ff5886"
231
  },
232
  {
233
+ "file": "model-card-example.json",
234
+ "bytes": 3774,
235
+ "sha256": "2e4b902caef5ffbc935f20f2812aa1799d83e5c4b4561cf79746cbbf847faa96"
236
  },
237
  {
238
+ "file": "peer-projection-provenance.json",
239
+ "bytes": 2281,
240
+ "sha256": "0dfae5adc5d3c028d6fd0f59ac8b3660b55f2e8b10787b5d840b1f5437d4f301"
241
  },
242
  {
243
  "file": "pyproject.toml",
244
  "bytes": 436,
245
+ "sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946"
246
+ },
247
+ {
248
+ "file": "quality-metrics.json",
249
+ "bytes": 1379273,
250
+ "sha256": "f2e1e71b391ccbac4ceeacbbd6a2cf128ace27594203a548d71c11aa8952f9f8"
251
  },
252
  {
253
  "file": "runtime-build-provenance.json",
254
+ "bytes": 3033,
255
+ "sha256": "8a22aed74b1210c44c918c1dc3f8ac9778b55ebcc70a9a805eb2f2de76a24870"
256
  },
257
  {
258
  "file": "runtime-fla-requirements.lock",
259
  "bytes": 654,
260
  "sha256": "35e1fedff9ca49092a7ada277bbd794abe7e8474e14a678a7e4bcec6406d0eb5"
261
  },
262
+ {
263
+ "file": "runtime-profile/l2norm_fwd_kernel.json",
264
+ "bytes": 13210,
265
+ "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc"
266
+ },
267
+ {
268
+ "file": "runtime-profile/profile.json",
269
+ "bytes": 18101,
270
+ "sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f"
271
+ },
272
  {
273
  "file": "runtime-provenance.json",
274
  "bytes": 5805,
 
276
  },
277
  {
278
  "file": "runtime.json",
279
+ "bytes": 1133,
280
+ "sha256": "9fe0a2bd626a8b974c85e4925cdc79dca320c9116c6d0196c12a87b9aeae8d2a"
281
  },
282
  {
283
  "file": "src/decision/__init__.py",
 
286
  },
287
  {
288
  "file": "src/decision/example.py",
289
+ "bytes": 3581,
290
+ "sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6"
291
  },
292
  {
293
  "file": "src/decision/model.py",
294
+ "bytes": 9415,
295
+ "sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0"
296
+ },
297
+ {
298
+ "file": "src/decision_local.egg-info/PKG-INFO",
299
+ "bytes": 225,
300
+ "sha256": "cc5266721b2e02c5c963b59856d979d619f7d9d687b60bda037c7bad2f22cd1d"
301
+ },
302
+ {
303
+ "file": "src/decision_local.egg-info/SOURCES.txt",
304
+ "bytes": 361,
305
+ "sha256": "741f317b6e9b27453e8f2fbeb76c0d1da326e89de7978fe87ce906ed1e2a68a9"
306
+ },
307
+ {
308
+ "file": "src/decision_local.egg-info/dependency_links.txt",
309
+ "bytes": 1,
310
+ "sha256": "01ba4719c80b6fe911b091a7c05124b64eeece964e09c058ef8f9805daca546b"
311
+ },
312
+ {
313
+ "file": "src/decision_local.egg-info/entry_points.txt",
314
+ "bytes": 59,
315
+ "sha256": "5400ff8d993d37849bc01b702e4674b8d9c4122af514101ace0bcd691eab39fb"
316
+ },
317
+ {
318
+ "file": "src/decision_local.egg-info/requires.txt",
319
+ "bytes": 31,
320
+ "sha256": "b8bf334329a333c3bba85388fbf41377b17305ee71b402c431c1af7445eadc08"
321
+ },
322
+ {
323
+ "file": "src/decision_local.egg-info/top_level.txt",
324
+ "bytes": 9,
325
+ "sha256": "834608781b00d5df58e3ae55a6b15c201aeebacc26ec99a1035fd75b83e16760"
326
  },
327
  {
328
  "file": "temperature.json",
329
+ "bytes": 13733,
330
+ "sha256": "63af0fbd1c817c4bc354aab81df66626bcd9908ac7a71bcbb921e63ba7361d44"
331
  },
332
  {
333
  "file": "tokenizer.json",
 
344
  "ATTRIBUTIONS.md": "copy",
345
  "Dockerfile.runtime": "copy",
346
  "EVALUATION.md": "copy",
 
347
  "LICENSE": "copy",
348
+ "NORMALIZATION_RUNTIME.md": "copy",
349
+ "QUESTION-SCALING.md": "copy",
350
  "QWEN-LICENSE": "copy",
351
+ "READING-SUPPLEMENT.md": "copy",
352
+ "README.md": "copy",
353
  "RUNTIME.md": "copy",
354
+ "RUNTIME_BINDING.json": "copy",
355
+ "SOURCE_BUNDLE_MANIFEST.json": "copy",
356
  "USAGE.md": "copy",
357
  "assets/architecture-atlas.pdf": "copy",
358
  "assets/architecture.pdf": "copy",
359
  "assets/architecture.png": "copy",
360
  "assets/architecture.svg": "copy",
361
+ "assets/decision-capabilities.pdf": "copy",
362
+ "assets/decision-capabilities.png": "copy",
363
+ "assets/decision-capabilities.svg": "copy",
364
  "assets/decision-family-header.png": "copy",
365
  "assets/decision-mark.png": "copy",
366
+ "assets/decision-quality.pdf": "copy",
367
+ "assets/decision-quality.png": "copy",
368
+ "assets/decision-quality.svg": "copy",
369
+ "assets/decision-question-scaling.pdf": "copy",
370
+ "assets/decision-question-scaling.png": "copy",
371
+ "assets/decision-question-scaling.svg": "copy",
372
  "assets/readout.pdf": "copy",
373
  "assets/readout.png": "copy",
374
  "assets/readout.svg": "copy",
 
379
  "code/decision_api.py": "copy",
380
  "code/decision_model.py": "copy",
381
  "code/predict.py": "copy",
382
+ "code/profile_guard.py": "copy",
383
+ "code/runtime_profile.py": "copy",
384
  "decision_config.json": "copy",
385
  "decision_head.safetensors": "hardlink",
386
+ "metrics/natural-confirmation.json": "copy",
387
+ "metrics/question-scaling.json": "copy",
 
388
  "model-card-example.json": "copy",
389
+ "peer-projection-provenance.json": "copy",
390
  "pyproject.toml": "copy",
391
+ "quality-metrics.json": "copy",
392
  "runtime-build-provenance.json": "copy",
393
  "runtime-fla-requirements.lock": "copy",
394
+ "runtime-profile/l2norm_fwd_kernel.json": "copy",
395
+ "runtime-profile/profile.json": "copy",
396
  "runtime-provenance.json": "copy",
397
  "runtime.json": "copy",
398
  "src/decision/__init__.py": "copy",
399
  "src/decision/example.py": "copy",
400
  "src/decision/model.py": "copy",
401
+ "src/decision_local.egg-info/PKG-INFO": "copy",
402
+ "src/decision_local.egg-info/SOURCES.txt": "copy",
403
+ "src/decision_local.egg-info/dependency_links.txt": "copy",
404
+ "src/decision_local.egg-info/entry_points.txt": "copy",
405
+ "src/decision_local.egg-info/requires.txt": "copy",
406
+ "src/decision_local.egg-info/top_level.txt": "copy",
407
  "temperature.json": "copy",
408
  "tokenizer.json": "copy",
409
+ "tokenizer_config.json": "copy"
 
 
 
 
 
410
  },
411
+ "scope": "Inference weights, original numerical code, calibrated runtime metadata, wrapper, model card, license, attribution and approved assets only."
 
 
 
412
  }
runtime-build-provenance.json CHANGED
@@ -25,7 +25,7 @@
25
  "build_exit_code": 0,
26
  "gpu_validation": {
27
  "status": "passed",
28
- "scope": "Real packaged three-question example and explicit complete-input overflow rejection for each bundle. Exact response equality against qualified runtime on this request; not a replacement for full benchmark rerun.",
29
  "models": {
30
  "Decision-1.0-Sol": {
31
  "bundle_manifest_sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c",
@@ -51,7 +51,8 @@
51
  "overflow_rejected": true,
52
  "board_product_name_verified": false
53
  }
54
- }
 
55
  },
56
  "built_utc": "2026-09-21T12:57:54.147566+00:00",
57
  "files": {
 
25
  "build_exit_code": 0,
26
  "gpu_validation": {
27
  "status": "passed",
28
+ "scope": "Real packaged three-question example and explicit complete-input overflow rejection for each bundle. Exact response equality against qualified runtime on this request; not a replacement for full benchmark rerun. These entries identify the initial v1.0 weights; they are not new measurements of a later weight update.",
29
  "models": {
30
  "Decision-1.0-Sol": {
31
  "bundle_manifest_sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c",
 
51
  "overflow_rejected": true,
52
  "board_product_name_verified": false
53
  }
54
+ },
55
+ "validation_release": "v1.0"
56
  },
57
  "built_utc": "2026-09-21T12:57:54.147566+00:00",
58
  "files": {
runtime-profile/l2norm_fwd_kernel.json ADDED
@@ -0,0 +1,646 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "kernel_name": "l2norm_fwd_kernel",
3
+ "triton_version": "3.7.1",
4
+ "autotune_entries": {
5
+ "878a9638751a7375a0fb15d3a7767659": {
6
+ "autotune_key": [
7
+ 128,
8
+ 1,
9
+ "torch.bfloat16",
10
+ "torch.bfloat16",
11
+ "torch.float32"
12
+ ],
13
+ "config": {
14
+ "kwargs": {
15
+ "BT": 32
16
+ },
17
+ "num_warps": 8,
18
+ "num_ctas": 1,
19
+ "num_stages": 3,
20
+ "maxnreg": null,
21
+ "pre_hook": null,
22
+ "ir_override": null
23
+ }
24
+ },
25
+ "eca743e0790be43b6867d648e9430f67": {
26
+ "autotune_key": [
27
+ 128,
28
+ 2,
29
+ "torch.bfloat16",
30
+ "torch.bfloat16",
31
+ "torch.float32"
32
+ ],
33
+ "config": {
34
+ "kwargs": {
35
+ "BT": 16
36
+ },
37
+ "num_warps": 1,
38
+ "num_ctas": 1,
39
+ "num_stages": 3,
40
+ "maxnreg": null,
41
+ "pre_hook": null,
42
+ "ir_override": null
43
+ }
44
+ },
45
+ "7e2d02ce45e4f072828b4c73e16fe0af": {
46
+ "autotune_key": [
47
+ 128,
48
+ 3,
49
+ "torch.bfloat16",
50
+ "torch.bfloat16",
51
+ "torch.float32"
52
+ ],
53
+ "config": {
54
+ "kwargs": {
55
+ "BT": 16
56
+ },
57
+ "num_warps": 2,
58
+ "num_ctas": 1,
59
+ "num_stages": 3,
60
+ "maxnreg": null,
61
+ "pre_hook": null,
62
+ "ir_override": null
63
+ }
64
+ },
65
+ "424d41ef00a3161badd7f960c34e6ece": {
66
+ "autotune_key": [
67
+ 128,
68
+ 4,
69
+ "torch.bfloat16",
70
+ "torch.bfloat16",
71
+ "torch.float32"
72
+ ],
73
+ "config": {
74
+ "kwargs": {
75
+ "BT": 32
76
+ },
77
+ "num_warps": 16,
78
+ "num_ctas": 1,
79
+ "num_stages": 3,
80
+ "maxnreg": null,
81
+ "pre_hook": null,
82
+ "ir_override": null
83
+ }
84
+ },
85
+ "6bc2b57d9035425140554c225e667aad": {
86
+ "autotune_key": [
87
+ 128,
88
+ 5,
89
+ "torch.bfloat16",
90
+ "torch.bfloat16",
91
+ "torch.float32"
92
+ ],
93
+ "config": {
94
+ "kwargs": {
95
+ "BT": 32
96
+ },
97
+ "num_warps": 16,
98
+ "num_ctas": 1,
99
+ "num_stages": 3,
100
+ "maxnreg": null,
101
+ "pre_hook": null,
102
+ "ir_override": null
103
+ }
104
+ },
105
+ "91b9289c43a2d1b21042009d44c57b8a": {
106
+ "autotune_key": [
107
+ 128,
108
+ 6,
109
+ "torch.bfloat16",
110
+ "torch.bfloat16",
111
+ "torch.float32"
112
+ ],
113
+ "config": {
114
+ "kwargs": {
115
+ "BT": 32
116
+ },
117
+ "num_warps": 16,
118
+ "num_ctas": 1,
119
+ "num_stages": 3,
120
+ "maxnreg": null,
121
+ "pre_hook": null,
122
+ "ir_override": null
123
+ }
124
+ },
125
+ "f44702dfc86a96df1de10439f5c5575a": {
126
+ "autotune_key": [
127
+ 128,
128
+ 7,
129
+ "torch.bfloat16",
130
+ "torch.bfloat16",
131
+ "torch.float32"
132
+ ],
133
+ "config": {
134
+ "kwargs": {
135
+ "BT": 32
136
+ },
137
+ "num_warps": 16,
138
+ "num_ctas": 1,
139
+ "num_stages": 3,
140
+ "maxnreg": null,
141
+ "pre_hook": null,
142
+ "ir_override": null
143
+ }
144
+ },
145
+ "87c993ac4858bff6042f319bec1647f7": {
146
+ "autotune_key": [
147
+ 128,
148
+ 8,
149
+ "torch.bfloat16",
150
+ "torch.bfloat16",
151
+ "torch.float32"
152
+ ],
153
+ "config": {
154
+ "kwargs": {
155
+ "BT": 32
156
+ },
157
+ "num_warps": 16,
158
+ "num_ctas": 1,
159
+ "num_stages": 3,
160
+ "maxnreg": null,
161
+ "pre_hook": null,
162
+ "ir_override": null
163
+ }
164
+ },
165
+ "b35c6bfa2cd85845ba58306ee63e28b1": {
166
+ "autotune_key": [
167
+ 128,
168
+ 9,
169
+ "torch.bfloat16",
170
+ "torch.bfloat16",
171
+ "torch.float32"
172
+ ],
173
+ "config": {
174
+ "kwargs": {
175
+ "BT": 8
176
+ },
177
+ "num_warps": 4,
178
+ "num_ctas": 1,
179
+ "num_stages": 3,
180
+ "maxnreg": null,
181
+ "pre_hook": null,
182
+ "ir_override": null
183
+ }
184
+ },
185
+ "3527d130e8f6f1415c6d24df9821e638": {
186
+ "autotune_key": [
187
+ 128,
188
+ 10,
189
+ "torch.bfloat16",
190
+ "torch.bfloat16",
191
+ "torch.float32"
192
+ ],
193
+ "config": {
194
+ "kwargs": {
195
+ "BT": 8
196
+ },
197
+ "num_warps": 2,
198
+ "num_ctas": 1,
199
+ "num_stages": 3,
200
+ "maxnreg": null,
201
+ "pre_hook": null,
202
+ "ir_override": null
203
+ }
204
+ },
205
+ "fe2e43a7c21c6a4814d2636f3004d7dc": {
206
+ "autotune_key": [
207
+ 128,
208
+ 11,
209
+ "torch.bfloat16",
210
+ "torch.bfloat16",
211
+ "torch.float32"
212
+ ],
213
+ "config": {
214
+ "kwargs": {
215
+ "BT": 8
216
+ },
217
+ "num_warps": 4,
218
+ "num_ctas": 1,
219
+ "num_stages": 3,
220
+ "maxnreg": null,
221
+ "pre_hook": null,
222
+ "ir_override": null
223
+ }
224
+ },
225
+ "c39c159634377c459eccde8a272f74fe": {
226
+ "autotune_key": [
227
+ 128,
228
+ 12,
229
+ "torch.bfloat16",
230
+ "torch.bfloat16",
231
+ "torch.float32"
232
+ ],
233
+ "config": {
234
+ "kwargs": {
235
+ "BT": 32
236
+ },
237
+ "num_warps": 8,
238
+ "num_ctas": 1,
239
+ "num_stages": 3,
240
+ "maxnreg": null,
241
+ "pre_hook": null,
242
+ "ir_override": null
243
+ }
244
+ },
245
+ "a10453829fb545fd2e01acfb4a809d17": {
246
+ "autotune_key": [
247
+ 128,
248
+ 13,
249
+ "torch.bfloat16",
250
+ "torch.bfloat16",
251
+ "torch.float32"
252
+ ],
253
+ "config": {
254
+ "kwargs": {
255
+ "BT": 8
256
+ },
257
+ "num_warps": 4,
258
+ "num_ctas": 1,
259
+ "num_stages": 3,
260
+ "maxnreg": null,
261
+ "pre_hook": null,
262
+ "ir_override": null
263
+ }
264
+ },
265
+ "bd8734e29b9704051fe467ede8e8cc03": {
266
+ "autotune_key": [
267
+ 128,
268
+ 14,
269
+ "torch.bfloat16",
270
+ "torch.bfloat16",
271
+ "torch.float32"
272
+ ],
273
+ "config": {
274
+ "kwargs": {
275
+ "BT": 32
276
+ },
277
+ "num_warps": 8,
278
+ "num_ctas": 1,
279
+ "num_stages": 3,
280
+ "maxnreg": null,
281
+ "pre_hook": null,
282
+ "ir_override": null
283
+ }
284
+ },
285
+ "d5ffd2f70a70e13ac03901bd22a150a7": {
286
+ "autotune_key": [
287
+ 128,
288
+ 15,
289
+ "torch.bfloat16",
290
+ "torch.bfloat16",
291
+ "torch.float32"
292
+ ],
293
+ "config": {
294
+ "kwargs": {
295
+ "BT": 32
296
+ },
297
+ "num_warps": 8,
298
+ "num_ctas": 1,
299
+ "num_stages": 3,
300
+ "maxnreg": null,
301
+ "pre_hook": null,
302
+ "ir_override": null
303
+ }
304
+ },
305
+ "d522f744092358110c9ad2cc5f54b501": {
306
+ "autotune_key": [
307
+ 128,
308
+ 16,
309
+ "torch.bfloat16",
310
+ "torch.bfloat16",
311
+ "torch.float32"
312
+ ],
313
+ "config": {
314
+ "kwargs": {
315
+ "BT": 32
316
+ },
317
+ "num_warps": 8,
318
+ "num_ctas": 1,
319
+ "num_stages": 3,
320
+ "maxnreg": null,
321
+ "pre_hook": null,
322
+ "ir_override": null
323
+ }
324
+ },
325
+ "6003bee20af205e16410f1983eadd67c": {
326
+ "autotune_key": [
327
+ 128,
328
+ 17,
329
+ "torch.bfloat16",
330
+ "torch.bfloat16",
331
+ "torch.float32"
332
+ ],
333
+ "config": {
334
+ "kwargs": {
335
+ "BT": 32
336
+ },
337
+ "num_warps": 8,
338
+ "num_ctas": 1,
339
+ "num_stages": 3,
340
+ "maxnreg": null,
341
+ "pre_hook": null,
342
+ "ir_override": null
343
+ }
344
+ },
345
+ "89d2cb2e8a818ce3ecb80bd340ba42b6": {
346
+ "autotune_key": [
347
+ 128,
348
+ 18,
349
+ "torch.bfloat16",
350
+ "torch.bfloat16",
351
+ "torch.float32"
352
+ ],
353
+ "config": {
354
+ "kwargs": {
355
+ "BT": 32
356
+ },
357
+ "num_warps": 8,
358
+ "num_ctas": 1,
359
+ "num_stages": 3,
360
+ "maxnreg": null,
361
+ "pre_hook": null,
362
+ "ir_override": null
363
+ }
364
+ },
365
+ "157376310735556f015b8ac2dc6e9e2e": {
366
+ "autotune_key": [
367
+ 128,
368
+ 19,
369
+ "torch.bfloat16",
370
+ "torch.bfloat16",
371
+ "torch.float32"
372
+ ],
373
+ "config": {
374
+ "kwargs": {
375
+ "BT": 32
376
+ },
377
+ "num_warps": 8,
378
+ "num_ctas": 1,
379
+ "num_stages": 3,
380
+ "maxnreg": null,
381
+ "pre_hook": null,
382
+ "ir_override": null
383
+ }
384
+ },
385
+ "e19965e082b861f637bf4bab6abe025b": {
386
+ "autotune_key": [
387
+ 128,
388
+ 20,
389
+ "torch.bfloat16",
390
+ "torch.bfloat16",
391
+ "torch.float32"
392
+ ],
393
+ "config": {
394
+ "kwargs": {
395
+ "BT": 32
396
+ },
397
+ "num_warps": 8,
398
+ "num_ctas": 1,
399
+ "num_stages": 3,
400
+ "maxnreg": null,
401
+ "pre_hook": null,
402
+ "ir_override": null
403
+ }
404
+ },
405
+ "49feda728414835e54048468abd808ab": {
406
+ "autotune_key": [
407
+ 128,
408
+ 21,
409
+ "torch.bfloat16",
410
+ "torch.bfloat16",
411
+ "torch.float32"
412
+ ],
413
+ "config": {
414
+ "kwargs": {
415
+ "BT": 32
416
+ },
417
+ "num_warps": 8,
418
+ "num_ctas": 1,
419
+ "num_stages": 3,
420
+ "maxnreg": null,
421
+ "pre_hook": null,
422
+ "ir_override": null
423
+ }
424
+ },
425
+ "6b419788330d3468ef22dc741ec9d7f7": {
426
+ "autotune_key": [
427
+ 128,
428
+ 22,
429
+ "torch.bfloat16",
430
+ "torch.bfloat16",
431
+ "torch.float32"
432
+ ],
433
+ "config": {
434
+ "kwargs": {
435
+ "BT": 32
436
+ },
437
+ "num_warps": 8,
438
+ "num_ctas": 1,
439
+ "num_stages": 3,
440
+ "maxnreg": null,
441
+ "pre_hook": null,
442
+ "ir_override": null
443
+ }
444
+ },
445
+ "8827081d0dda52c9376b3a1d97a82a8e": {
446
+ "autotune_key": [
447
+ 128,
448
+ 23,
449
+ "torch.bfloat16",
450
+ "torch.bfloat16",
451
+ "torch.float32"
452
+ ],
453
+ "config": {
454
+ "kwargs": {
455
+ "BT": 32
456
+ },
457
+ "num_warps": 8,
458
+ "num_ctas": 1,
459
+ "num_stages": 3,
460
+ "maxnreg": null,
461
+ "pre_hook": null,
462
+ "ir_override": null
463
+ }
464
+ },
465
+ "49e16fa1a715a8913c54c4ff9ce61aa9": {
466
+ "autotune_key": [
467
+ 128,
468
+ 24,
469
+ "torch.bfloat16",
470
+ "torch.bfloat16",
471
+ "torch.float32"
472
+ ],
473
+ "config": {
474
+ "kwargs": {
475
+ "BT": 32
476
+ },
477
+ "num_warps": 8,
478
+ "num_ctas": 1,
479
+ "num_stages": 3,
480
+ "maxnreg": null,
481
+ "pre_hook": null,
482
+ "ir_override": null
483
+ }
484
+ },
485
+ "7bf76d12f90b95bbddff57fbc8fe3c39": {
486
+ "autotune_key": [
487
+ 128,
488
+ 25,
489
+ "torch.bfloat16",
490
+ "torch.bfloat16",
491
+ "torch.float32"
492
+ ],
493
+ "config": {
494
+ "kwargs": {
495
+ "BT": 32
496
+ },
497
+ "num_warps": 8,
498
+ "num_ctas": 1,
499
+ "num_stages": 3,
500
+ "maxnreg": null,
501
+ "pre_hook": null,
502
+ "ir_override": null
503
+ }
504
+ },
505
+ "3f2cd7a3a429540e08f0e4dd8ac9afc5": {
506
+ "autotune_key": [
507
+ 128,
508
+ 26,
509
+ "torch.bfloat16",
510
+ "torch.bfloat16",
511
+ "torch.float32"
512
+ ],
513
+ "config": {
514
+ "kwargs": {
515
+ "BT": 32
516
+ },
517
+ "num_warps": 8,
518
+ "num_ctas": 1,
519
+ "num_stages": 3,
520
+ "maxnreg": null,
521
+ "pre_hook": null,
522
+ "ir_override": null
523
+ }
524
+ },
525
+ "837afa436bfab6256a535e9360403f34": {
526
+ "autotune_key": [
527
+ 128,
528
+ 27,
529
+ "torch.bfloat16",
530
+ "torch.bfloat16",
531
+ "torch.float32"
532
+ ],
533
+ "config": {
534
+ "kwargs": {
535
+ "BT": 32
536
+ },
537
+ "num_warps": 8,
538
+ "num_ctas": 1,
539
+ "num_stages": 3,
540
+ "maxnreg": null,
541
+ "pre_hook": null,
542
+ "ir_override": null
543
+ }
544
+ },
545
+ "d4e411f29afbb396dcfe1455427e5178": {
546
+ "autotune_key": [
547
+ 128,
548
+ 28,
549
+ "torch.bfloat16",
550
+ "torch.bfloat16",
551
+ "torch.float32"
552
+ ],
553
+ "config": {
554
+ "kwargs": {
555
+ "BT": 32
556
+ },
557
+ "num_warps": 8,
558
+ "num_ctas": 1,
559
+ "num_stages": 3,
560
+ "maxnreg": null,
561
+ "pre_hook": null,
562
+ "ir_override": null
563
+ }
564
+ },
565
+ "1b96f1802f4f63f5c872de3d1972ad8a": {
566
+ "autotune_key": [
567
+ 128,
568
+ 29,
569
+ "torch.bfloat16",
570
+ "torch.bfloat16",
571
+ "torch.float32"
572
+ ],
573
+ "config": {
574
+ "kwargs": {
575
+ "BT": 32
576
+ },
577
+ "num_warps": 8,
578
+ "num_ctas": 1,
579
+ "num_stages": 3,
580
+ "maxnreg": null,
581
+ "pre_hook": null,
582
+ "ir_override": null
583
+ }
584
+ },
585
+ "aec76b66508d5e884ffcc743e8f55a92": {
586
+ "autotune_key": [
587
+ 128,
588
+ 30,
589
+ "torch.bfloat16",
590
+ "torch.bfloat16",
591
+ "torch.float32"
592
+ ],
593
+ "config": {
594
+ "kwargs": {
595
+ "BT": 32
596
+ },
597
+ "num_warps": 8,
598
+ "num_ctas": 1,
599
+ "num_stages": 3,
600
+ "maxnreg": null,
601
+ "pre_hook": null,
602
+ "ir_override": null
603
+ }
604
+ },
605
+ "439d493be8fe1159bd65368245a4aa60": {
606
+ "autotune_key": [
607
+ 128,
608
+ 31,
609
+ "torch.bfloat16",
610
+ "torch.bfloat16",
611
+ "torch.float32"
612
+ ],
613
+ "config": {
614
+ "kwargs": {
615
+ "BT": 32
616
+ },
617
+ "num_warps": 8,
618
+ "num_ctas": 1,
619
+ "num_stages": 3,
620
+ "maxnreg": null,
621
+ "pre_hook": null,
622
+ "ir_override": null
623
+ }
624
+ },
625
+ "801422dba59f6c5e144328aabdd135a3": {
626
+ "autotune_key": [
627
+ 128,
628
+ 32,
629
+ "torch.bfloat16",
630
+ "torch.bfloat16",
631
+ "torch.float32"
632
+ ],
633
+ "config": {
634
+ "kwargs": {
635
+ "BT": 32
636
+ },
637
+ "num_warps": 8,
638
+ "num_ctas": 1,
639
+ "num_stages": 3,
640
+ "maxnreg": null,
641
+ "pre_hook": null,
642
+ "ir_override": null
643
+ }
644
+ }
645
+ }
646
+ }
runtime-profile/profile.json ADDED
@@ -0,0 +1,822 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "decision-fla-l2norm-profile-v1",
3
+ "status": "diagnostic-only-awaiting-clean-room-proof",
4
+ "target": "Decision Sol Qwen3.5-2B pointer Stage4; one profile per process",
5
+ "model_changes": false,
6
+ "numerical_tolerance_unchanged": 1e-06,
7
+ "cache_mode": "strict",
8
+ "supported": {
9
+ "head_dimension": 128,
10
+ "key_heads": 16,
11
+ "body_dtype": "bfloat16",
12
+ "output_dtype": "bfloat16",
13
+ "rstd_dtype": "float32",
14
+ "batch_size_min": 1,
15
+ "batch_size_max": 8,
16
+ "padded_tokens_max": 16384,
17
+ "NB_min": 1,
18
+ "NB_max": 32,
19
+ "NB_formula": "ceil(B*padded_tokens*16/65536)",
20
+ "unknown_key": "raise before kernel/autotune"
21
+ },
22
+ "runtime": {
23
+ "torch": "2.12.0+git6bbd260",
24
+ "hip": "7.2.53211",
25
+ "triton": "3.7.1",
26
+ "fla": "0.5.2",
27
+ "gpu_arch": "gfx942"
28
+ },
29
+ "files": [
30
+ {
31
+ "file": "l2norm_fwd_kernel.json",
32
+ "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc"
33
+ }
34
+ ],
35
+ "fla_source_sha256": {
36
+ "modules/l2norm.py": "30f7feebdbfa87b8e90c143ec62a7fa0781bfd736c856c0ec69452f0c778be3f",
37
+ "ops/utils/cache.py": "a64b09ffd3f51aed547ad72559dee94cb9aeafd2a5f02c6c8e60a6e020147a9c"
38
+ },
39
+ "source_evidence": [
40
+ {
41
+ "key": [
42
+ 128,
43
+ 1,
44
+ "torch.bfloat16",
45
+ "torch.bfloat16",
46
+ "torch.float32"
47
+ ],
48
+ "source_autotune_sha256": "030be4b902759f8a02c66edfe9724c8a04d320c32eb8298ef42da42f8ff02a54",
49
+ "chosen_config": {
50
+ "kwargs": {
51
+ "BT": 32
52
+ },
53
+ "num_warps": 8,
54
+ "num_ctas": 1,
55
+ "num_stages": 3,
56
+ "maxnreg": null,
57
+ "pre_hook": null,
58
+ "ir_override": null
59
+ }
60
+ },
61
+ {
62
+ "key": [
63
+ 128,
64
+ 7,
65
+ "torch.bfloat16",
66
+ "torch.bfloat16",
67
+ "torch.float32"
68
+ ],
69
+ "source_autotune_sha256": "b868fb0bcb06cfdc5b4432392e35ade0daeeb40374645dbfd65476996362dc48",
70
+ "chosen_config": {
71
+ "kwargs": {
72
+ "BT": 32
73
+ },
74
+ "num_warps": 16,
75
+ "num_ctas": 1,
76
+ "num_stages": 3,
77
+ "maxnreg": null,
78
+ "pre_hook": null,
79
+ "ir_override": null
80
+ }
81
+ },
82
+ {
83
+ "key": [
84
+ 128,
85
+ 8,
86
+ "torch.bfloat16",
87
+ "torch.bfloat16",
88
+ "torch.float32"
89
+ ],
90
+ "source_autotune_sha256": "ee3b7b3a9facff1f50feacbc32427f005e6d344b1e452e9daabc68b19a9fed0d",
91
+ "chosen_config": {
92
+ "kwargs": {
93
+ "BT": 32
94
+ },
95
+ "num_warps": 16,
96
+ "num_ctas": 1,
97
+ "num_stages": 3,
98
+ "maxnreg": null,
99
+ "pre_hook": null,
100
+ "ir_override": null
101
+ }
102
+ },
103
+ {
104
+ "key": [
105
+ 128,
106
+ 9,
107
+ "torch.bfloat16",
108
+ "torch.bfloat16",
109
+ "torch.float32"
110
+ ],
111
+ "source_autotune_sha256": "fd0b189927eb2475fd73e9a71899a6655ae4d73a010c618b00316f9eb15e429e",
112
+ "chosen_config": {
113
+ "kwargs": {
114
+ "BT": 8
115
+ },
116
+ "num_warps": 4,
117
+ "num_ctas": 1,
118
+ "num_stages": 3,
119
+ "maxnreg": null,
120
+ "pre_hook": null,
121
+ "ir_override": null
122
+ }
123
+ },
124
+ {
125
+ "key": [
126
+ 128,
127
+ 5,
128
+ "torch.bfloat16",
129
+ "torch.bfloat16",
130
+ "torch.float32"
131
+ ],
132
+ "source_autotune_sha256": "a8ad1053d222fb9d5751a6ebe11fc58679a9e44366be3dc9a2c214042485778c",
133
+ "chosen_config": {
134
+ "kwargs": {
135
+ "BT": 32
136
+ },
137
+ "num_warps": 16,
138
+ "num_ctas": 1,
139
+ "num_stages": 3,
140
+ "maxnreg": null,
141
+ "pre_hook": null,
142
+ "ir_override": null
143
+ }
144
+ },
145
+ {
146
+ "key": [
147
+ 128,
148
+ 2,
149
+ "torch.bfloat16",
150
+ "torch.bfloat16",
151
+ "torch.float32"
152
+ ],
153
+ "source_autotune_sha256": "00499403ea55da8fc3e1110dfb41a94204f19813ac8302dc2cd0894b2e1f51a9",
154
+ "chosen_config": {
155
+ "kwargs": {
156
+ "BT": 16
157
+ },
158
+ "num_warps": 1,
159
+ "num_ctas": 1,
160
+ "num_stages": 3,
161
+ "maxnreg": null,
162
+ "pre_hook": null,
163
+ "ir_override": null
164
+ }
165
+ },
166
+ {
167
+ "key": [
168
+ 128,
169
+ 12,
170
+ "torch.bfloat16",
171
+ "torch.bfloat16",
172
+ "torch.float32"
173
+ ],
174
+ "source_autotune_sha256": "bad146284201145ccafd77500eba52f4b20483596065a3da91e6b1bccb0548e3",
175
+ "chosen_config": {
176
+ "kwargs": {
177
+ "BT": 32
178
+ },
179
+ "num_warps": 8,
180
+ "num_ctas": 1,
181
+ "num_stages": 3,
182
+ "maxnreg": null,
183
+ "pre_hook": null,
184
+ "ir_override": null
185
+ }
186
+ },
187
+ {
188
+ "key": [
189
+ 128,
190
+ 10,
191
+ "torch.bfloat16",
192
+ "torch.bfloat16",
193
+ "torch.float32"
194
+ ],
195
+ "source_autotune_sha256": "cb8bd713aee627cb50fcca3ed750e7ac9976ee56f9d6d3d44ef469325049822c",
196
+ "chosen_config": {
197
+ "kwargs": {
198
+ "BT": 8
199
+ },
200
+ "num_warps": 2,
201
+ "num_ctas": 1,
202
+ "num_stages": 3,
203
+ "maxnreg": null,
204
+ "pre_hook": null,
205
+ "ir_override": null
206
+ }
207
+ },
208
+ {
209
+ "key": [
210
+ 128,
211
+ 13,
212
+ "torch.bfloat16",
213
+ "torch.bfloat16",
214
+ "torch.float32"
215
+ ],
216
+ "source_autotune_sha256": "f2c9fe94fbf5ff3f48b4ec875cbfb8016148c772fe085b69d6e42dd8053f9549",
217
+ "chosen_config": {
218
+ "kwargs": {
219
+ "BT": 8
220
+ },
221
+ "num_warps": 4,
222
+ "num_ctas": 1,
223
+ "num_stages": 3,
224
+ "maxnreg": null,
225
+ "pre_hook": null,
226
+ "ir_override": null
227
+ }
228
+ },
229
+ {
230
+ "key": [
231
+ 128,
232
+ 11,
233
+ "torch.bfloat16",
234
+ "torch.bfloat16",
235
+ "torch.float32"
236
+ ],
237
+ "source_autotune_sha256": "0d2cc4c185b6c73918b1397271e98276a6ee8aa510d345506b8fd30343f4e38d",
238
+ "chosen_config": {
239
+ "kwargs": {
240
+ "BT": 8
241
+ },
242
+ "num_warps": 4,
243
+ "num_ctas": 1,
244
+ "num_stages": 3,
245
+ "maxnreg": null,
246
+ "pre_hook": null,
247
+ "ir_override": null
248
+ }
249
+ },
250
+ {
251
+ "key": [
252
+ 128,
253
+ 3,
254
+ "torch.bfloat16",
255
+ "torch.bfloat16",
256
+ "torch.float32"
257
+ ],
258
+ "source_autotune_sha256": "a426f2ff58b25d5f0e7b23876254d3f4fc255ac596a902eeb14e47917688f508",
259
+ "chosen_config": {
260
+ "kwargs": {
261
+ "BT": 16
262
+ },
263
+ "num_warps": 2,
264
+ "num_ctas": 1,
265
+ "num_stages": 3,
266
+ "maxnreg": null,
267
+ "pre_hook": null,
268
+ "ir_override": null
269
+ }
270
+ },
271
+ {
272
+ "key": [
273
+ 128,
274
+ 4,
275
+ "torch.bfloat16",
276
+ "torch.bfloat16",
277
+ "torch.float32"
278
+ ],
279
+ "source_autotune_sha256": "615e132647ea6735a52e431485cd2fd4706dd754e7d4aec356d0a006e18d9166",
280
+ "chosen_config": {
281
+ "kwargs": {
282
+ "BT": 32
283
+ },
284
+ "num_warps": 16,
285
+ "num_ctas": 1,
286
+ "num_stages": 3,
287
+ "maxnreg": null,
288
+ "pre_hook": null,
289
+ "ir_override": null
290
+ }
291
+ },
292
+ {
293
+ "key": [
294
+ 128,
295
+ 18,
296
+ "torch.bfloat16",
297
+ "torch.bfloat16",
298
+ "torch.float32"
299
+ ],
300
+ "source_autotune_sha256": "ef64a3156f4dd428d7b9f80f9435c25eabbcfc192882c199c6fe5fc1c2316be8",
301
+ "chosen_config": {
302
+ "kwargs": {
303
+ "BT": 32
304
+ },
305
+ "num_warps": 8,
306
+ "num_ctas": 1,
307
+ "num_stages": 3,
308
+ "maxnreg": null,
309
+ "pre_hook": null,
310
+ "ir_override": null
311
+ }
312
+ },
313
+ {
314
+ "key": [
315
+ 128,
316
+ 6,
317
+ "torch.bfloat16",
318
+ "torch.bfloat16",
319
+ "torch.float32"
320
+ ],
321
+ "source_autotune_sha256": "adca9d81bb4df7c1a64dc7e525544b05c362d943c95446f4871bf92f09fb603b",
322
+ "chosen_config": {
323
+ "kwargs": {
324
+ "BT": 32
325
+ },
326
+ "num_warps": 16,
327
+ "num_ctas": 1,
328
+ "num_stages": 3,
329
+ "maxnreg": null,
330
+ "pre_hook": null,
331
+ "ir_override": null
332
+ }
333
+ }
334
+ ],
335
+ "per_NB_policy": [
336
+ {
337
+ "NB": 1,
338
+ "basis": "original-production-chosen",
339
+ "config": {
340
+ "kwargs": {
341
+ "BT": 32
342
+ },
343
+ "num_warps": 8,
344
+ "num_ctas": 1,
345
+ "num_stages": 3,
346
+ "maxnreg": null,
347
+ "pre_hook": null,
348
+ "ir_override": null
349
+ }
350
+ },
351
+ {
352
+ "NB": 2,
353
+ "basis": "original-production-chosen",
354
+ "config": {
355
+ "kwargs": {
356
+ "BT": 16
357
+ },
358
+ "num_warps": 1,
359
+ "num_ctas": 1,
360
+ "num_stages": 3,
361
+ "maxnreg": null,
362
+ "pre_hook": null,
363
+ "ir_override": null
364
+ }
365
+ },
366
+ {
367
+ "NB": 3,
368
+ "basis": "original-production-chosen",
369
+ "config": {
370
+ "kwargs": {
371
+ "BT": 16
372
+ },
373
+ "num_warps": 2,
374
+ "num_ctas": 1,
375
+ "num_stages": 3,
376
+ "maxnreg": null,
377
+ "pre_hook": null,
378
+ "ir_override": null
379
+ }
380
+ },
381
+ {
382
+ "NB": 4,
383
+ "basis": "original-production-chosen",
384
+ "config": {
385
+ "kwargs": {
386
+ "BT": 32
387
+ },
388
+ "num_warps": 16,
389
+ "num_ctas": 1,
390
+ "num_stages": 3,
391
+ "maxnreg": null,
392
+ "pre_hook": null,
393
+ "ir_override": null
394
+ }
395
+ },
396
+ {
397
+ "NB": 5,
398
+ "basis": "original-production-chosen",
399
+ "config": {
400
+ "kwargs": {
401
+ "BT": 32
402
+ },
403
+ "num_warps": 16,
404
+ "num_ctas": 1,
405
+ "num_stages": 3,
406
+ "maxnreg": null,
407
+ "pre_hook": null,
408
+ "ir_override": null
409
+ }
410
+ },
411
+ {
412
+ "NB": 6,
413
+ "basis": "original-production-chosen",
414
+ "config": {
415
+ "kwargs": {
416
+ "BT": 32
417
+ },
418
+ "num_warps": 16,
419
+ "num_ctas": 1,
420
+ "num_stages": 3,
421
+ "maxnreg": null,
422
+ "pre_hook": null,
423
+ "ir_override": null
424
+ }
425
+ },
426
+ {
427
+ "NB": 7,
428
+ "basis": "original-production-chosen",
429
+ "config": {
430
+ "kwargs": {
431
+ "BT": 32
432
+ },
433
+ "num_warps": 16,
434
+ "num_ctas": 1,
435
+ "num_stages": 3,
436
+ "maxnreg": null,
437
+ "pre_hook": null,
438
+ "ir_override": null
439
+ }
440
+ },
441
+ {
442
+ "NB": 8,
443
+ "basis": "original-production-chosen",
444
+ "config": {
445
+ "kwargs": {
446
+ "BT": 32
447
+ },
448
+ "num_warps": 16,
449
+ "num_ctas": 1,
450
+ "num_stages": 3,
451
+ "maxnreg": null,
452
+ "pre_hook": null,
453
+ "ir_override": null
454
+ }
455
+ },
456
+ {
457
+ "NB": 9,
458
+ "basis": "original-production-chosen",
459
+ "config": {
460
+ "kwargs": {
461
+ "BT": 8
462
+ },
463
+ "num_warps": 4,
464
+ "num_ctas": 1,
465
+ "num_stages": 3,
466
+ "maxnreg": null,
467
+ "pre_hook": null,
468
+ "ir_override": null
469
+ }
470
+ },
471
+ {
472
+ "NB": 10,
473
+ "basis": "original-production-chosen",
474
+ "config": {
475
+ "kwargs": {
476
+ "BT": 8
477
+ },
478
+ "num_warps": 2,
479
+ "num_ctas": 1,
480
+ "num_stages": 3,
481
+ "maxnreg": null,
482
+ "pre_hook": null,
483
+ "ir_override": null
484
+ }
485
+ },
486
+ {
487
+ "NB": 11,
488
+ "basis": "original-production-chosen",
489
+ "config": {
490
+ "kwargs": {
491
+ "BT": 8
492
+ },
493
+ "num_warps": 4,
494
+ "num_ctas": 1,
495
+ "num_stages": 3,
496
+ "maxnreg": null,
497
+ "pre_hook": null,
498
+ "ir_override": null
499
+ }
500
+ },
501
+ {
502
+ "NB": 12,
503
+ "basis": "original-production-chosen",
504
+ "config": {
505
+ "kwargs": {
506
+ "BT": 32
507
+ },
508
+ "num_warps": 8,
509
+ "num_ctas": 1,
510
+ "num_stages": 3,
511
+ "maxnreg": null,
512
+ "pre_hook": null,
513
+ "ir_override": null
514
+ }
515
+ },
516
+ {
517
+ "NB": 13,
518
+ "basis": "original-production-chosen",
519
+ "config": {
520
+ "kwargs": {
521
+ "BT": 8
522
+ },
523
+ "num_warps": 4,
524
+ "num_ctas": 1,
525
+ "num_stages": 3,
526
+ "maxnreg": null,
527
+ "pre_hook": null,
528
+ "ir_override": null
529
+ }
530
+ },
531
+ {
532
+ "NB": 14,
533
+ "basis": "prospective-fixed-original-NB18-setting",
534
+ "config": {
535
+ "kwargs": {
536
+ "BT": 32
537
+ },
538
+ "num_warps": 8,
539
+ "num_ctas": 1,
540
+ "num_stages": 3,
541
+ "maxnreg": null,
542
+ "pre_hook": null,
543
+ "ir_override": null
544
+ }
545
+ },
546
+ {
547
+ "NB": 15,
548
+ "basis": "prospective-fixed-original-NB18-setting",
549
+ "config": {
550
+ "kwargs": {
551
+ "BT": 32
552
+ },
553
+ "num_warps": 8,
554
+ "num_ctas": 1,
555
+ "num_stages": 3,
556
+ "maxnreg": null,
557
+ "pre_hook": null,
558
+ "ir_override": null
559
+ }
560
+ },
561
+ {
562
+ "NB": 16,
563
+ "basis": "prospective-fixed-original-NB18-setting",
564
+ "config": {
565
+ "kwargs": {
566
+ "BT": 32
567
+ },
568
+ "num_warps": 8,
569
+ "num_ctas": 1,
570
+ "num_stages": 3,
571
+ "maxnreg": null,
572
+ "pre_hook": null,
573
+ "ir_override": null
574
+ }
575
+ },
576
+ {
577
+ "NB": 17,
578
+ "basis": "prospective-fixed-original-NB18-setting",
579
+ "config": {
580
+ "kwargs": {
581
+ "BT": 32
582
+ },
583
+ "num_warps": 8,
584
+ "num_ctas": 1,
585
+ "num_stages": 3,
586
+ "maxnreg": null,
587
+ "pre_hook": null,
588
+ "ir_override": null
589
+ }
590
+ },
591
+ {
592
+ "NB": 18,
593
+ "basis": "original-production-chosen",
594
+ "config": {
595
+ "kwargs": {
596
+ "BT": 32
597
+ },
598
+ "num_warps": 8,
599
+ "num_ctas": 1,
600
+ "num_stages": 3,
601
+ "maxnreg": null,
602
+ "pre_hook": null,
603
+ "ir_override": null
604
+ }
605
+ },
606
+ {
607
+ "NB": 19,
608
+ "basis": "prospective-fixed-original-NB18-setting",
609
+ "config": {
610
+ "kwargs": {
611
+ "BT": 32
612
+ },
613
+ "num_warps": 8,
614
+ "num_ctas": 1,
615
+ "num_stages": 3,
616
+ "maxnreg": null,
617
+ "pre_hook": null,
618
+ "ir_override": null
619
+ }
620
+ },
621
+ {
622
+ "NB": 20,
623
+ "basis": "prospective-fixed-original-NB18-setting",
624
+ "config": {
625
+ "kwargs": {
626
+ "BT": 32
627
+ },
628
+ "num_warps": 8,
629
+ "num_ctas": 1,
630
+ "num_stages": 3,
631
+ "maxnreg": null,
632
+ "pre_hook": null,
633
+ "ir_override": null
634
+ }
635
+ },
636
+ {
637
+ "NB": 21,
638
+ "basis": "prospective-fixed-original-NB18-setting",
639
+ "config": {
640
+ "kwargs": {
641
+ "BT": 32
642
+ },
643
+ "num_warps": 8,
644
+ "num_ctas": 1,
645
+ "num_stages": 3,
646
+ "maxnreg": null,
647
+ "pre_hook": null,
648
+ "ir_override": null
649
+ }
650
+ },
651
+ {
652
+ "NB": 22,
653
+ "basis": "prospective-fixed-original-NB18-setting",
654
+ "config": {
655
+ "kwargs": {
656
+ "BT": 32
657
+ },
658
+ "num_warps": 8,
659
+ "num_ctas": 1,
660
+ "num_stages": 3,
661
+ "maxnreg": null,
662
+ "pre_hook": null,
663
+ "ir_override": null
664
+ }
665
+ },
666
+ {
667
+ "NB": 23,
668
+ "basis": "prospective-fixed-original-NB18-setting",
669
+ "config": {
670
+ "kwargs": {
671
+ "BT": 32
672
+ },
673
+ "num_warps": 8,
674
+ "num_ctas": 1,
675
+ "num_stages": 3,
676
+ "maxnreg": null,
677
+ "pre_hook": null,
678
+ "ir_override": null
679
+ }
680
+ },
681
+ {
682
+ "NB": 24,
683
+ "basis": "prospective-fixed-original-NB18-setting",
684
+ "config": {
685
+ "kwargs": {
686
+ "BT": 32
687
+ },
688
+ "num_warps": 8,
689
+ "num_ctas": 1,
690
+ "num_stages": 3,
691
+ "maxnreg": null,
692
+ "pre_hook": null,
693
+ "ir_override": null
694
+ }
695
+ },
696
+ {
697
+ "NB": 25,
698
+ "basis": "prospective-fixed-original-NB18-setting",
699
+ "config": {
700
+ "kwargs": {
701
+ "BT": 32
702
+ },
703
+ "num_warps": 8,
704
+ "num_ctas": 1,
705
+ "num_stages": 3,
706
+ "maxnreg": null,
707
+ "pre_hook": null,
708
+ "ir_override": null
709
+ }
710
+ },
711
+ {
712
+ "NB": 26,
713
+ "basis": "prospective-fixed-original-NB18-setting",
714
+ "config": {
715
+ "kwargs": {
716
+ "BT": 32
717
+ },
718
+ "num_warps": 8,
719
+ "num_ctas": 1,
720
+ "num_stages": 3,
721
+ "maxnreg": null,
722
+ "pre_hook": null,
723
+ "ir_override": null
724
+ }
725
+ },
726
+ {
727
+ "NB": 27,
728
+ "basis": "prospective-fixed-original-NB18-setting",
729
+ "config": {
730
+ "kwargs": {
731
+ "BT": 32
732
+ },
733
+ "num_warps": 8,
734
+ "num_ctas": 1,
735
+ "num_stages": 3,
736
+ "maxnreg": null,
737
+ "pre_hook": null,
738
+ "ir_override": null
739
+ }
740
+ },
741
+ {
742
+ "NB": 28,
743
+ "basis": "prospective-fixed-original-NB18-setting",
744
+ "config": {
745
+ "kwargs": {
746
+ "BT": 32
747
+ },
748
+ "num_warps": 8,
749
+ "num_ctas": 1,
750
+ "num_stages": 3,
751
+ "maxnreg": null,
752
+ "pre_hook": null,
753
+ "ir_override": null
754
+ }
755
+ },
756
+ {
757
+ "NB": 29,
758
+ "basis": "prospective-fixed-original-NB18-setting",
759
+ "config": {
760
+ "kwargs": {
761
+ "BT": 32
762
+ },
763
+ "num_warps": 8,
764
+ "num_ctas": 1,
765
+ "num_stages": 3,
766
+ "maxnreg": null,
767
+ "pre_hook": null,
768
+ "ir_override": null
769
+ }
770
+ },
771
+ {
772
+ "NB": 30,
773
+ "basis": "prospective-fixed-original-NB18-setting",
774
+ "config": {
775
+ "kwargs": {
776
+ "BT": 32
777
+ },
778
+ "num_warps": 8,
779
+ "num_ctas": 1,
780
+ "num_stages": 3,
781
+ "maxnreg": null,
782
+ "pre_hook": null,
783
+ "ir_override": null
784
+ }
785
+ },
786
+ {
787
+ "NB": 31,
788
+ "basis": "prospective-fixed-original-NB18-setting",
789
+ "config": {
790
+ "kwargs": {
791
+ "BT": 32
792
+ },
793
+ "num_warps": 8,
794
+ "num_ctas": 1,
795
+ "num_stages": 3,
796
+ "maxnreg": null,
797
+ "pre_hook": null,
798
+ "ir_override": null
799
+ }
800
+ },
801
+ {
802
+ "NB": 32,
803
+ "basis": "prospective-fixed-original-NB18-setting",
804
+ "config": {
805
+ "kwargs": {
806
+ "BT": 32
807
+ },
808
+ "num_warps": 8,
809
+ "num_ctas": 1,
810
+ "num_stages": 3,
811
+ "maxnreg": null,
812
+ "pre_hook": null,
813
+ "ir_override": null
814
+ }
815
+ }
816
+ ],
817
+ "unseen_NB_claim": "Deterministic prospective settings, not proof of matching an undefined prior autotune choice. No speed claim.",
818
+ "other_kernels": "Unmodified FLA/Transformers dispatch; this profile changes only l2norm_fwd_kernel launch settings.",
819
+ "compiled_binaries_included": false,
820
+ "private_paths_included": false,
821
+ "builder_sha256": "833d5dcfad43a94c239ea2f5ad68d69b0c501e82d9a3b3ef530b5b85b72db379"
822
+ }
runtime.json CHANGED
@@ -2,22 +2,24 @@
2
  "torch": "2.12.0+git6bbd260",
3
  "torch_git": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5",
4
  "hip": "7.2.53211",
 
5
  "triton": "3.7.1",
6
  "fla": "0.5.2",
7
- "image_digest": "sha256:670f9f4cced18cccfb196188a3d42ff81ab617f17c447bfe033e150824d3448b",
8
- "wheels": {
9
- "fla_core-0.5.2-py3-none-any.whl": "5e830c85bad3d0d34677f98ac7074d08687a3756f0f0499d95ceb96eb6920761",
10
- "flash_linear_attention-0.5.2-py3-none-any.whl": "dcf405d81f5426393b59037097aa700d0f4a841465d5028d5aa543f4502f2400"
11
- },
12
- "gated_delta": "fla.ops.gated_delta_rule.chunk",
13
- "causal_conv": "transformers-reference-PyTorch",
14
- "full_attention": "sdpa",
15
- "runtime_installation": "isolated PYTHONPATH, no shared image mutation",
16
- "warm_start": "successful pointer head warmup checkpoint100; no failed smoke weights reused",
17
- "transformers": "5.17.0",
18
  "tokenizers": "0.23.2",
19
  "safetensors": "0.8.0",
20
- "huggingface-hub": "1.31.0",
21
- "accelerate": "1.15.0",
22
- "numpy": "2.3.5"
 
 
 
 
 
 
 
 
 
 
23
  }
 
2
  "torch": "2.12.0+git6bbd260",
3
  "torch_git": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5",
4
  "hip": "7.2.53211",
5
+ "transformers": "5.17.0",
6
  "triton": "3.7.1",
7
  "fla": "0.5.2",
8
+ "gdn_implementation": "fla.ops.gated_delta_rule.chunk",
9
+ "gdn_new_implementation": true,
 
 
 
 
 
 
 
 
 
10
  "tokenizers": "0.23.2",
11
  "safetensors": "0.8.0",
12
+ "gated_delta": "fla.ops.gated_delta_rule.chunk",
13
+ "normalization_profile": {
14
+ "kind": "decision-fla-l2norm-profile-v1",
15
+ "profile_file": "runtime-profile/profile.json",
16
+ "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f",
17
+ "guard_file": "code/profile_guard.py",
18
+ "guard_sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452",
19
+ "loader_file": "code/runtime_profile.py",
20
+ "loader_sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda",
21
+ "validated_arch": "gfx942",
22
+ "activation": "automatic before FLA import by public from_pretrained and bundle DecisionEngine",
23
+ "process_scope": "one profile per fresh Python process; same-profile loads allowed; unprofiled/different-profile mixing rejected"
24
+ }
25
  }
src/decision/example.py CHANGED
@@ -52,6 +52,9 @@ def main():
52
  'bundle_manifest_sha256': hashlib.sha256((model.bundle_path / 'bundle-manifest.json').read_bytes()).hexdigest(),
53
  'runtime': model.runtime, 'revision': args.revision, 'device': args.device,
54
  'model_name': response['model'], 'example_source_sha256': hashlib.sha256(Path(__file__).read_bytes()).hexdigest()}
 
 
 
55
  args.output.parent.mkdir(parents=True, exist_ok=True)
56
  args.output.write_text(json.dumps(record, ensure_ascii=False, indent=2) + '\n')
57
  print(json.dumps(response, ensure_ascii=False, indent=2))
 
52
  'bundle_manifest_sha256': hashlib.sha256((model.bundle_path / 'bundle-manifest.json').read_bytes()).hexdigest(),
53
  'runtime': model.runtime, 'revision': args.revision, 'device': args.device,
54
  'model_name': response['model'], 'example_source_sha256': hashlib.sha256(Path(__file__).read_bytes()).hexdigest()}
55
+ if model.runtime.get('normalization_profile') is not None:
56
+ import sys
57
+ record['normalization_telemetry'] = dict(sys.modules['_decision_process_normalization_profile_v1'].telemetry)
58
  args.output.parent.mkdir(parents=True, exist_ok=True)
59
  args.output.write_text(json.dumps(record, ensure_ascii=False, indent=2) + '\n')
60
  print(json.dumps(response, ensure_ascii=False, indent=2))
src/decision/model.py CHANGED
@@ -131,10 +131,21 @@ class DecisionModel:
131
  isinstance(v, bool) or not isinstance(v, (int, float)) or not math.isfinite(v) or v <= 0
132
  for v in temperatures.values()):
133
  raise ValueError('Bundle must contain finite positive temperatures for all three types')
134
- runtime = _runtime_report(json.loads((path / 'runtime.json').read_text()), device, allow_unvalidated_runtime)
 
 
 
 
 
 
 
 
 
 
 
 
135
  names = {'Qwen/Qwen3.5-2B': 'Decision-1.0-Sol', 'Qwen/Qwen3.5-4B': 'Decision-1.0-Nox'}
136
  name = model_name or config.get('model_name') or names.get(config.get('base_model'), path.name)
137
- api = _load_api(path)
138
  engine = api.DecisionEngine(path, path / 'code', device=device, max_length=maximum,
139
  batch_size=manifest['production_batch_size'], temperatures=temperatures,
140
  model_name=name)
 
131
  isinstance(v, bool) or not isinstance(v, (int, float)) or not math.isfinite(v) or v <= 0
132
  for v in temperatures.values()):
133
  raise ValueError('Bundle must contain finite positive temperatures for all three types')
134
+ api = _load_api(path)
135
+ expected_runtime = json.loads((path / 'runtime.json').read_text())
136
+ if expected_runtime.get('normalization_profile') is not None:
137
+ if not hasattr(api, 'prepare_runtime_profile'):
138
+ raise RuntimeError('Profiled bundle omits its automatic runtime entrypoint')
139
+ profile = api.prepare_runtime_profile(path, device=device)
140
+ else:
141
+ import sys
142
+ if '_decision_process_normalization_profile_v1' in sys.modules:
143
+ raise RuntimeError('Use separate processes for profiled and unprofiled models')
144
+ profile = None
145
+ runtime = _runtime_report(expected_runtime, device, allow_unvalidated_runtime)
146
+ runtime['normalization_profile'] = profile
147
  names = {'Qwen/Qwen3.5-2B': 'Decision-1.0-Sol', 'Qwen/Qwen3.5-4B': 'Decision-1.0-Nox'}
148
  name = model_name or config.get('model_name') or names.get(config.get('base_model'), path.name)
 
149
  engine = api.DecisionEngine(path, path / 'code', device=device, max_length=maximum,
150
  batch_size=manifest['production_batch_size'], temperatures=temperatures,
151
  model_name=name)
src/decision_local.egg-info/PKG-INFO ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ Metadata-Version: 2.4
2
+ Name: decision-local
3
+ Version: 1.0.0
4
+ Summary: Local typed inference for exported Decision decoder models
5
+ Requires-Python: >=3.10
6
+ Provides-Extra: hub
7
+ Requires-Dist: huggingface-hub==1.31.0; extra == "hub"
src/decision_local.egg-info/SOURCES.txt ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ pyproject.toml
2
+ src/decision/__init__.py
3
+ src/decision/example.py
4
+ src/decision/model.py
5
+ src/decision_local.egg-info/PKG-INFO
6
+ src/decision_local.egg-info/SOURCES.txt
7
+ src/decision_local.egg-info/dependency_links.txt
8
+ src/decision_local.egg-info/entry_points.txt
9
+ src/decision_local.egg-info/requires.txt
10
+ src/decision_local.egg-info/top_level.txt
11
+ tests/test_wrapper.py
src/decision_local.egg-info/dependency_links.txt ADDED
@@ -0,0 +1 @@
 
 
1
+
src/decision_local.egg-info/entry_points.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ [console_scripts]
2
+ decision-example = decision.example:main
src/decision_local.egg-info/requires.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+
2
+ [hub]
3
+ huggingface-hub==1.31.0
src/decision_local.egg-info/top_level.txt ADDED
@@ -0,0 +1 @@
 
 
1
+ decision
temperature.json CHANGED
@@ -1,22 +1,420 @@
1
  {
2
- "temperature": 0.7033302993804421,
 
 
 
 
 
 
 
 
 
 
 
3
  "temperatures": {
4
- "choice": 0.7033302993804421,
5
- "noul": 0.7033302993804421,
6
- "score": 0.7033302993804421
7
  },
8
- "method": "one-global-positive-temperature; family-macro-development-Brier",
9
- "bounds": [
10
- 0.25,
11
- 4.0
12
- ],
13
- "dev_examples": 256,
14
- "families": 8,
15
- "raw_brier": 0.045776584869665994,
16
- "calibrated_dev_brier": 0.04377952758544348,
17
- "boundary_solution": false,
18
  "ranking_unchanged": true,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
19
  "heldout_used": false,
20
- "dev_sha256": "a38f9be168553d5a82a91991aca86aaa602e1951957bac7c475abcec8dcc232d",
21
- "raw_predictions_sha256": "38df5ace521ee8ec7decc752ab7a49e5e7b2cda751c7d7d6a4fdf00ad8af4202"
 
 
 
 
 
 
 
22
  }
 
1
  {
2
+ "version": "stage4-nine-family-macro-nll-beta-bisection-v1",
3
+ "objective": "equal-family macro NLL",
4
+ "input_sha256": {
5
+ "cal": "5ba3bd7527da831d77cec8c6a2f260a71e10a57742e356510c719f8f30d61a70",
6
+ "predictions": "b9e2cf2c466e6f9422e627905625400d683c6ce0cecb015c16e5784b0c3c577c",
7
+ "inference-summary": "8e537d9729e9f56636eef327063e356e9f32bc96b46edd252263bb880cb450b1",
8
+ "inference-binding": "8d7d7a5cbb58104b5bf2037088b770d1173e719a1701d56001efc35c7d8b7224",
9
+ "selection": "b1d31cb6eea09b75471f256845530217cb0fd19d668c524c6e4e092267c8e9cd",
10
+ "model-manifest": "55fd2542828d86625123f5b0e59b2f8e967ec9db38ec6d3f67b64c1f079408b6",
11
+ "plan": "e5086ac34dbf190bd1f1ce6f0a426a37710160878b67120aab9bdbfed49e8348"
12
+ },
13
+ "temperature": 0.9982273489600023,
14
  "temperatures": {
15
+ "choice": 0.9982273489600023,
16
+ "noul": 0.9982273489600023,
17
+ "score": 0.9982273489600023
18
  },
 
 
 
 
 
 
 
 
 
 
19
  "ranking_unchanged": true,
20
+ "ranking_claim_scope": "Positive global scaling preserves mathematical logit ordering; finite-precision inference parity is audited separately",
21
+ "source_sha256": "80e008d3a0eab5604de4b15bc897ddfa26c6761430753478ad5972a6cf08e5df",
22
+ "family_rows": {
23
+ "stage4_dense_table": 100,
24
+ "stage4_scope": 100,
25
+ "stage4_arithmetic": 100,
26
+ "stage4_ordinal": 100,
27
+ "stage4_natural_nli": 200,
28
+ "stage4_automaton": 100,
29
+ "stage4_boolean": 100,
30
+ "stage4_relations": 100,
31
+ "stage4_registers": 100
32
+ },
33
+ "selected_checkpoint": "checkpoint-001563",
34
+ "selected_arm": "ce",
35
+ "fit": {
36
+ "temperature": 0.9982273489600023,
37
+ "inverse_temperature": 1.0017757989117855,
38
+ "status": "relative-bracket-converged",
39
+ "boundary_hit": false,
40
+ "iterations": 38,
41
+ "beta_final_bracket": [
42
+ 1.0017757988754965,
43
+ 1.0017757989480742
44
+ ],
45
+ "objective_at_T1": 0.6400866469651806,
46
+ "objective_at_selected_T": 0.6400860719946871,
47
+ "derivative_at_selected_beta": 5.17162410925694e-12,
48
+ "initial_boundary_derivatives": {
49
+ "beta_0.05": -2.9645769693802926,
50
+ "beta_20": 0.25425204205854235
51
+ },
52
+ "derivative_at_beta1": -0.0006479866804508699,
53
+ "initial_boundary_objectives": [
54
+ 1.299894618307853,
55
+ 5.139564840374214
56
+ ]
57
+ },
58
+ "raw": {
59
+ "temperature": 1.0,
60
+ "family_macro_nll": 0.6400866469651806,
61
+ "family_macro_brier": 0.3141964724320727,
62
+ "family_macro_native_accuracy": 0.7488888888888889,
63
+ "by_family": {
64
+ "stage4_dense_table": {
65
+ "nll": 0.44557900925725874,
66
+ "brier": 0.21564659042540416,
67
+ "native_accuracy": 0.88,
68
+ "rows": 100
69
+ },
70
+ "stage4_scope": {
71
+ "nll": 0.0498655140340889,
72
+ "brier": 0.028826434428688895,
73
+ "native_accuracy": 0.98,
74
+ "rows": 100
75
+ },
76
+ "stage4_arithmetic": {
77
+ "nll": 1.0564804127990468,
78
+ "brier": 0.3403965985820037,
79
+ "native_accuracy": 0.74,
80
+ "rows": 100
81
+ },
82
+ "stage4_ordinal": {
83
+ "nll": 0.4764526628150301,
84
+ "brier": 0.25096833388492373,
85
+ "native_accuracy": 0.79,
86
+ "rows": 100
87
+ },
88
+ "stage4_natural_nli": {
89
+ "nll": 0.5360083749004084,
90
+ "brier": 0.3208918487912171,
91
+ "native_accuracy": 0.76,
92
+ "rows": 200
93
+ },
94
+ "stage4_automaton": {
95
+ "nll": 1.3367082473383853,
96
+ "brier": 0.6598012810538971,
97
+ "native_accuracy": 0.41,
98
+ "rows": 100
99
+ },
100
+ "stage4_boolean": {
101
+ "nll": 0.45878482504605034,
102
+ "brier": 0.3056423821560782,
103
+ "native_accuracy": 0.75,
104
+ "rows": 100
105
+ },
106
+ "stage4_relations": {
107
+ "nll": 0.8561881989781089,
108
+ "brier": 0.39314463269485195,
109
+ "native_accuracy": 0.69,
110
+ "rows": 100
111
+ },
112
+ "stage4_registers": {
113
+ "nll": 0.5447125775182473,
114
+ "brier": 0.31245014987158926,
115
+ "native_accuracy": 0.74,
116
+ "rows": 100
117
+ }
118
+ },
119
+ "by_task_type": {
120
+ "score": {
121
+ "family_macro_nll": 0.48249895787513475,
122
+ "family_macro_brier": 0.24589504270711923,
123
+ "family_macro_native_accuracy": 0.8325,
124
+ "by_family": {
125
+ "stage4_dense_table": {
126
+ "nll": 0.4885452529352393,
127
+ "brier": 0.24082175152931473,
128
+ "native_accuracy": 0.875,
129
+ "rows": 16
130
+ },
131
+ "stage4_ordinal": {
132
+ "nll": 0.4764526628150301,
133
+ "brier": 0.25096833388492373,
134
+ "native_accuracy": 0.79,
135
+ "rows": 100
136
+ }
137
+ }
138
+ },
139
+ "choice": {
140
+ "family_macro_nll": 0.7578712143392436,
141
+ "family_macro_brier": 0.3425364771873384,
142
+ "family_macro_native_accuracy": 0.7228736900165471,
143
+ "by_family": {
144
+ "stage4_scope": {
145
+ "nll": 0.06248730610766334,
146
+ "brier": 0.03778805355766209,
147
+ "native_accuracy": 0.972972972972973,
148
+ "rows": 74
149
+ },
150
+ "stage4_arithmetic": {
151
+ "nll": 1.0564804127990468,
152
+ "brier": 0.3403965985820037,
153
+ "native_accuracy": 0.74,
154
+ "rows": 100
155
+ },
156
+ "stage4_natural_nli": {
157
+ "nll": 0.5360083749004084,
158
+ "brier": 0.3208918487912171,
159
+ "native_accuracy": 0.76,
160
+ "rows": 200
161
+ },
162
+ "stage4_automaton": {
163
+ "nll": 1.6447244044728033,
164
+ "brier": 0.7579729767796513,
165
+ "native_accuracy": 0.3142857142857143,
166
+ "rows": 70
167
+ },
168
+ "stage4_relations": {
169
+ "nll": 1.3089671086886498,
170
+ "brier": 0.5900762617472654,
171
+ "native_accuracy": 0.5166666666666667,
172
+ "rows": 60
173
+ },
174
+ "stage4_registers": {
175
+ "nll": 0.4960359448890203,
176
+ "brier": 0.26223171850149674,
177
+ "native_accuracy": 0.8133333333333334,
178
+ "rows": 75
179
+ },
180
+ "stage4_dense_table": {
181
+ "nll": 0.20039494851711318,
182
+ "brier": 0.08839788235207212,
183
+ "native_accuracy": 0.9428571428571428,
184
+ "rows": 35
185
+ }
186
+ }
187
+ },
188
+ "noul": {
189
+ "family_macro_nll": 0.427528942482448,
190
+ "family_macro_brier": 0.2664778929921975,
191
+ "family_macro_native_accuracy": 0.7816780045351474,
192
+ "by_family": {
193
+ "stage4_scope": {
194
+ "nll": 0.013941951978530863,
195
+ "brier": 0.003320287676995936,
196
+ "native_accuracy": 1.0,
197
+ "rows": 26
198
+ },
199
+ "stage4_boolean": {
200
+ "nll": 0.45878482504605034,
201
+ "brier": 0.3056423821560782,
202
+ "native_accuracy": 0.75,
203
+ "rows": 100
204
+ },
205
+ "stage4_relations": {
206
+ "nll": 0.17701983441229752,
207
+ "brier": 0.09774718911623191,
208
+ "native_accuracy": 0.95,
209
+ "rows": 40
210
+ },
211
+ "stage4_dense_table": {
212
+ "nll": 0.6066806873604711,
213
+ "brier": 0.29831806399487465,
214
+ "native_accuracy": 0.8367346938775511,
215
+ "rows": 49
216
+ },
217
+ "stage4_registers": {
218
+ "nll": 0.690742475405928,
219
+ "brier": 0.463105443981867,
220
+ "native_accuracy": 0.52,
221
+ "rows": 25
222
+ },
223
+ "stage4_automaton": {
224
+ "nll": 0.6180038806914102,
225
+ "brier": 0.4307339910271372,
226
+ "native_accuracy": 0.6333333333333333,
227
+ "rows": 30
228
+ }
229
+ }
230
+ }
231
+ },
232
+ "proper_score_convention": "multiclass sum-squared Brier, including two entries for noul; NLL in nats; equal family weight within each reported task type"
233
+ },
234
+ "calibrated": {
235
+ "temperature": 0.9982273489600023,
236
+ "family_macro_nll": 0.6400860719946871,
237
+ "family_macro_brier": 0.31418442074316877,
238
+ "family_macro_native_accuracy": 0.7488888888888889,
239
+ "by_family": {
240
+ "stage4_dense_table": {
241
+ "nll": 0.445938296164393,
242
+ "brier": 0.2156572009593377,
243
+ "native_accuracy": 0.88,
244
+ "rows": 100
245
+ },
246
+ "stage4_scope": {
247
+ "nll": 0.04986573972928388,
248
+ "brier": 0.02882890500497602,
249
+ "native_accuracy": 0.98,
250
+ "rows": 100
251
+ },
252
+ "stage4_arithmetic": {
253
+ "nll": 1.056168898096333,
254
+ "brier": 0.3402942706104855,
255
+ "native_accuracy": 0.74,
256
+ "rows": 100
257
+ },
258
+ "stage4_ordinal": {
259
+ "nll": 0.4763746819749599,
260
+ "brier": 0.2509243877991786,
261
+ "native_accuracy": 0.79,
262
+ "rows": 100
263
+ },
264
+ "stage4_natural_nli": {
265
+ "nll": 0.5358254005746533,
266
+ "brier": 0.3208298384181523,
267
+ "native_accuracy": 0.76,
268
+ "rows": 200
269
+ },
270
+ "stage4_automaton": {
271
+ "nll": 1.3368995049684889,
272
+ "brier": 0.6598855736978168,
273
+ "native_accuracy": 0.41,
274
+ "rows": 100
275
+ },
276
+ "stage4_boolean": {
277
+ "nll": 0.45874553976333154,
278
+ "brier": 0.30562878190540066,
279
+ "native_accuracy": 0.75,
280
+ "rows": 100
281
+ },
282
+ "stage4_relations": {
283
+ "nll": 0.8561430453929648,
284
+ "brier": 0.3931506584302739,
285
+ "native_accuracy": 0.69,
286
+ "rows": 100
287
+ },
288
+ "stage4_registers": {
289
+ "nll": 0.5448135412877755,
290
+ "brier": 0.31246016986289765,
291
+ "native_accuracy": 0.74,
292
+ "rows": 100
293
+ }
294
+ },
295
+ "by_task_type": {
296
+ "score": {
297
+ "family_macro_nll": 0.4824392478193705,
298
+ "family_macro_brier": 0.24583611006762263,
299
+ "family_macro_native_accuracy": 0.8325,
300
+ "by_family": {
301
+ "stage4_dense_table": {
302
+ "nll": 0.48850381366378104,
303
+ "brier": 0.24074783233606664,
304
+ "native_accuracy": 0.875,
305
+ "rows": 16
306
+ },
307
+ "stage4_ordinal": {
308
+ "nll": 0.4763746819749599,
309
+ "brier": 0.2509243877991786,
310
+ "native_accuracy": 0.79,
311
+ "rows": 100
312
+ }
313
+ }
314
+ },
315
+ "choice": {
316
+ "family_macro_nll": 0.757831004980382,
317
+ "family_macro_brier": 0.34252562919037327,
318
+ "family_macro_native_accuracy": 0.7228736900165471,
319
+ "by_family": {
320
+ "stage4_scope": {
321
+ "nll": 0.06250739149911984,
322
+ "brier": 0.03779678126229391,
323
+ "native_accuracy": 0.972972972972973,
324
+ "rows": 74
325
+ },
326
+ "stage4_arithmetic": {
327
+ "nll": 1.056168898096333,
328
+ "brier": 0.3402942706104855,
329
+ "native_accuracy": 0.74,
330
+ "rows": 100
331
+ },
332
+ "stage4_natural_nli": {
333
+ "nll": 0.5358254005746533,
334
+ "brier": 0.3208298384181523,
335
+ "native_accuracy": 0.76,
336
+ "rows": 200
337
+ },
338
+ "stage4_automaton": {
339
+ "nll": 1.6449019417622601,
340
+ "brier": 0.758039538459674,
341
+ "native_accuracy": 0.3142857142857143,
342
+ "rows": 70
343
+ },
344
+ "stage4_relations": {
345
+ "nll": 1.3088570377916526,
346
+ "brier": 0.5900881012282487,
347
+ "native_accuracy": 0.5166666666666667,
348
+ "rows": 60
349
+ },
350
+ "stage4_registers": {
351
+ "nll": 0.4961240001814749,
352
+ "brier": 0.2622496602881456,
353
+ "native_accuracy": 0.8133333333333334,
354
+ "rows": 75
355
+ },
356
+ "stage4_dense_table": {
357
+ "nll": 0.20043236495718011,
358
+ "brier": 0.0883812140656126,
359
+ "native_accuracy": 0.9428571428571428,
360
+ "rows": 35
361
+ }
362
+ }
363
+ },
364
+ "noul": {
365
+ "family_macro_nll": 0.42770221644099665,
366
+ "family_macro_brier": 0.26650089018224704,
367
+ "family_macro_native_accuracy": 0.7816780045351474,
368
+ "by_family": {
369
+ "stage4_scope": {
370
+ "nll": 0.013885653922827704,
371
+ "brier": 0.0033049495033789502,
372
+ "native_accuracy": 1.0,
373
+ "rows": 26
374
+ },
375
+ "stage4_boolean": {
376
+ "nll": 0.45874553976333154,
377
+ "brier": 0.30562878190540066,
378
+ "native_accuracy": 0.75,
379
+ "rows": 100
380
+ },
381
+ "stage4_relations": {
382
+ "nll": 0.17707205679493318,
383
+ "brier": 0.09774449423331166,
384
+ "native_accuracy": 0.95,
385
+ "rows": 40
386
+ },
387
+ "stage4_dense_table": {
388
+ "nll": 0.6074007311085204,
389
+ "brier": 0.298375760944087,
390
+ "native_accuracy": 0.8367346938775511,
391
+ "rows": 49
392
+ },
393
+ "stage4_registers": {
394
+ "nll": 0.6908821646066775,
395
+ "brier": 0.46309169858715404,
396
+ "native_accuracy": 0.52,
397
+ "rows": 25
398
+ },
399
+ "stage4_automaton": {
400
+ "nll": 0.6182271524496894,
401
+ "brier": 0.4308596559201498,
402
+ "native_accuracy": 0.6333333333333333,
403
+ "rows": 30
404
+ }
405
+ }
406
+ }
407
+ },
408
+ "proper_score_convention": "multiclass sum-squared Brier, including two entries for noul; NLL in nats; equal family weight within each reported task type"
409
+ },
410
  "heldout_used": false,
411
+ "SELECT_used_for_temperature": false,
412
+ "generalization_improvement_claim": false,
413
+ "raw_predictions_sha256": "b9e2cf2c466e6f9422e627905625400d683c6ce0cecb015c16e5784b0c3c577c",
414
+ "dev_sha256": "5ba3bd7527da831d77cec8c6a2f260a71e10a57742e356510c719f8f30d61a70",
415
+ "cal_sha256": "5ba3bd7527da831d77cec8c6a2f260a71e10a57742e356510c719f8f30d61a70",
416
+ "calibration_dataset_role": "CAL; dev_sha256 is a legacy portable-bundle schema alias, not SELECT",
417
+ "source_calibration_artifact_sha256": "f6be19545ba1f81456b201b5cf83360e9b80e722080b0ce6c997daa4f228d11c",
418
+ "export_metadata_bridge_sha256": "e4cd2ceaa9bd9697f5c47f5a4504526612e6f23fd87a095fe9c92fd0fae9ae31",
419
+ "temperature_refit_performed_by_bridge": false
420
  }