Update official TypeSafe SDK usage and current Decision model cards
Browse files- .gitattributes +1 -0
- DIAGNOSTICS.md +26 -16
- EVALUATION.md +10 -6
- MATERIALS.json +22 -16
- README.md +63 -17
- SENSITIVITY.md +9 -5
- TASKS.md +64 -64
- USAGE.md +46 -53
- WEIGHTING.md +9 -5
- assets/decision-matrix.pdf +0 -0
- assets/decision-matrix.png +2 -2
- assets/decision-matrix.svg +782 -612
- assets/decision-ranking.pdf +0 -0
- assets/decision-ranking.png +2 -2
- assets/decision-ranking.svg +168 -134
- assets/decision-sol-2b-header.png +3 -0
- metrics/benchmark.json +0 -0
- metrics/evaluation-provenance.json +288 -645
- release-manifest.json +50 -38
.gitattributes
CHANGED
|
@@ -56,3 +56,4 @@ assets/decision-question-scaling.png filter=lfs diff=lfs merge=lfs -text
|
|
| 56 |
assets/decision-expanded-v5-600px.png filter=lfs diff=lfs merge=lfs -text
|
| 57 |
assets/decision-matrix.png filter=lfs diff=lfs merge=lfs -text
|
| 58 |
assets/decision-ranking.png filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 56 |
assets/decision-expanded-v5-600px.png filter=lfs diff=lfs merge=lfs -text
|
| 57 |
assets/decision-matrix.png filter=lfs diff=lfs merge=lfs -text
|
| 58 |
assets/decision-ranking.png filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
assets/decision-sol-2b-header.png filter=lfs diff=lfs merge=lfs -text
|
DIAGNOSTICS.md
CHANGED
|
@@ -6,16 +6,18 @@ These axes remain separate from headline accuracy. Probability metrics use the s
|
|
| 6 |
|
| 7 |
| Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
|
| 8 |
|---|---:|---:|---:|---:|---:|---:|
|
| 9 |
-
| Lux | 1046/1046 | 0.
|
| 10 |
-
| Nox | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
|
| 11 |
| Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
|
| 12 |
| Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
|
| 13 |
| Qwen3.5-9B | 1046/1046 | 0.3673 | 0.7597 | 9.75 | 47.42 | 0.0872 |
|
| 14 |
| Decider | 1046/1046 | 0.4122 | 0.7969 | 8.09 | 35.18 | 0.1172 |
|
| 15 |
| Qwen3.5-4B | 1046/1046 | 0.4199 | 0.8186 | 8.98 | 34.70 | 0.1233 |
|
| 16 |
-
| Sol | 1046/1046 | 0.5454 | 1.1094 | 11.25 | 16.16 | 0.2073 |
|
|
|
|
| 17 |
| Kev-0.8B | 1046/1046 | 0.4839 | 0.9489 | 2.58 | 21.61 | 0.1782 |
|
| 18 |
| Qwen3.5-2B | 1046/1046 | 0.5708 | 1.2327 | 12.30 | 11.76 | 0.2699 |
|
|
|
|
| 19 |
| Laya · English | 1046/1046 | 0.5832 | 1.2275 | 12.90 | 0.48 | 0.2780 |
|
| 20 |
| Laya · Multilingual | 1046/1046 | 0.6873 | 1.4653 | 21.36 | 0.00 | 0.3460 |
|
| 21 |
| Jev | 1046/1046 | 0.1912 | 0.5743 | 3.35 | 76.96 | 0.0346 |
|
|
@@ -26,16 +28,18 @@ Coverage at an error threshold keeps whole confidence-tie groups together. These
|
|
| 26 |
|
| 27 |
| Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
|
| 28 |
|---|---:|---:|---:|---:|
|
| 29 |
-
| Lux | 36/36 | 77.78 | 8.33 | 0.
|
| 30 |
-
| Nox | 36/36 | 63.89 | 16.67 | 0.0795 |
|
| 31 |
| Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
|
| 32 |
| Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
|
| 33 |
| Qwen3.5-9B | 36/36 | 75.00 | 11.11 | 0.1157 |
|
| 34 |
| Decider | 36/36 | 83.33 | 11.11 | 0.0701 |
|
| 35 |
| Qwen3.5-4B | 36/36 | 77.78 | 13.89 | 0.1390 |
|
| 36 |
-
| Sol | 36/36 | 41.67 | 38.89 | 0.0694 |
|
|
|
|
| 37 |
| Kev-0.8B | 36/36 | 55.56 | 16.67 | 0.0808 |
|
| 38 |
| Qwen3.5-2B | 36/36 | 55.56 | 30.56 | 0.2163 |
|
|
|
|
| 39 |
| Laya · English | 36/36 | 44.44 | 19.44 | 0.0928 |
|
| 40 |
| Laya · Multilingual | 36/36 | 50.00 | 22.22 | 0.1491 |
|
| 41 |
| Jev | 36/36 | 86.11 | 0.00 | 0.0208 |
|
|
@@ -46,36 +50,40 @@ The 36 paired permutations test the same semantics under changed option order. S
|
|
| 46 |
|
| 47 |
| Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
|
| 48 |
|---|---:|---:|---:|---:|---:|---:|
|
| 49 |
-
| Lux | 110/110 |
|
| 50 |
-
| Nox | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
|
| 51 |
| Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
|
| 52 |
| Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
|
| 53 |
| Qwen3.5-9B | 110/110 | 80.91 | 67.62 | 7.27 | 0.7585 | 19.09 |
|
| 54 |
| Decider | 110/110 | 69.09 | 68.82 | 16.36 | 0.6772 | 13.59 |
|
| 55 |
| Qwen3.5-4B | 110/110 | 72.73 | 59.89 | 0.91 | 0.8120 | 22.35 |
|
| 56 |
-
| Sol | 110/110 | 65.45 | 77.19 | 24.55 | 0.5424 | 5.68 |
|
|
|
|
| 57 |
| Kev-0.8B | 110/110 | 87.27 | 42.11 | 0.00 | 0.9796 | 40.34 |
|
| 58 |
| Qwen3.5-2B | 110/110 | 51.82 | 65.19 | 9.09 | 0.7370 | 4.68 |
|
|
|
|
| 59 |
| Laya · English | 110/110 | 49.09 | 67.96 | 0.00 | 0.7497 | -5.58 |
|
| 60 |
| Laya · Multilingual | 110/110 | 37.27 | 75.70 | 29.09 | 0.5307 | -0.49 |
|
| 61 |
| Jev | 110/110 | 93.64 | 62.04 | 16.36 | 0.7657 | 30.30 |
|
| 62 |
|
| 63 |
-
The 110 unknowable examples have no scored true class and are excluded from accuracy. Confidence is compared with matched evidence-bearing controls.
|
| 64 |
|
| 65 |
## Native contract coverage
|
| 66 |
|
| 67 |
| Model | Original probability rows | Transfer probability rows | Transfer truncated questions |
|
| 68 |
|---|---:|---:|---:|
|
| 69 |
-
| Lux | 2720/2720 | 1264/1264 | 0 |
|
| 70 |
-
| Nox | 2720/2720 | 1264/1264 | 0 |
|
| 71 |
| Kev-9B | 2720/2720 | 1264/1264 | 0 |
|
| 72 |
| Kev-4B | 2720/2720 | 1264/1264 | 0 |
|
| 73 |
| Qwen3.5-9B | 2720/2720 | 1264/1264 | 0 |
|
| 74 |
| Decider | 2720/2720 | 1264/1264 | 0 |
|
| 75 |
| Qwen3.5-4B | 2720/2720 | 1264/1264 | 0 |
|
| 76 |
-
| Sol | 2720/2720 | 1264/1264 | 0 |
|
|
|
|
| 77 |
| Kev-0.8B | 2720/2720 | 1264/1264 | 0 |
|
| 78 |
| Qwen3.5-2B | 2720/2720 | 1264/1264 | 0 |
|
|
|
|
| 79 |
| Laya · English | 2720/2720 | 1264/1264 | 34 |
|
| 80 |
| Laya · Multilingual | 2720/2720 | 1264/1264 | 14 |
|
| 81 |
| Jev | 2720/2720 | 1264/1264 | — / not observable |
|
|
@@ -86,16 +94,18 @@ Transfer coverage includes all 1,264 questions: clean, unknown-evidence and orde
|
|
| 86 |
|
| 87 |
| Model | Overall % | 95% component-bootstrap interval |
|
| 88 |
|---|---:|---:|
|
| 89 |
-
| Lux | 76.
|
| 90 |
-
| Nox | 73.09 | 71.57–74.56 |
|
| 91 |
| Kev-9B | 71.89 | 70.42–73.35 |
|
| 92 |
| Kev-4B | 70.09 | 68.45–71.63 |
|
| 93 |
| Qwen3.5-9B | 69.73 | 68.27–71.20 |
|
| 94 |
| Decider | 67.71 | 66.10–69.34 |
|
| 95 |
| Qwen3.5-4B | 67.29 | 65.89–68.69 |
|
| 96 |
-
| Sol | 66.32 | 64.77–67.85 |
|
|
|
|
| 97 |
| Kev-0.8B | 58.28 | 56.64–59.89 |
|
| 98 |
| Qwen3.5-2B | 57.24 | 55.74–58.76 |
|
|
|
|
| 99 |
| Laya · English | 51.03 | 49.43–52.68 |
|
| 100 |
| Laya · Multilingual | 47.19 | 45.58–48.82 |
|
| 101 |
| Jev | 81.05 | 79.70–82.35 |
|
|
|
|
| 6 |
|
| 7 |
| Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
|
| 8 |
|---|---:|---:|---:|---:|---:|---:|
|
| 9 |
+
| Lux-9B | 1046/1046 | 0.3004 | 0.6086 | 5.89 | 63.38 | 0.0614 |
|
| 10 |
+
| Nox-4B | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
|
| 11 |
| Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
|
| 12 |
| Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
|
| 13 |
| Qwen3.5-9B | 1046/1046 | 0.3673 | 0.7597 | 9.75 | 47.42 | 0.0872 |
|
| 14 |
| Decider | 1046/1046 | 0.4122 | 0.7969 | 8.09 | 35.18 | 0.1172 |
|
| 15 |
| Qwen3.5-4B | 1046/1046 | 0.4199 | 0.8186 | 8.98 | 34.70 | 0.1233 |
|
| 16 |
+
| Sol-2B | 1046/1046 | 0.5454 | 1.1094 | 11.25 | 16.16 | 0.2073 |
|
| 17 |
+
| Eos-0.8B | 1046/1046 | 0.5847 | 1.1406 | 13.96 | 10.23 | 0.2611 |
|
| 18 |
| Kev-0.8B | 1046/1046 | 0.4839 | 0.9489 | 2.58 | 21.61 | 0.1782 |
|
| 19 |
| Qwen3.5-2B | 1046/1046 | 0.5708 | 1.2327 | 12.30 | 11.76 | 0.2699 |
|
| 20 |
+
| Kai-0.6B | 1046/1046 | 0.6392 | 1.2391 | 14.36 | 4.02 | 0.3345 |
|
| 21 |
| Laya · English | 1046/1046 | 0.5832 | 1.2275 | 12.90 | 0.48 | 0.2780 |
|
| 22 |
| Laya · Multilingual | 1046/1046 | 0.6873 | 1.4653 | 21.36 | 0.00 | 0.3460 |
|
| 23 |
| Jev | 1046/1046 | 0.1912 | 0.5743 | 3.35 | 76.96 | 0.0346 |
|
|
|
|
| 28 |
|
| 29 |
| Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
|
| 30 |
|---|---:|---:|---:|---:|
|
| 31 |
+
| Lux-9B | 36/36 | 77.78 | 8.33 | 0.0819 |
|
| 32 |
+
| Nox-4B | 36/36 | 63.89 | 16.67 | 0.0795 |
|
| 33 |
| Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
|
| 34 |
| Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
|
| 35 |
| Qwen3.5-9B | 36/36 | 75.00 | 11.11 | 0.1157 |
|
| 36 |
| Decider | 36/36 | 83.33 | 11.11 | 0.0701 |
|
| 37 |
| Qwen3.5-4B | 36/36 | 77.78 | 13.89 | 0.1390 |
|
| 38 |
+
| Sol-2B | 36/36 | 41.67 | 38.89 | 0.0694 |
|
| 39 |
+
| Eos-0.8B | 36/36 | 55.56 | 11.11 | 0.1505 |
|
| 40 |
| Kev-0.8B | 36/36 | 55.56 | 16.67 | 0.0808 |
|
| 41 |
| Qwen3.5-2B | 36/36 | 55.56 | 30.56 | 0.2163 |
|
| 42 |
+
| Kai-0.6B | 36/36 | 44.44 | 13.89 | 0.1147 |
|
| 43 |
| Laya · English | 36/36 | 44.44 | 19.44 | 0.0928 |
|
| 44 |
| Laya · Multilingual | 36/36 | 50.00 | 22.22 | 0.1491 |
|
| 45 |
| Jev | 36/36 | 86.11 | 0.00 | 0.0208 |
|
|
|
|
| 50 |
|
| 51 |
| Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
|
| 52 |
|---|---:|---:|---:|---:|---:|---:|
|
| 53 |
+
| Lux-9B | 110/110 | 88.18 | 63.68 | 16.36 | 0.7292 | 26.22 |
|
| 54 |
+
| Nox-4B | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
|
| 55 |
| Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
|
| 56 |
| Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
|
| 57 |
| Qwen3.5-9B | 110/110 | 80.91 | 67.62 | 7.27 | 0.7585 | 19.09 |
|
| 58 |
| Decider | 110/110 | 69.09 | 68.82 | 16.36 | 0.6772 | 13.59 |
|
| 59 |
| Qwen3.5-4B | 110/110 | 72.73 | 59.89 | 0.91 | 0.8120 | 22.35 |
|
| 60 |
+
| Sol-2B | 110/110 | 65.45 | 77.19 | 24.55 | 0.5424 | 5.68 |
|
| 61 |
+
| Eos-0.8B | 110/110 | 55.45 | 63.01 | 9.09 | 0.7772 | 1.93 |
|
| 62 |
| Kev-0.8B | 110/110 | 87.27 | 42.11 | 0.00 | 0.9796 | 40.34 |
|
| 63 |
| Qwen3.5-2B | 110/110 | 51.82 | 65.19 | 9.09 | 0.7370 | 4.68 |
|
| 64 |
+
| Kai-0.6B | 110/110 | 37.27 | 52.11 | 18.18 | 0.8263 | 0.62 |
|
| 65 |
| Laya · English | 110/110 | 49.09 | 67.96 | 0.00 | 0.7497 | -5.58 |
|
| 66 |
| Laya · Multilingual | 110/110 | 37.27 | 75.70 | 29.09 | 0.5307 | -0.49 |
|
| 67 |
| Jev | 110/110 | 93.64 | 62.04 | 16.36 | 0.7657 | 30.30 |
|
| 68 |
|
| 69 |
+
The 110 unknowable examples have no scored true class and are excluded from accuracy. Confidence is compared with matched evidence-bearing controls. These variants can combine evidence removal, candidate deletion and option permutation. Their confidence shifts are descriptive and do not isolate pure abstention or evidence sensitivity; these are not correctness scores.
|
| 70 |
|
| 71 |
## Native contract coverage
|
| 72 |
|
| 73 |
| Model | Original probability rows | Transfer probability rows | Transfer truncated questions |
|
| 74 |
|---|---:|---:|---:|
|
| 75 |
+
| Lux-9B | 2720/2720 | 1264/1264 | 0 |
|
| 76 |
+
| Nox-4B | 2720/2720 | 1264/1264 | 0 |
|
| 77 |
| Kev-9B | 2720/2720 | 1264/1264 | 0 |
|
| 78 |
| Kev-4B | 2720/2720 | 1264/1264 | 0 |
|
| 79 |
| Qwen3.5-9B | 2720/2720 | 1264/1264 | 0 |
|
| 80 |
| Decider | 2720/2720 | 1264/1264 | 0 |
|
| 81 |
| Qwen3.5-4B | 2720/2720 | 1264/1264 | 0 |
|
| 82 |
+
| Sol-2B | 2720/2720 | 1264/1264 | 0 |
|
| 83 |
+
| Eos-0.8B | 2720/2720 | 1264/1264 | 0 |
|
| 84 |
| Kev-0.8B | 2720/2720 | 1264/1264 | 0 |
|
| 85 |
| Qwen3.5-2B | 2720/2720 | 1264/1264 | 0 |
|
| 86 |
+
| Kai-0.6B | 2720/2720 | 1264/1264 | 0 |
|
| 87 |
| Laya · English | 2720/2720 | 1264/1264 | 34 |
|
| 88 |
| Laya · Multilingual | 2720/2720 | 1264/1264 | 14 |
|
| 89 |
| Jev | 2720/2720 | 1264/1264 | — / not observable |
|
|
|
|
| 94 |
|
| 95 |
| Model | Overall % | 95% component-bootstrap interval |
|
| 96 |
|---|---:|---:|
|
| 97 |
+
| Lux-9B | 76.94 | 75.60–78.24 |
|
| 98 |
+
| Nox-4B | 73.09 | 71.57–74.56 |
|
| 99 |
| Kev-9B | 71.89 | 70.42–73.35 |
|
| 100 |
| Kev-4B | 70.09 | 68.45–71.63 |
|
| 101 |
| Qwen3.5-9B | 69.73 | 68.27–71.20 |
|
| 102 |
| Decider | 67.71 | 66.10–69.34 |
|
| 103 |
| Qwen3.5-4B | 67.29 | 65.89–68.69 |
|
| 104 |
+
| Sol-2B | 66.32 | 64.77–67.85 |
|
| 105 |
+
| Eos-0.8B | 61.89 | 60.26–63.53 |
|
| 106 |
| Kev-0.8B | 58.28 | 56.64–59.89 |
|
| 107 |
| Qwen3.5-2B | 57.24 | 55.74–58.76 |
|
| 108 |
+
| Kai-0.6B | 53.52 | 51.85–55.25 |
|
| 109 |
| Laya · English | 51.03 | 49.43–52.68 |
|
| 110 |
| Laya · Multilingual | 47.19 | 45.58–48.82 |
|
| 111 |
| Jev | 81.05 | 79.70–82.35 |
|
EVALUATION.md
CHANGED
|
@@ -1,29 +1,33 @@
|
|
| 1 |
# Evaluation
|
| 2 |
|
| 3 |
-
The comparison covers **3,766 scored decisions across 54 tasks** and all
|
| 4 |
|
| 5 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 6 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 7 |
-
|
|
| 8 |
-
|
|
| 9 |
-
| Nox | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 10 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 11 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 12 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
| 13 |
| Decider | 2B | 64.01 | 46.58 | 92.03 | 84.38 | 69.31 | 67.71 |
|
| 14 |
| Qwen3.5-4B | 4B | 69.89 | 43.33 | 87.97 | 79.79 | 68.83 | 67.29 |
|
|
|
|
|
|
|
| 15 |
| Kev-0.8B | 0.8B | 60.14 | 42.29 | 67.81 | 68.75 | 61.19 | 58.28 |
|
| 16 |
| Qwen3.5-2B | 2B | 57.12 | 39.00 | 73.75 | 72.29 | 56.31 | 57.24 |
|
|
|
|
| 17 |
| Laya · English | 0.421B | 56.54 | 35.33 | 51.41 | 63.75 | 53.06 | 51.03 |
|
| 18 |
| Laya · Multilingual | 0.322B | 47.25 | 38.92 | 50.78 | 57.29 | 47.13 | 47.19 |
|
| 19 |
| Jev | — | 79.10 | 66.38 | 94.53 | 89.79 | 87.19 | 81.05 |
|
| 20 |
|
| 21 |
-
Accuracy (%).
|
| 22 |
|
| 23 |
## Scope and weighting
|
| 24 |
|
| 25 |
The overall score weights **Decisions 30%, Composition 25%, Reading 15%, Inference 15%, Transfer 15%**. The first four panels contain 880, 880, 480 and 480 questions and retain their original family/source weights. Transfer is micro-accuracy over 1,046 clean knowable questions from the frozen upstream transfer test; 110 missing-evidence questions and 108 variants remain separate diagnostics. No latency, calibration error or consistency score is averaged into accuracy.
|
| 26 |
|
|
|
|
|
|
|
| 27 |
These are **outcome-informed product-priority weights, chosen after observing benchmark results**. The data are observed regression tests, not a fresh blind test. Reweighting is not a training improvement. [Weight sensitivity](SENSITIVITY.md) retains the prior weighting and original four-panel comparison for the same model weights. Training, checkpoint selection and calibration do not use these test labels.
|
| 28 |
|
| 29 |
## Full results
|
|
@@ -32,6 +36,6 @@ These are **outcome-informed product-priority weights, chosen after observing be
|
|
| 32 |
|
| 33 |
## Model and API scope
|
| 34 |
|
| 35 |
-
Decision models return typed Choice, Noul and Score answers in the SystemOne format. Encoder and decoder references are evaluated through their published native interfaces. Untuned Qwen models use the frozen letter-logit readout, with no generated-text parsing. Kev uses the matched BF16 backbone with FP32 head and shipped temperature; date-fact injection is disabled. This differs from the authors’ default FP32 reproduction. Laya English and multilingual are measured separately. Jev is a recorded hosted-service snapshot.
|
| 36 |
|
| 37 |
The tests assess decisions from the supplied state and fixed choices. They do not establish live fact retrieval, universally superior reasoning or cross-hardware speed rankings. [Immutable identities and evaluation provenance](metrics/evaluation-provenance.json).
|
|
|
|
| 1 |
# Evaluation
|
| 2 |
|
| 3 |
+
The comparison covers **3,766 scored decisions across 54 tasks** and all 15 displayed models. All models use the same requested examples, candidate order and scoring rules. Each model keeps its native admission, truncation and probability semantics.
|
| 4 |
|
| 5 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 6 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 9B | **84.10** | **51.75** | 89.69 | **91.46** | 77.34 | **76.94** |
|
| 8 |
+
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
|
|
|
| 9 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 10 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 11 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
| 12 |
| Decider | 2B | 64.01 | 46.58 | 92.03 | 84.38 | 69.31 | 67.71 |
|
| 13 |
| Qwen3.5-4B | 4B | 69.89 | 43.33 | 87.97 | 79.79 | 68.83 | 67.29 |
|
| 14 |
+
| Sol-2B | 2B | 73.75 | 46.08 | 76.56 | 84.17 | 57.07 | 66.32 |
|
| 15 |
+
| Eos-0.8B | 0.8B | 65.94 | 46.04 | 70.31 | 81.67 | 52.01 | 61.89 |
|
| 16 |
| Kev-0.8B | 0.8B | 60.14 | 42.29 | 67.81 | 68.75 | 61.19 | 58.28 |
|
| 17 |
| Qwen3.5-2B | 2B | 57.12 | 39.00 | 73.75 | 72.29 | 56.31 | 57.24 |
|
| 18 |
+
| Kai-0.6B | 0.6B | 57.96 | 40.83 | 54.69 | 69.79 | 48.37 | 53.52 |
|
| 19 |
| Laya · English | 0.421B | 56.54 | 35.33 | 51.41 | 63.75 | 53.06 | 51.03 |
|
| 20 |
| Laya · Multilingual | 0.322B | 47.25 | 38.92 | 50.78 | 57.29 | 47.13 | 47.19 |
|
| 21 |
| Jev | — | 79.10 | 66.38 | 94.53 | 89.79 | 87.19 | 81.05 |
|
| 22 |
|
| 23 |
+
Accuracy (%). Open model rows are sorted by overall score; Jev is the frontier reference at the end. Bold marks Decision-family cells strictly above every external open or untuned reference in that metric; Jev and the other Decision models are excluded from that threshold.
|
| 24 |
|
| 25 |
## Scope and weighting
|
| 26 |
|
| 27 |
The overall score weights **Decisions 30%, Composition 25%, Reading 15%, Inference 15%, Transfer 15%**. The first four panels contain 880, 880, 480 and 480 questions and retain their original family/source weights. Transfer is micro-accuracy over 1,046 clean knowable questions from the frozen upstream transfer test; 110 missing-evidence questions and 108 variants remain separate diagnostics. No latency, calibration error or consistency score is averaged into accuracy.
|
| 28 |
|
| 29 |
+
Reported benchmarks cover English and Chinese; these are evaluation languages, not a restriction on accepted input languages.
|
| 30 |
+
|
| 31 |
These are **outcome-informed product-priority weights, chosen after observing benchmark results**. The data are observed regression tests, not a fresh blind test. Reweighting is not a training improvement. [Weight sensitivity](SENSITIVITY.md) retains the prior weighting and original four-panel comparison for the same model weights. Training, checkpoint selection and calibration do not use these test labels.
|
| 32 |
|
| 33 |
## Full results
|
|
|
|
| 36 |
|
| 37 |
## Model and API scope
|
| 38 |
|
| 39 |
+
Decision models return typed Choice, Noul and Score answers in the SystemOne format. Encoder and decoder references are evaluated through their published native interfaces. Untuned Qwen models use the frozen letter-logit readout, with no generated-text parsing. Kev uses the matched BF16 backbone with FP32 head and shipped temperature; date-fact injection is disabled. This differs from the authors’ default FP32 reproduction. Laya English and multilingual are measured separately. Kai-0.6B and Eos-0.8B use their independently verified published native interfaces on the same 54 tasks and requested denominators. Jev is a recorded hosted-service snapshot.
|
| 40 |
|
| 41 |
The tests assess decisions from the supplied state and fixed choices. They do not establish live fact retrieval, universally superior reasoning or cross-hardware speed rankings. [Immutable identities and evaluation provenance](metrics/evaluation-provenance.json).
|
MATERIALS.json
CHANGED
|
@@ -1,20 +1,26 @@
|
|
| 1 |
{
|
| 2 |
-
"
|
| 3 |
-
"
|
|
|
|
|
|
|
|
|
|
| 4 |
"files": {
|
| 5 |
-
"
|
| 6 |
-
"
|
| 7 |
-
"
|
| 8 |
-
"
|
| 9 |
-
"SENSITIVITY.md": "
|
| 10 |
-
"
|
| 11 |
-
"
|
| 12 |
-
"
|
| 13 |
-
"assets/decision-
|
| 14 |
-
"assets/decision-
|
| 15 |
-
"assets/decision-
|
| 16 |
-
"assets/decision-
|
| 17 |
-
"assets/decision-
|
| 18 |
-
"assets/decision-
|
|
|
|
|
|
|
|
|
|
| 19 |
}
|
| 20 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"kind": "latest-public-comparison-documents",
|
| 3 |
+
"public_models": 15,
|
| 4 |
+
"scored_questions": 3766,
|
| 5 |
+
"tasks": 54,
|
| 6 |
+
"source_statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc",
|
| 7 |
"files": {
|
| 8 |
+
".gitattributes": "8ec513d4464879c383554c84c1acff60508f0748470daa16f475d05ac804e2ba",
|
| 9 |
+
"DIAGNOSTICS.md": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73",
|
| 10 |
+
"EVALUATION.md": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e",
|
| 11 |
+
"README.md": "cc5fdd26eb3b5b0e88107b4b84fc3e12fc96d30ac9e8fb35cddc32a92df7f4ab",
|
| 12 |
+
"SENSITIVITY.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
|
| 13 |
+
"TASKS.md": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476",
|
| 14 |
+
"USAGE.md": "398c56cbb43047629551827fc5c0ba24f4dac2d7b01a5e6e3aeeb85b951706a0",
|
| 15 |
+
"WEIGHTING.md": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4",
|
| 16 |
+
"assets/decision-matrix.pdf": "a7e091eb7b4df6e3dfda49b58be3ba046c3e471a9954e3ae6c023e6732dd4bcd",
|
| 17 |
+
"assets/decision-matrix.png": "64e496d7692fe5404cff4664ab3fe93bec7396e19c31e7df3fdb097d50cc9e77",
|
| 18 |
+
"assets/decision-matrix.svg": "5b83c10c82f0a7f0c23278fd91bde9fbfb0710e76ec0e88dc8843394dabea6fc",
|
| 19 |
+
"assets/decision-ranking.pdf": "118307de70f311bc6a0787610d76297fccbfd10cadd916ca386f765df084b060",
|
| 20 |
+
"assets/decision-ranking.png": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6",
|
| 21 |
+
"assets/decision-ranking.svg": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62",
|
| 22 |
+
"assets/decision-sol-2b-header.png": "9616b7828774561692b642236d74b72de548e2dfc2ae83f1cae2b70bb3b69b08",
|
| 23 |
+
"metrics/benchmark.json": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56",
|
| 24 |
+
"metrics/evaluation-provenance.json": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5"
|
| 25 |
}
|
| 26 |
}
|
README.md
CHANGED
|
@@ -1,11 +1,9 @@
|
|
| 1 |
---
|
| 2 |
license: apache-2.0
|
| 3 |
-
language:
|
| 4 |
-
- en
|
| 5 |
-
- zh
|
| 6 |
base_model: Qwen/Qwen3.5-2B
|
| 7 |
base_model_relation: finetune
|
| 8 |
tags:
|
|
|
|
| 9 |
- decision-model
|
| 10 |
- classification
|
| 11 |
- qwen3_5
|
|
@@ -14,13 +12,15 @@ tags:
|
|
| 14 |
- rocm
|
| 15 |
---
|
| 16 |
|
| 17 |
-
|
|
|
|
|
|
|
|
|
|
| 18 |
|
| 19 |
*Sol, Latin for sun.*
|
| 20 |
|
| 21 |
-
|
| 22 |
|
| 23 |
-
**1.884B parameters · 16K complete-question budget · English / Chinese evaluated · Apache 2.0**
|
| 24 |
|
| 25 |
[Decision family](https://huggingface.co/collections/llm-semantic-router/decision-10-6ab12177bd0002394d8409f9)
|
| 26 |
|
|
@@ -32,20 +32,22 @@ tags:
|
|
| 32 |
|
| 33 |
## Measured capability
|
| 34 |
|
| 35 |
-
**66.32%
|
| 36 |
|
| 37 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 38 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 39 |
-
| Sol | 2B | 73.75 | 46.08 | 76.56 | 84.17 | 57.07 | 66.32 |
|
| 40 |
-
| Lux | 9B | **
|
| 41 |
-
| Nox | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 42 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 43 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 44 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
| 45 |
| Decider | 2B | 64.01 | 46.58 | 92.03 | 84.38 | 69.31 | 67.71 |
|
| 46 |
| Qwen3.5-4B | 4B | 69.89 | 43.33 | 87.97 | 79.79 | 68.83 | 67.29 |
|
|
|
|
| 47 |
| Kev-0.8B | 0.8B | 60.14 | 42.29 | 67.81 | 68.75 | 61.19 | 58.28 |
|
| 48 |
| Qwen3.5-2B | 2B | 57.12 | 39.00 | 73.75 | 72.29 | 56.31 | 57.24 |
|
|
|
|
| 49 |
| Laya · English | 0.421B | 56.54 | 35.33 | 51.41 | 63.75 | 53.06 | 51.03 |
|
| 50 |
| Laya · Multilingual | 0.322B | 47.25 | 38.92 | 50.78 | 57.29 | 47.13 | 47.19 |
|
| 51 |
| Jev | — | 79.10 | 66.38 | 94.53 | 89.79 | 87.19 | 81.05 |
|
|
@@ -64,19 +66,63 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading *
|
|
| 64 |
|
| 65 |
Same inputs and physical AMD gfx942 GPU; 30 measured requests per point across three blocks. Python request latency includes tokenization and inference, excluding loading and network. [p95 and all three native types](QUESTION-SCALING.md).
|
| 66 |
|
| 67 |
-
##
|
| 68 |
|
| 69 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 70 |
|
| 71 |
```python
|
| 72 |
-
from
|
| 73 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 74 |
|
| 75 |
-
|
| 76 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 77 |
```
|
| 78 |
|
| 79 |
-
[
|
| 80 |
|
| 81 |
The complete state, question and candidates must fit 16,384 tokens; overflow is rejected. The bundled normalization profile loads automatically. AMD gfx942 is validated; CPU/MPS are unsupported and NVIDIA is unqualified. Use a fresh Python process when switching profiles.
|
| 82 |
|
|
|
|
| 1 |
---
|
| 2 |
license: apache-2.0
|
|
|
|
|
|
|
|
|
|
| 3 |
base_model: Qwen/Qwen3.5-2B
|
| 4 |
base_model_relation: finetune
|
| 5 |
tags:
|
| 6 |
+
- multilingual
|
| 7 |
- decision-model
|
| 8 |
- classification
|
| 9 |
- qwen3_5
|
|
|
|
| 12 |
- rocm
|
| 13 |
---
|
| 14 |
|
| 15 |
+

|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
# Decision-1.0-Sol-2B
|
| 19 |
|
| 20 |
*Sol, Latin for sun.*
|
| 21 |
|
| 22 |
+
Give Sol a state, questions and possible answers. It returns decisions and probabilities with labels defined at runtime.
|
| 23 |
|
|
|
|
| 24 |
|
| 25 |
[Decision family](https://huggingface.co/collections/llm-semantic-router/decision-10-6ab12177bd0002394d8409f9)
|
| 26 |
|
|
|
|
| 32 |
|
| 33 |
## Measured capability
|
| 34 |
|
| 35 |
+
**66.32% weighted accuracy** across 3,766 decisions and 54 tasks. Compare decision, reading and transfer capabilities in the complete results below.
|
| 36 |
|
| 37 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 38 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 39 |
+
| Sol-2B | 2B | 73.75 | 46.08 | 76.56 | 84.17 | 57.07 | 66.32 |
|
| 40 |
+
| Lux-9B | 9B | **84.10** | **51.75** | 89.69 | **91.46** | 77.34 | **76.94** |
|
| 41 |
+
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 42 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 43 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 44 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
| 45 |
| Decider | 2B | 64.01 | 46.58 | 92.03 | 84.38 | 69.31 | 67.71 |
|
| 46 |
| Qwen3.5-4B | 4B | 69.89 | 43.33 | 87.97 | 79.79 | 68.83 | 67.29 |
|
| 47 |
+
| Eos-0.8B | 0.8B | 65.94 | 46.04 | 70.31 | 81.67 | 52.01 | 61.89 |
|
| 48 |
| Kev-0.8B | 0.8B | 60.14 | 42.29 | 67.81 | 68.75 | 61.19 | 58.28 |
|
| 49 |
| Qwen3.5-2B | 2B | 57.12 | 39.00 | 73.75 | 72.29 | 56.31 | 57.24 |
|
| 50 |
+
| Kai-0.6B | 0.6B | 57.96 | 40.83 | 54.69 | 69.79 | 48.37 | 53.52 |
|
| 51 |
| Laya · English | 0.421B | 56.54 | 35.33 | 51.41 | 63.75 | 53.06 | 51.03 |
|
| 52 |
| Laya · Multilingual | 0.322B | 47.25 | 38.92 | 50.78 | 57.29 | 47.13 | 47.19 |
|
| 53 |
| Jev | — | 79.10 | 66.38 | 94.53 | 89.79 | 87.19 | 81.05 |
|
|
|
|
| 66 |
|
| 67 |
Same inputs and physical AMD gfx942 GPU; 30 measured requests per point across three blocks. Python request latency includes tokenization and inference, excluding loading and network. [p95 and all three native types](QUESTION-SCALING.md).
|
| 68 |
|
| 69 |
+
## Use Sol-2B
|
| 70 |
|
| 71 |
+
Use the [official TypeSafe Python SDK](https://docs.typesafe.ai/sdk/python/usage) with your SystemOne-compatible endpoint, configured to serve `Decision-1.0-Sol-2B`. Replace the example URL and API key with your own.
|
| 72 |
+
|
| 73 |
+
```bash
|
| 74 |
+
pip install typesafe-sdk
|
| 75 |
+
```
|
| 76 |
|
| 77 |
```python
|
| 78 |
+
from typesafe_sdk import Choice, Noul, TypeSafeClient
|
| 79 |
+
|
| 80 |
+
with TypeSafeClient(
|
| 81 |
+
api_key="YOUR_ENDPOINT_API_KEY",
|
| 82 |
+
base_url="https://your-decision-endpoint.example",
|
| 83 |
+
model="Decision-1.0-Sol-2B",
|
| 84 |
+
) as client:
|
| 85 |
+
result = client.system_one(
|
| 86 |
+
state="Customer reports a duplicate charge and asks for a refund.",
|
| 87 |
+
questions={
|
| 88 |
+
"route": Choice(
|
| 89 |
+
instructions="Which team should handle this request?",
|
| 90 |
+
criteria={"billing": "Payments and refunds", "technical": "Product faults"},
|
| 91 |
+
),
|
| 92 |
+
"refund_requested": Noul(instructions="Did the customer request a refund?"),
|
| 93 |
+
},
|
| 94 |
+
)
|
| 95 |
+
print(result.choices["route"].choice)
|
| 96 |
+
print(result.nouls["refund_requested"].noul)
|
| 97 |
+
```
|
| 98 |
|
| 99 |
+
The same request with curl:
|
| 100 |
+
|
| 101 |
+
```bash
|
| 102 |
+
curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
|
| 103 |
+
-H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
|
| 104 |
+
-H 'Content-Type: application/json' \
|
| 105 |
+
--data-raw '{
|
| 106 |
+
"model": "Decision-1.0-Sol-2B",
|
| 107 |
+
"state": "Customer reports a duplicate charge and asks for a refund.",
|
| 108 |
+
"questions": {
|
| 109 |
+
"route": {
|
| 110 |
+
"type": "choice",
|
| 111 |
+
"instructions": "Which team should handle this request?",
|
| 112 |
+
"criteria": {
|
| 113 |
+
"billing": "Payments and refunds",
|
| 114 |
+
"technical": "Product faults"
|
| 115 |
+
}
|
| 116 |
+
},
|
| 117 |
+
"refund_requested": {
|
| 118 |
+
"type": "noul",
|
| 119 |
+
"instructions": "Did the customer request a refund?"
|
| 120 |
+
}
|
| 121 |
+
}
|
| 122 |
+
}'
|
| 123 |
```
|
| 124 |
|
| 125 |
+
[Typed request and response guide](USAGE.md) · [Model runtime requirements](RUNTIME.md)
|
| 126 |
|
| 127 |
The complete state, question and candidates must fit 16,384 tokens; overflow is rejected. The bundled normalization profile loads automatically. AMD gfx942 is validated; CPU/MPS are unsupported and NVIDIA is unqualified. Use a fresh Python process when switching profiles.
|
| 128 |
|
SENSITIVITY.md
CHANGED
|
@@ -1,17 +1,21 @@
|
|
| 1 |
-
|
| 2 |
|
| 3 |
-
|
|
|
|
|
|
|
| 4 |
|---|---:|---:|---:|
|
| 5 |
-
| Lux | 76.
|
| 6 |
-
| Nox | 73.09 | 72.42 | 75.03 |
|
| 7 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 8 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
| 9 |
| Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
|
| 10 |
| Decider | 67.71 | 67.97 | 71.75 |
|
| 11 |
| Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
|
| 12 |
-
| Sol | 66.32 | 65.48 | 70.14 |
|
|
|
|
| 13 |
| Kev-0.8B | 58.28 | 58.33 | 59.75 |
|
| 14 |
| Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
|
|
|
|
| 15 |
| Laya · English | 51.03 | 50.85 | 51.76 |
|
| 16 |
| Laya · Multilingual | 47.19 | 47.18 | 48.56 |
|
| 17 |
| Jev | 81.05 | 81.45 | 82.45 |
|
|
|
|
| 1 |
+
# Weight sensitivity
|
| 2 |
|
| 3 |
+
The current product-priority weights were chosen after observing results. This comparison holds every model and prediction fixed; reweighting is not a training improvement.
|
| 4 |
+
|
| 5 |
+
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 76.94 | 76.60 | 79.25 |
|
| 8 |
+
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
| 11 |
| Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
|
| 12 |
| Decider | 67.71 | 67.97 | 71.75 |
|
| 13 |
| Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
|
| 14 |
+
| Sol-2B | 66.32 | 65.48 | 70.14 |
|
| 15 |
+
| Eos-0.8B | 61.89 | 61.19 | 65.99 |
|
| 16 |
| Kev-0.8B | 58.28 | 58.33 | 59.75 |
|
| 17 |
| Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
|
| 18 |
+
| Kai-0.6B | 53.52 | 53.05 | 55.82 |
|
| 19 |
| Laya · English | 51.03 | 50.85 | 51.76 |
|
| 20 |
| Laya · Multilingual | 47.19 | 47.18 | 48.56 |
|
| 21 |
| Jev | 81.05 | 81.45 | 82.45 |
|
TASKS.md
CHANGED
|
@@ -5,93 +5,93 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 5 |
<details>
|
| 6 |
<summary>Decisions · 10 tasks</summary>
|
| 7 |
|
| 8 |
-
| Task | n | Lux | Nox | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol | Kev-0.8B | Qwen3.5-2B | Laya · English | Laya · Multilingual | Jev |
|
| 9 |
-
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 10 |
-
| News classification | 128 | 88.28 | 85.16 | 87.50 | 87.50 | 85.94 | 86.72 | 84.38 | 83.59 | 86.72 | 80.47 | 91.41 | 89.84 | 85.16 |
|
| 11 |
-
| Boolean constraints | 64 |
|
| 12 |
-
| Entity classification | 112 | 96.43 | 96.43 | 98.21 | 98.21 | 96.43 | 98.21 | 97.32 | 95.54 | 96.43 | 92.86 | 83.93 | 57.14 | 96.43 |
|
| 13 |
-
| Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 81.25 | 81.25 | 100.00 |
|
| 14 |
-
| Evidence placement | 96 |
|
| 15 |
-
| Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 59.38 | 81.25 | 18.75 | 12.50 | 100.00 |
|
| 16 |
-
| Relation composition | 96 | 61.46 | 52.08 | 51.04 | 62.50 | 36.46 | 54.17 | 51.04 | 37.50 | 44.79 | 48.96 | 25.00 | 35.42 | 56.25 |
|
| 17 |
-
| Scoped evidence | 96 | **
|
| 18 |
-
| State tracking | 96 |
|
| 19 |
-
| In / out of menu | 64 | **
|
| 20 |
|
| 21 |
</details>
|
| 22 |
|
| 23 |
<details>
|
| 24 |
<summary>Composition · 10 tasks</summary>
|
| 25 |
|
| 26 |
-
| Task | n | Lux | Nox | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol | Kev-0.8B | Qwen3.5-2B | Laya · English | Laya · Multilingual | Jev |
|
| 27 |
-
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 28 |
-
| Record identity | 80 | **
|
| 29 |
-
| Capacity assignment | 80 |
|
| 30 |
-
| Constraint assignment | 80 | **
|
| 31 |
-
| Intent routing · EN | 120 | 90.
|
| 32 |
-
| Intent routing · ZH | 120 |
|
| 33 |
-
| Multiset reconciliation | 80 | 26.25 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 25.00 | 27.50 | 26.25 | 32.50 | 58.75 |
|
| 34 |
-
| Ordered service loss | 80 |
|
| 35 |
-
| Conflicting rule closure | 80 |
|
| 36 |
-
| Temporal exclusion | 80 |
|
| 37 |
-
| Transaction recovery | 80 |
|
| 38 |
|
| 39 |
</details>
|
| 40 |
|
| 41 |
<details>
|
| 42 |
<summary>Reading · 3 tasks</summary>
|
| 43 |
|
| 44 |
-
| Task | n | Lux | Nox | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol | Kev-0.8B | Qwen3.5-2B | Laya · English | Laya · Multilingual | Jev |
|
| 45 |
-
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 46 |
-
| Yes / no reading | 160 |
|
| 47 |
-
| Reading · EN | 160 |
|
| 48 |
-
| Reading · ZH | 160 | 88.75 | 70.62 | 81.25 | 74.38 | 91.88 | 91.25 | 91.25 | 70.00 | 58.75 | 81.25 | 28.75 | 28.12 | 96.25 |
|
| 49 |
|
| 50 |
</details>
|
| 51 |
|
| 52 |
<details>
|
| 53 |
<summary>Inference · 4 tasks</summary>
|
| 54 |
|
| 55 |
-
| Task | n | Lux | Nox | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol | Kev-0.8B | Qwen3.5-2B | Laya · English | Laya · Multilingual | Jev |
|
| 56 |
-
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 57 |
-
| Contextual reasoning | 120 | **83.33** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 42.50 | 56.67 | 30.00 | 20.00 | 86.67 |
|
| 58 |
-
| Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 66.67 | 82.50 | 57.50 | 55.00 | 90.83 |
|
| 59 |
-
| Textual entailment | 120 |
|
| 60 |
-
| Scientific inference | 120 | 96.67 | 96.67 | 96.67 | 96.67 | 98.33 | 95.83 | 96.67 | 95.83 | 88.33 | 85.83 | 95.00 | 81.67 | 99.17 |
|
| 61 |
|
| 62 |
</details>
|
| 63 |
|
| 64 |
<details>
|
| 65 |
<summary>Transfer · 27 tasks</summary>
|
| 66 |
|
| 67 |
-
| Task | n | Lux | Nox | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol | Kev-0.8B | Qwen3.5-2B | Laya · English | Laya · Multilingual | Jev |
|
| 68 |
-
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 69 |
-
| Buried emotion | 20 |
|
| 70 |
-
| Buried paraphrase | 20 |
|
| 71 |
-
| Buried entailment | 20 | 90.00 | 90.00 | 95.00 | 90.00 | 95.00 | 90.00 | 75.00 | 75.00 | 85.00 | 65.00 | 60.00 | 55.00 | 95.00 |
|
| 72 |
-
| Buried offensive-language detection | 20 |
|
| 73 |
-
| Combined policy conditions | 32 | **
|
| 74 |
-
| Policy exceptions | 32 |
|
| 75 |
-
| Policy negation | 32 | 87.50 | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 71.88 | 46.88 | 53.12 | 50.00 | 90.62 |
|
| 76 |
-
| Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 97.50 | 62.50 | 67.50 | 50.00 | 100.00 |
|
| 77 |
-
| Deadline contrast | 40 |
|
| 78 |
-
| Emotion | 80 |
|
| 79 |
-
| MMLU | 80 |
|
| 80 |
-
| MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 22.50 | 28.00 | 11.00 | 11.50 | 84.00 |
|
| 81 |
-
| Paraphrase | 80 | **
|
| 82 |
-
| Question entailment | 80 |
|
| 83 |
-
| Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 96.25 | 96.25 | 90.00 | 72.50 | 100.00 |
|
| 84 |
-
| Offensive-language detection | 80 | 73.75 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 72.50 | 83.75 | 81.25 | 82.50 | 76.25 |
|
| 85 |
-
| Evidence control · age eligibility | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 90.00 | 50.00 | 100.00 |
|
| 86 |
-
| Evidence control · authorization | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 80.00 | 100.00 | 50.00 | 100.00 | 50.00 | 70.00 | 50.00 | 100.00 |
|
| 87 |
-
| Evidence control · deadline | 10 |
|
| 88 |
-
| Evidence control · late fee | 10 |
|
| 89 |
-
| Evidence control · quantity limit | 10 | 100.00 | 80.00 | 100.00 | 100.00 | 70.00 | 80.00 | 70.00 | 70.00 | 100.00 | 20.00 | 60.00 | 30.00 | 100.00 |
|
| 90 |
-
| Evidence control · return window | 10 | 60.00 | 40.00 | 70.00 | 70.00 | 60.00 | 60.00 | 40.00 | 40.00 | 80.00 | 50.00 | 60.00 | 40.00 | 50.00 |
|
| 91 |
-
| Evidence control · shipping delay | 10 | 90.00 | 40.00 | 90.00 | 80.00 | 80.00 | 80.00 | 60.00 | 80.00 | 100.00 | 30.00 | 30.00 | 30.00 | 90.00 |
|
| 92 |
-
| Evidence control · sla response | 10 | 70.00 | 40.00 | 100.00 | 100.00 | 50.00 | 30.00 | 50.00 | 40.00 | 100.00 | 50.00 | 50.00 | 30.00 | 100.00 |
|
| 93 |
-
| Evidence control · spend threshold | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 90.00 | 100.00 | 100.00 | 50.00 | 50.00 | 100.00 |
|
| 94 |
-
| Evidence control · volume discount | 10 | 100.00 | 80.00 | 100.00 | 100.00 | 100.00 | 90.00 | 100.00 | 90.00 | 100.00 | 20.00 | 20.00 | 30.00 | 100.00 |
|
| 95 |
-
| Evidence control · warranty claim | 10 |
|
| 96 |
|
| 97 |
</details>
|
|
|
|
| 5 |
<details>
|
| 6 |
<summary>Decisions · 10 tasks</summary>
|
| 7 |
|
| 8 |
+
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 9 |
+
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 10 |
+
| News classification | 128 | 88.28 | 85.16 | 87.50 | 87.50 | 85.94 | 86.72 | 84.38 | 83.59 | 82.03 | 86.72 | 80.47 | 76.56 | 91.41 | 89.84 | 85.16 |
|
| 11 |
+
| Boolean constraints | 64 | **100.00** | 93.75 | 95.31 | 71.88 | 98.44 | 71.88 | 62.50 | 50.00 | 82.81 | 81.25 | 37.50 | 53.12 | 43.75 | 39.06 | 100.00 |
|
| 12 |
+
| Entity classification | 112 | 96.43 | 96.43 | 98.21 | 98.21 | 96.43 | 98.21 | 97.32 | 95.54 | 91.96 | 96.43 | 92.86 | 84.82 | 83.93 | 57.14 | 96.43 |
|
| 13 |
+
| Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 89.06 | 81.25 | 81.25 | 100.00 |
|
| 14 |
+
| Evidence placement | 96 | 64.58 | **100.00** | 50.00 | 41.67 | 79.17 | 8.33 | 72.92 | **98.96** | 19.79 | 33.33 | 9.38 | **100.00** | 95.83 | 54.17 | 30.21 |
|
| 15 |
+
| Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 84.38 | 59.38 | 81.25 | 6.25 | 18.75 | 12.50 | 100.00 |
|
| 16 |
+
| Relation composition | 96 | 61.46 | 52.08 | 51.04 | 62.50 | 36.46 | 54.17 | 51.04 | 37.50 | **64.58** | 44.79 | 48.96 | 40.62 | 25.00 | 35.42 | 56.25 |
|
| 17 |
+
| Scoped evidence | 96 | **97.92** | **78.12** | 65.62 | 47.92 | 55.21 | 47.92 | 51.04 | **79.17** | 25.00 | 10.42 | 36.46 | 25.00 | 37.50 | 36.46 | 89.58 |
|
| 18 |
+
| State tracking | 96 | 32.29 | 35.42 | 38.54 | 34.38 | 31.25 | 29.17 | 28.12 | 27.08 | 38.54 | 31.25 | 25.00 | 29.17 | 23.96 | 29.17 | 33.33 |
|
| 19 |
+
| In / out of menu | 64 | **100.00** | **100.00** | 84.38 | 75.00 | 71.88 | 59.38 | 60.94 | **100.00** | 70.31 | 57.81 | 62.50 | 75.00 | 64.06 | 37.50 | 100.00 |
|
| 20 |
|
| 21 |
</details>
|
| 22 |
|
| 23 |
<details>
|
| 24 |
<summary>Composition · 10 tasks</summary>
|
| 25 |
|
| 26 |
+
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 27 |
+
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 28 |
+
| Record identity | 80 | **63.75** | 53.75 | 45.00 | 50.00 | 57.50 | 56.25 | 50.00 | 50.00 | **60.00** | 50.00 | 46.25 | 51.25 | 47.50 | 50.00 | 76.25 |
|
| 29 |
+
| Capacity assignment | 80 | 62.50 | 46.25 | 50.00 | 50.00 | 50.00 | 51.25 | 50.00 | 47.50 | 50.00 | 50.00 | 50.00 | 50.00 | 63.75 | 50.00 | 68.75 |
|
| 30 |
+
| Constraint assignment | 80 | **37.50** | **41.25** | 27.50 | 32.50 | 25.00 | 23.75 | 23.75 | 26.25 | 25.00 | 20.00 | 21.25 | 26.25 | 22.50 | 27.50 | 55.00 |
|
| 31 |
+
| Intent routing · EN | 120 | 90.00 | 91.67 | 86.67 | 87.50 | 88.33 | 90.00 | 91.67 | 90.83 | 91.67 | 83.33 | 70.83 | 81.67 | 74.17 | 70.83 | 91.67 |
|
| 32 |
+
| Intent routing · ZH | 120 | 87.50 | 87.50 | 85.83 | 86.67 | 86.67 | 88.33 | 86.67 | 87.50 | 85.00 | 85.83 | 74.17 | 79.17 | 44.17 | 73.33 | 88.33 |
|
| 33 |
+
| Multiset reconciliation | 80 | 26.25 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 26.25 | 25.00 | 27.50 | 25.00 | 26.25 | 32.50 | 58.75 |
|
| 34 |
+
| Ordered service loss | 80 | 23.75 | **30.00** | 25.00 | 20.00 | 20.00 | 22.50 | 21.25 | 25.00 | 20.00 | 27.50 | 20.00 | 20.00 | 20.00 | 17.50 | 42.50 |
|
| 35 |
+
| Conflicting rule closure | 80 | 31.25 | 33.75 | 27.50 | 33.75 | 27.50 | 26.25 | 25.00 | 25.00 | 27.50 | 27.50 | 27.50 | 25.00 | 18.75 | 20.00 | 66.25 |
|
| 36 |
+
| Temporal exclusion | 80 | 45.00 | 42.50 | 28.75 | 45.00 | 36.25 | 42.50 | 31.25 | 41.25 | 32.50 | 21.25 | 28.75 | 17.50 | 12.50 | 28.75 | 41.25 |
|
| 37 |
+
| Transaction recovery | 80 | 50.00 | **62.50** | 43.75 | 51.25 | 35.00 | 41.25 | 26.25 | 38.75 | 42.50 | 32.50 | 23.75 | 32.50 | 23.75 | 18.75 | 75.00 |
|
| 38 |
|
| 39 |
</details>
|
| 40 |
|
| 41 |
<details>
|
| 42 |
<summary>Reading · 3 tasks</summary>
|
| 43 |
|
| 44 |
+
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 45 |
+
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 46 |
+
| Yes / no reading | 160 | 88.75 | 86.25 | 91.25 | 88.75 | 84.38 | 91.88 | 83.75 | 83.12 | 75.62 | 74.38 | 65.00 | 74.38 | 69.38 | 69.38 | 92.50 |
|
| 47 |
+
| Reading · EN | 160 | 92.50 | 73.12 | 83.12 | 75.62 | 98.75 | 93.12 | 93.12 | 70.00 | 68.12 | 63.75 | 83.75 | 38.12 | 38.12 | 36.25 | 96.88 |
|
| 48 |
+
| Reading · ZH | 160 | 88.75 | 70.62 | 81.25 | 74.38 | 91.88 | 91.25 | 91.25 | 70.00 | 61.88 | 58.75 | 81.25 | 31.87 | 28.75 | 28.12 | 96.25 |
|
| 49 |
|
| 50 |
</details>
|
| 51 |
|
| 52 |
<details>
|
| 53 |
<summary>Inference · 4 tasks</summary>
|
| 54 |
|
| 55 |
+
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 56 |
+
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 57 |
+
| Contextual reasoning | 120 | **83.33** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 70.00 | 42.50 | 56.67 | 46.67 | 30.00 | 20.00 | 86.67 |
|
| 58 |
+
| Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 81.67 | 66.67 | 82.50 | 55.83 | 57.50 | 55.00 | 90.83 |
|
| 59 |
+
| Textual entailment | 120 | **92.50** | 89.17 | 90.83 | 89.17 | 81.67 | 90.83 | 80.00 | 88.33 | 84.17 | 77.50 | 64.17 | 78.33 | 72.50 | 72.50 | 82.50 |
|
| 60 |
+
| Scientific inference | 120 | 96.67 | 96.67 | 96.67 | 96.67 | 98.33 | 95.83 | 96.67 | 95.83 | 90.83 | 88.33 | 85.83 | 98.33 | 95.00 | 81.67 | 99.17 |
|
| 61 |
|
| 62 |
</details>
|
| 63 |
|
| 64 |
<details>
|
| 65 |
<summary>Transfer · 27 tasks</summary>
|
| 66 |
|
| 67 |
+
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 68 |
+
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 69 |
+
| Buried emotion | 20 | 45.00 | 25.00 | 65.00 | 55.00 | 40.00 | 85.00 | 45.00 | 20.00 | 40.00 | 40.00 | 55.00 | 45.00 | 60.00 | 35.00 | 70.00 |
|
| 70 |
+
| Buried paraphrase | 20 | 80.00 | 75.00 | 80.00 | 75.00 | 80.00 | 75.00 | 75.00 | 70.00 | 70.00 | 60.00 | 70.00 | 35.00 | 55.00 | 70.00 | 95.00 |
|
| 71 |
+
| Buried entailment | 20 | 90.00 | 90.00 | 95.00 | 90.00 | 95.00 | 90.00 | 75.00 | 75.00 | 85.00 | 85.00 | 65.00 | 55.00 | 60.00 | 55.00 | 95.00 |
|
| 72 |
+
| Buried offensive-language detection | 20 | 40.00 | 45.00 | 80.00 | 65.00 | 45.00 | 80.00 | 40.00 | 35.00 | 15.00 | 60.00 | 35.00 | 75.00 | 70.00 | 80.00 | 55.00 |
|
| 73 |
+
| Combined policy conditions | 32 | **96.88** | 75.00 | 68.75 | 78.12 | 53.12 | 53.12 | 59.38 | 78.12 | 37.50 | 40.62 | 62.50 | 62.50 | 62.50 | 43.75 | 96.88 |
|
| 74 |
+
| Policy exceptions | 32 | 90.62 | 96.88 | 87.50 | 96.88 | 59.38 | 56.25 | 75.00 | 71.88 | 53.12 | 65.62 | 40.62 | 53.12 | 43.75 | 37.50 | 100.00 |
|
| 75 |
+
| Policy negation | 32 | 87.50 | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 68.75 | 71.88 | 46.88 | 56.25 | 53.12 | 50.00 | 90.62 |
|
| 76 |
+
| Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 50.00 | 97.50 | 62.50 | 50.00 | 67.50 | 50.00 | 100.00 |
|
| 77 |
+
| Deadline contrast | 40 | **90.00** | 70.00 | 87.50 | 75.00 | 77.50 | 30.00 | 52.50 | 40.00 | 50.00 | 45.00 | 25.00 | 50.00 | 35.00 | 22.50 | 92.50 |
|
| 78 |
+
| Emotion | 80 | 60.00 | 60.00 | 67.50 | 65.00 | 52.50 | 86.25 | 65.00 | 22.50 | 45.00 | 60.00 | 67.50 | 53.75 | 62.50 | 56.25 | 67.50 |
|
| 79 |
+
| MMLU | 80 | 73.75 | 61.25 | 75.00 | 68.75 | 73.75 | 62.50 | 66.25 | 52.50 | 47.50 | 51.25 | 53.75 | 37.50 | 22.50 | 27.50 | 88.75 |
|
| 80 |
+
| MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 17.00 | 22.50 | 28.00 | 12.50 | 11.00 | 11.50 | 84.00 |
|
| 81 |
+
| Paraphrase | 80 | **90.00** | 85.00 | 81.25 | 81.25 | 88.75 | 77.50 | 83.75 | 80.00 | 71.25 | 58.75 | 70.00 | 47.50 | 86.25 | 75.00 | 87.50 |
|
| 82 |
+
| Question entailment | 80 | 91.25 | 91.25 | 95.00 | 91.25 | 90.00 | 87.50 | 87.50 | 83.75 | 81.25 | 81.25 | 63.75 | 71.25 | 80.00 | 73.75 | 91.25 |
|
| 83 |
+
| Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 97.50 | 96.25 | 96.25 | 90.00 | 90.00 | 72.50 | 100.00 |
|
| 84 |
+
| Offensive-language detection | 80 | 73.75 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 52.50 | 72.50 | 83.75 | 78.75 | 81.25 | 82.50 | 76.25 |
|
| 85 |
+
| Evidence control · age eligibility | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 50.00 | 90.00 | 50.00 | 100.00 |
|
| 86 |
+
| Evidence control · authorization | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 80.00 | 100.00 | 50.00 | 50.00 | 100.00 | 50.00 | 50.00 | 70.00 | 50.00 | 100.00 |
|
| 87 |
+
| Evidence control · deadline | 10 | **100.00** | 60.00 | 70.00 | 60.00 | 80.00 | 30.00 | 50.00 | 30.00 | 40.00 | 40.00 | 30.00 | 50.00 | 30.00 | 30.00 | 100.00 |
|
| 88 |
+
| Evidence control · late fee | 10 | 100.00 | 90.00 | 100.00 | 100.00 | 90.00 | 70.00 | 80.00 | 80.00 | 60.00 | 90.00 | 70.00 | 30.00 | 40.00 | 40.00 | 100.00 |
|
| 89 |
+
| Evidence control · quantity limit | 10 | 100.00 | 80.00 | 100.00 | 100.00 | 70.00 | 80.00 | 70.00 | 70.00 | 50.00 | 100.00 | 20.00 | 20.00 | 60.00 | 30.00 | 100.00 |
|
| 90 |
+
| Evidence control · return window | 10 | 60.00 | 40.00 | 70.00 | 70.00 | 60.00 | 60.00 | 40.00 | 40.00 | 50.00 | 80.00 | 50.00 | 50.00 | 60.00 | 40.00 | 50.00 |
|
| 91 |
+
| Evidence control · shipping delay | 10 | 90.00 | 40.00 | 90.00 | 80.00 | 80.00 | 80.00 | 60.00 | 80.00 | 20.00 | 100.00 | 30.00 | 20.00 | 30.00 | 30.00 | 90.00 |
|
| 92 |
+
| Evidence control · sla response | 10 | 70.00 | 40.00 | 100.00 | 100.00 | 50.00 | 30.00 | 50.00 | 40.00 | 30.00 | 100.00 | 50.00 | 30.00 | 50.00 | 30.00 | 100.00 |
|
| 93 |
+
| Evidence control · spend threshold | 10 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 90.00 | 70.00 | 100.00 | 100.00 | 50.00 | 50.00 | 50.00 | 100.00 |
|
| 94 |
+
| Evidence control · volume discount | 10 | 100.00 | 80.00 | 100.00 | 100.00 | 100.00 | 90.00 | 100.00 | 90.00 | 90.00 | 100.00 | 20.00 | 40.00 | 20.00 | 30.00 | 100.00 |
|
| 95 |
+
| Evidence control · warranty claim | 10 | 50.00 | 70.00 | 80.00 | 100.00 | 60.00 | 40.00 | 50.00 | 50.00 | 50.00 | 50.00 | 50.00 | 20.00 | 40.00 | 30.00 | 90.00 |
|
| 96 |
|
| 97 |
</details>
|
USAGE.md
CHANGED
|
@@ -1,66 +1,59 @@
|
|
| 1 |
-
#
|
| 2 |
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
Install from a downloaded model repository containing this `pyproject.toml` and `src/decision/`:
|
| 6 |
|
| 7 |
```bash
|
| 8 |
-
|
| 9 |
-
# Optional, only for resolving models through Hugging Face:
|
| 10 |
-
python -m pip install '.[hub]'
|
| 11 |
```
|
| 12 |
|
| 13 |
-
There is no requirement to install an identically named package from PyPI. These commands install this repository's wrapper. They deliberately do not replace your GPU PyTorch installation. The full inference environment is recorded in the model's `runtime.json`: the qualified run used PyTorch 2.12.0+git6bbd260 / ROCm 7.2.53211, Transformers 5.17.0, FLA 0.5.2, Triton 3.7.1, tokenizers 0.23.2 and safetensors 0.8.0. The recorded PyTorch build is not promised to exist on ordinary PyPI. Use the public digest-pinned build recipe in [RUNTIME.md](RUNTIME.md); installation of this lightweight wrapper alone is not installation of that GPU runtime.
|
| 14 |
-
|
| 15 |
-
Local loading is offline and requires no Hub client or credentials:
|
| 16 |
-
|
| 17 |
```python
|
| 18 |
-
from
|
| 19 |
-
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
"
|
| 27 |
-
|
| 28 |
-
"
|
| 29 |
-
|
| 30 |
-
"billing": "
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
}
|
| 34 |
-
|
| 35 |
-
)
|
| 36 |
-
print(result)
|
| 37 |
```
|
| 38 |
|
| 39 |
-
The saved model-card example runner includes Choice, Noul and Score together, compares its actual output with the direct frozen engine, and checks that an overflowing input raises an error:
|
| 40 |
-
|
| 41 |
```bash
|
| 42 |
-
|
| 43 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
```
|
| 45 |
|
| 46 |
-
The
|
| 47 |
-
|
| 48 |
-
## Interface
|
| 49 |
-
|
| 50 |
-
| Type | Input criteria | Answer |
|
| 51 |
-
|---|---|---|
|
| 52 |
-
| Choice | Ordered mapping of 2–255 external IDs to complete descriptions | `probabilities`, selected `choice` ID, `confidence` |
|
| 53 |
-
| Noul | Optional mapping containing only `false` / `true` descriptions | `noul`: P(true); a hard judgment uses `>= 0.5` |
|
| 54 |
-
| Score | Ordered list of 2–10 rubric descriptions | `probabilities` over string indices, expected level index `score`, `legend`, `confidence` |
|
| 55 |
-
|
| 56 |
-
Response shape is `{"model": name, "answers": {question_name: answer}, "usage": {"input_tokens": total, "scored_questions": count}}`. Choice ties select the earliest candidate in insertion order. Score returns the expected ordinal **index**, not an arbitrary supplied numeric value; this adapter does not implement a supplied-values extension. Noul 0.5 is interpreted as true. Confidence is `(K * max(p) - 1)/(K - 1)`, clipped to [0,1], and is not a claimed reproduction of Jev's confidence statistic. Shipped temperature calibration does not make every confidence value a correctness guarantee.
|
| 57 |
-
|
| 58 |
-
Question names are preserved as opaque bookkeeping IDs; candidate IDs and descriptions use the frozen renderer. Native objects are deterministically serialized with sorted JSON keys; strings preserve their contents. Do not interpret object/string field-order differences as identical token inputs. Non-finite or non-JSON inputs are rejected.
|
| 59 |
-
|
| 60 |
-
The bound is **16,384 tokens per complete question**, including its state, instructions, all candidates and readout suffix. Questions are separate sequences, grouped in fixed batches of eight; the state is repeated for each question and counted repeatedly in `usage.input_tokens`. All questions are encoded before any forward pass. If any exceeds the bound, the whole call raises `ValueError` without truncation or partial answers. A lower `max_length` can be chosen at load time; a higher limit is rejected. This native wrapper does not impose the Studio's separate 16-question UI limit.
|
| 61 |
-
|
| 62 |
-
## Runtime boundary
|
| 63 |
|
| 64 |
-
|
| 65 |
|
| 66 |
-
|
|
|
|
| 1 |
+
# Use Sol-2B
|
| 2 |
|
| 3 |
+
The examples below use the [official TypeSafe SDK](https://docs.typesafe.ai/sdk/python/usage) and the standard [SystemOne HTTP request](https://docs.typesafe.ai/api). Configure your endpoint to serve `Decision-1.0-Sol-2B`, then replace the example URL and API key. A Hugging Face model repository is a weights download, not an inference endpoint.
|
|
|
|
|
|
|
| 4 |
|
| 5 |
```bash
|
| 6 |
+
pip install typesafe-sdk
|
|
|
|
|
|
|
| 7 |
```
|
| 8 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 9 |
```python
|
| 10 |
+
from typesafe_sdk import Choice, Noul, TypeSafeClient
|
| 11 |
+
|
| 12 |
+
with TypeSafeClient(
|
| 13 |
+
api_key="YOUR_ENDPOINT_API_KEY",
|
| 14 |
+
base_url="https://your-decision-endpoint.example",
|
| 15 |
+
model="Decision-1.0-Sol-2B",
|
| 16 |
+
) as client:
|
| 17 |
+
result = client.system_one(
|
| 18 |
+
state="Customer reports a duplicate charge and asks for a refund.",
|
| 19 |
+
questions={
|
| 20 |
+
"route": Choice(
|
| 21 |
+
instructions="Which team should handle this request?",
|
| 22 |
+
criteria={"billing": "Payments and refunds", "technical": "Product faults"},
|
| 23 |
+
),
|
| 24 |
+
"refund_requested": Noul(instructions="Did the customer request a refund?"),
|
| 25 |
+
},
|
| 26 |
+
)
|
| 27 |
+
print(result.choices["route"].choice)
|
| 28 |
+
print(result.nouls["refund_requested"].noul)
|
| 29 |
```
|
| 30 |
|
|
|
|
|
|
|
| 31 |
```bash
|
| 32 |
+
curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
|
| 33 |
+
-H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
|
| 34 |
+
-H 'Content-Type: application/json' \
|
| 35 |
+
--data-raw '{
|
| 36 |
+
"model": "Decision-1.0-Sol-2B",
|
| 37 |
+
"state": "Customer reports a duplicate charge and asks for a refund.",
|
| 38 |
+
"questions": {
|
| 39 |
+
"route": {
|
| 40 |
+
"type": "choice",
|
| 41 |
+
"instructions": "Which team should handle this request?",
|
| 42 |
+
"criteria": {
|
| 43 |
+
"billing": "Payments and refunds",
|
| 44 |
+
"technical": "Product faults"
|
| 45 |
+
}
|
| 46 |
+
},
|
| 47 |
+
"refund_requested": {
|
| 48 |
+
"type": "noul",
|
| 49 |
+
"instructions": "Did the customer request a refund?"
|
| 50 |
+
}
|
| 51 |
+
}
|
| 52 |
+
}'
|
| 53 |
```
|
| 54 |
|
| 55 |
+
The state can be text or JSON-compatible structured data. Question IDs and Choice IDs are preserved in the response. Choice uses 2–255 options; Noul returns `noul`, the probability of a condition being true; Score uses 2–10 rubric descriptions ordered from index zero. A request can contain many questions.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
|
| 57 |
+
Choice returns `choice`, `probabilities` and `confidence`. Score returns an expected zero-based `score`, a probability distribution and `legend`. Local Decision confidence is normalized maximum probability, `(K × max(p) − 1)/(K − 1)`; it does not reproduce an unpublished provider confidence statistic. An HTTP integration must supply `usage.input_tokens` and `usage.output_tokens`; counting generated tokens as zero is appropriate for this non-generative model, not a claim about provider billing.
|
| 58 |
|
| 59 |
+
The native runtime processes independent complete questions in batches of eight. Each complete rendered question, including state, instructions and criteria, must fit 16,384 tokens; overflow is rejected. [Runtime and hardware requirements](RUNTIME.md).
|
WEIGHTING.md
CHANGED
|
@@ -1,17 +1,21 @@
|
|
| 1 |
-
|
| 2 |
|
| 3 |
-
|
|
|
|
|
|
|
| 4 |
|---|---:|---:|---:|
|
| 5 |
-
| Lux | 76.
|
| 6 |
-
| Nox | 73.09 | 72.42 | 75.03 |
|
| 7 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 8 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
| 9 |
| Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
|
| 10 |
| Decider | 67.71 | 67.97 | 71.75 |
|
| 11 |
| Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
|
| 12 |
-
| Sol | 66.32 | 65.48 | 70.14 |
|
|
|
|
| 13 |
| Kev-0.8B | 58.28 | 58.33 | 59.75 |
|
| 14 |
| Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
|
|
|
|
| 15 |
| Laya · English | 51.03 | 50.85 | 51.76 |
|
| 16 |
| Laya · Multilingual | 47.19 | 47.18 | 48.56 |
|
| 17 |
| Jev | 81.05 | 81.45 | 82.45 |
|
|
|
|
| 1 |
+
# Weight sensitivity
|
| 2 |
|
| 3 |
+
The current product-priority weights were chosen after observing results. This comparison holds every model and prediction fixed; reweighting is not a training improvement.
|
| 4 |
+
|
| 5 |
+
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 76.94 | 76.60 | 79.25 |
|
| 8 |
+
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
| 11 |
| Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
|
| 12 |
| Decider | 67.71 | 67.97 | 71.75 |
|
| 13 |
| Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
|
| 14 |
+
| Sol-2B | 66.32 | 65.48 | 70.14 |
|
| 15 |
+
| Eos-0.8B | 61.89 | 61.19 | 65.99 |
|
| 16 |
| Kev-0.8B | 58.28 | 58.33 | 59.75 |
|
| 17 |
| Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
|
| 18 |
+
| Kai-0.6B | 53.52 | 53.05 | 55.82 |
|
| 19 |
| Laya · English | 51.03 | 50.85 | 51.76 |
|
| 20 |
| Laya · Multilingual | 47.19 | 47.18 | 48.56 |
|
| 21 |
| Jev | 81.05 | 81.45 | 82.45 |
|
assets/decision-matrix.pdf
CHANGED
|
Binary files a/assets/decision-matrix.pdf and b/assets/decision-matrix.pdf differ
|
|
|
assets/decision-matrix.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
assets/decision-matrix.svg
CHANGED
|
|
|
|
assets/decision-ranking.pdf
CHANGED
|
Binary files a/assets/decision-ranking.pdf and b/assets/decision-ranking.pdf differ
|
|
|
assets/decision-ranking.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
assets/decision-ranking.svg
CHANGED
|
|
|
|
assets/decision-sol-2b-header.png
ADDED
|
Git LFS Details
|
metrics/benchmark.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
metrics/evaluation-provenance.json
CHANGED
|
@@ -1,676 +1,319 @@
|
|
| 1 |
{
|
| 2 |
"frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
|
| 3 |
-
"
|
| 4 |
-
"
|
| 5 |
-
"qualified_runtime": {
|
| 6 |
-
"python": "3.12.13",
|
| 7 |
-
"numpy": "2.3.5"
|
| 8 |
-
},
|
| 9 |
-
"source_sha256": "f2f72fe161944599b1114df29ff095786bf661cca59afd9d784b7ca6d1430c99",
|
| 10 |
-
"model_manifest": {
|
| 11 |
"Jev": {
|
| 12 |
-
"
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
"path": "eval/heldout/v3/official/normalized/core.jsonl",
|
| 20 |
-
"sha256": "54707354bfe5d96b5de56e6dc1b38cba0a610eea516b28fbff013b94f422b397",
|
| 21 |
-
"bytes": 368716
|
| 22 |
-
},
|
| 23 |
-
"v4": {
|
| 24 |
-
"path": "eval/heldout/v4/official/normalized/core.jsonl",
|
| 25 |
-
"sha256": "d6583f903e974d70747fae2944014828954dc4d7344ceddcbb2cd1b06fbfc799",
|
| 26 |
-
"bytes": 183671
|
| 27 |
-
},
|
| 28 |
-
"v5": {
|
| 29 |
-
"path": "eval/heldout/v5/official/normalized/core.jsonl",
|
| 30 |
-
"sha256": "f883164c4c83db8e42ebe03f7e165afbd2f81f2b541bcf8008a2cc06a60e1ee7",
|
| 31 |
-
"bytes": 200316
|
| 32 |
-
}
|
| 33 |
-
},
|
| 34 |
-
"transfer_rows": {
|
| 35 |
-
"path": "analysis/decision-benchmark-v3-official/scored/ROWS.json",
|
| 36 |
-
"sha256": "be29a7f3804ad84993ffb06f05ada7e2d8ed25317504c4fc17fe5dc306fd788c"
|
| 37 |
-
},
|
| 38 |
-
"evidence": [
|
| 39 |
-
{
|
| 40 |
-
"path": "eval/expanded-public-extension-v1/COMPARATOR-REGISTRY.json",
|
| 41 |
-
"sha256": "b80be4270721cbeeac9e28beb10c6f19f75c1cbf2553e38d3d59e3a42d56a91c"
|
| 42 |
-
},
|
| 43 |
-
{
|
| 44 |
-
"path": "analysis/decision-benchmark-v3-official/scored/REPORT.json",
|
| 45 |
-
"sha256": "2fdbaf7db1e4b877762e33a87b23a236a065eaa3078b14c2d7d579ad44f9ea24"
|
| 46 |
-
},
|
| 47 |
-
{
|
| 48 |
-
"path": "analysis/decision-benchmark-v3-official/INDEPENDENT-PROJECTION-REVIEW.json",
|
| 49 |
-
"sha256": "b9506539e2c4a863aef16f7dab0517c175e85863416bc622c60df05489a9eb52"
|
| 50 |
-
}
|
| 51 |
-
]
|
| 52 |
-
},
|
| 53 |
-
"Lux": {
|
| 54 |
-
"panels": {
|
| 55 |
-
"old_core": {
|
| 56 |
-
"path": "results/lux-new-host-v1/quality/lux/old_core/normalized.jsonl",
|
| 57 |
-
"sha256": "a659a7c7fef2871bbf806e842ea8ca81f645dcf610f0f2f6622d8d32e76a0836"
|
| 58 |
-
},
|
| 59 |
-
"v3_core": {
|
| 60 |
-
"path": "results/lux-new-host-v1/quality/lux/v3_core/normalized.jsonl",
|
| 61 |
-
"sha256": "da8903a07e36a9687f6ee1d0fd8dbc684e61c49ef5fea30c346e46bff5a90e94"
|
| 62 |
-
},
|
| 63 |
-
"v4": {
|
| 64 |
-
"path": "results/lux-new-host-v1/quality/lux/v4/normalized.jsonl",
|
| 65 |
-
"sha256": "26cd7482693ec1ffa1a93a156abfc88378e4872e320f3b9b775763635086d488"
|
| 66 |
-
},
|
| 67 |
-
"v5": {
|
| 68 |
-
"path": "results/lux-new-host-v1/quality/lux/v5/normalized.jsonl",
|
| 69 |
-
"sha256": "759c773e629a16317271de850780a8734533c795046122f850914c4482d1e72b"
|
| 70 |
-
}
|
| 71 |
-
},
|
| 72 |
-
"transfer_rows": {
|
| 73 |
-
"path": "analysis/decision-benchmark-v3-baselines/Lux/ROWS.json",
|
| 74 |
-
"sha256": "bafb55fc86762061e537382360eda14a8a8f8c5e4e111c2fda1f82cd6c06793f"
|
| 75 |
-
},
|
| 76 |
-
"evidence": [
|
| 77 |
-
{
|
| 78 |
-
"path": "results/lux-new-host-v1/quality/lux/COMPLETE.json",
|
| 79 |
-
"sha256": "9600ff792cbda13af5505c804cf63af94d2ee895f2912f0645a7d493ae1b7a9c"
|
| 80 |
-
},
|
| 81 |
-
{
|
| 82 |
-
"path": "results/lux-new-host-v1/quality/lux/metadata.json",
|
| 83 |
-
"sha256": "0d4509fae7eb3630118eaebc24e75a7b02f717d9a0ad74811fa27358c33c9cc5"
|
| 84 |
-
},
|
| 85 |
-
{
|
| 86 |
-
"path": "analysis/decision-benchmark-v3-baselines/Lux-projected/COMPLETE.json",
|
| 87 |
-
"sha256": "f58d02df78f297d4feebe39facfac072951111d8d6a19e2c1069f1b90250c622"
|
| 88 |
-
},
|
| 89 |
-
{
|
| 90 |
-
"path": "analysis/decision-benchmark-v3-baselines/Lux/REPORT.json",
|
| 91 |
-
"sha256": "f31703c342efe043a1e1534da189808bb7e0de0641fbaf3f91faa4a1f928a299"
|
| 92 |
-
},
|
| 93 |
-
{
|
| 94 |
-
"path": "analysis/decoder4b/lux9b-training-v1/CHECKPOINT-HELDOUT-COLLECTED.json",
|
| 95 |
-
"sha256": "d1c42515ab225b27ea011a6cf2262fe12dd7001e3b4d2cb64a23e7642c7a0d91"
|
| 96 |
-
}
|
| 97 |
-
]
|
| 98 |
-
},
|
| 99 |
-
"Nox": {
|
| 100 |
-
"panels": {
|
| 101 |
-
"old_core": {
|
| 102 |
-
"path": "results/nox-null-description-v1/full-regression/old_core/normalized.jsonl",
|
| 103 |
-
"sha256": "dbd18273dbbb9d09ba37d19dfedb0fcd6b64841d528632f04d39fd7955799967"
|
| 104 |
-
},
|
| 105 |
-
"v3_core": {
|
| 106 |
-
"path": "results/nox-null-description-v1/full-regression/v3_core/normalized.jsonl",
|
| 107 |
-
"sha256": "5286d1c3a78649e3284a018b1518f94cf32d8f0ad43a49c6a14c993524a3aff8"
|
| 108 |
-
},
|
| 109 |
-
"v4": {
|
| 110 |
-
"path": "results/nox-null-description-v1/full-regression/v4/normalized.jsonl",
|
| 111 |
-
"sha256": "077b752f4866bd0fc83db9b31c6227fd5015ce3585d2138c1a7a803c433d472b"
|
| 112 |
-
},
|
| 113 |
-
"v5": {
|
| 114 |
-
"path": "results/nox-null-description-v1/full-regression/v5/normalized.jsonl",
|
| 115 |
-
"sha256": "d57f07df59ff9b0e6ace0658e7917746e2b8d3bf518309a3858045cd175d79df"
|
| 116 |
-
}
|
| 117 |
-
},
|
| 118 |
-
"transfer_rows": {
|
| 119 |
-
"path": "analysis/nox-null-description-candidate-v4/transfer/ROWS.json",
|
| 120 |
-
"sha256": "7cee9e036ee53a3c41918cfafc73ad0d68f16b266abcae43077d1d69247afdfe"
|
| 121 |
-
},
|
| 122 |
-
"evidence": [
|
| 123 |
-
{
|
| 124 |
-
"path": "results/nox-null-description-v1/COMPLETE.json",
|
| 125 |
-
"sha256": "b813aa5db8c43b728b94abbaef20cefc8d3d84b51f7873ca46b3afb59aab6fe4"
|
| 126 |
-
},
|
| 127 |
-
{
|
| 128 |
-
"path": "results/nox-null-description-v1/full-regression/COMPLETE.json",
|
| 129 |
-
"sha256": "ad80e5bfb2558ee3b8fb1fc80617fcefbc1fbb8e58b8155e3cf5dfde38528754"
|
| 130 |
-
},
|
| 131 |
-
{
|
| 132 |
-
"path": "results/nox-null-description-v1/transfer-v9/COMPLETE.json",
|
| 133 |
-
"sha256": "93a8767f7245b93764e527e616f236b5fa46b24a087a7e8ff66d35eee9f9fd0f"
|
| 134 |
-
},
|
| 135 |
-
{
|
| 136 |
-
"path": "results/nox-null-description-v1/full-regression-ACTUAL-EXIT.json",
|
| 137 |
-
"sha256": "dca6e05615c9c82a2d506438fb4e7c9abfe03053661515ffc9a4c5b176fd3ed8"
|
| 138 |
-
},
|
| 139 |
-
{
|
| 140 |
-
"path": "results/nox-null-description-v1/transfer-v9-ACTUAL-EXIT.json",
|
| 141 |
-
"sha256": "09c4c05549253c386f72b52e8ad4fbbc47d05af59715caebf9d01a6e8734d0af"
|
| 142 |
-
},
|
| 143 |
-
{
|
| 144 |
-
"path": "analysis/nox-null-description-candidate-v4/transfer/REPORT.json",
|
| 145 |
-
"sha256": "f02e447787c0c5afcb8acbd06408407719ebfc191e8501e8a887cd96c150f81b"
|
| 146 |
-
}
|
| 147 |
-
]
|
| 148 |
},
|
| 149 |
"kev-9b": {
|
| 150 |
-
"
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
|
| 154 |
-
|
| 155 |
-
"
|
| 156 |
-
|
| 157 |
-
"sha256": "bb90e5701a151272f97caa4107888ba2379969b4e3f22abb4670a528b50d7729"
|
| 158 |
-
},
|
| 159 |
-
"v4": {
|
| 160 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-9b-v4-normalized.jsonl",
|
| 161 |
-
"sha256": "b9c50612c0be1b7b5647779d8489c185ea7f02673da176b1d965673da604d644"
|
| 162 |
-
},
|
| 163 |
-
"v5": {
|
| 164 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-9b-v5-normalized.jsonl",
|
| 165 |
-
"sha256": "2893714030357718e662633fddf31a328e56a02fc3ae71ed34612f480e50d4ee"
|
| 166 |
-
}
|
| 167 |
-
},
|
| 168 |
-
"transfer_rows": {
|
| 169 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-9b/v9-transfer-v9-test-published-temperature-rows.json",
|
| 170 |
-
"sha256": "e712bf23acea6a9431001f5424639d1e3738789907ff2e367db7c9dad4e0d239"
|
| 171 |
-
},
|
| 172 |
-
"evidence": [
|
| 173 |
-
{
|
| 174 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-9b/REPORT.json",
|
| 175 |
-
"sha256": "96dcf11961f510d05c011aa3fd8dfa6043348471b8e800d6bac2d691d0815d54"
|
| 176 |
-
},
|
| 177 |
-
{
|
| 178 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/POINTS.json",
|
| 179 |
-
"sha256": "a3aa6d501a14df1cf5e7c98ecc6161c88483801310c56913465059d28c4d88ea"
|
| 180 |
-
}
|
| 181 |
-
]
|
| 182 |
},
|
| 183 |
"kev-4b": {
|
| 184 |
-
"
|
| 185 |
-
|
| 186 |
-
|
| 187 |
-
|
| 188 |
-
|
| 189 |
-
"
|
| 190 |
-
|
| 191 |
-
"sha256": "2ec54da92c4e72412e89b3c07013b54729f5899ba8542eeadf962b04a1f3a4b8"
|
| 192 |
-
},
|
| 193 |
-
"v4": {
|
| 194 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-4b-v4-normalized.jsonl",
|
| 195 |
-
"sha256": "88c576c3d62ec11686dd5ebf360971a9087779e45a346897e2badf874a754a12"
|
| 196 |
-
},
|
| 197 |
-
"v5": {
|
| 198 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-4b-v5-normalized.jsonl",
|
| 199 |
-
"sha256": "d33bbb60134314dcb550060176ed0a3ae91ec0c6b033cceb40e3255b33666898"
|
| 200 |
-
}
|
| 201 |
-
},
|
| 202 |
-
"transfer_rows": {
|
| 203 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-4b/v9-transfer-v9-test-published-temperature-rows.json",
|
| 204 |
-
"sha256": "c2cb16a936be4c10ca6b1870c69426ddd6b51629c401f71ae1da652b3bd2ac7e"
|
| 205 |
-
},
|
| 206 |
-
"evidence": [
|
| 207 |
-
{
|
| 208 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-4b/REPORT.json",
|
| 209 |
-
"sha256": "366fb0d27a4d732e5e270c264521d9a93ccabc4f0ddf5b7d3e9f1ecc5f460406"
|
| 210 |
-
},
|
| 211 |
-
{
|
| 212 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/POINTS.json",
|
| 213 |
-
"sha256": "a3aa6d501a14df1cf5e7c98ecc6161c88483801310c56913465059d28c4d88ea"
|
| 214 |
-
}
|
| 215 |
-
]
|
| 216 |
},
|
| 217 |
"Qwen3.5-9B": {
|
| 218 |
-
"
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
|
| 222 |
-
|
| 223 |
-
"
|
| 224 |
-
|
| 225 |
-
"sha256": "d081a8158863e85d5e49cb853fee87192119767124f53f9e6c782392f8edb018"
|
| 226 |
-
},
|
| 227 |
-
"v4": {
|
| 228 |
-
"path": "results/lux-new-host-v1/quality/base9b/v4/normalized.jsonl",
|
| 229 |
-
"sha256": "cc2775884d84c49e5155afbebc7f4562ec86f6f6d32765ffcbdafe653ca7d132"
|
| 230 |
-
},
|
| 231 |
-
"v5": {
|
| 232 |
-
"path": "results/lux-new-host-v1/quality/base9b/v5/normalized.jsonl",
|
| 233 |
-
"sha256": "ce6bf5c286e5920055d7de36fbd910311f7d3185380c63681c7baea6d0e3494a"
|
| 234 |
-
}
|
| 235 |
-
},
|
| 236 |
-
"transfer_rows": {
|
| 237 |
-
"path": "analysis/decision-benchmark-v3-baselines/Qwen3.5-9B/ROWS.json",
|
| 238 |
-
"sha256": "5acdd41f616a3ba63a5407af9c8371268012792fd324b70230d899bbb37101b4"
|
| 239 |
-
},
|
| 240 |
-
"evidence": [
|
| 241 |
-
{
|
| 242 |
-
"path": "results/lux-new-host-v1/quality/base9b/COMPLETE.json",
|
| 243 |
-
"sha256": "702c65f5c290f5cdda0495479f361101dbe976330233c5359455ba736245c3aa"
|
| 244 |
-
},
|
| 245 |
-
{
|
| 246 |
-
"path": "results/lux-new-host-v1/quality/base9b/metadata.json",
|
| 247 |
-
"sha256": "eaa29f71aba60e86068e6c5d1b5782ea73b0214b81a297b57419c4f0a2fc6c04"
|
| 248 |
-
},
|
| 249 |
-
{
|
| 250 |
-
"path": "results/lux-new-host-v1/quality/base9b/old_core/complete.json",
|
| 251 |
-
"sha256": "d28f2a646a908fc775e3727569a608f30ae5820f88a6173750a1031604e84f47"
|
| 252 |
-
},
|
| 253 |
-
{
|
| 254 |
-
"path": "results/lux-new-host-v1/quality/base9b/v3_core/complete.json",
|
| 255 |
-
"sha256": "3a10bd5e2121a81f8ec6f76b2526723a17689ec34d13f062c5259983203992cc"
|
| 256 |
-
},
|
| 257 |
-
{
|
| 258 |
-
"path": "results/lux-new-host-v1/quality/base9b/v4/complete.json",
|
| 259 |
-
"sha256": "073239ab7dd21768bccdfa4aaf3217c99dc7ff9b00fc308ffbe889c72fe002b4"
|
| 260 |
-
},
|
| 261 |
-
{
|
| 262 |
-
"path": "results/lux-new-host-v1/quality/base9b/v5/complete.json",
|
| 263 |
-
"sha256": "9a8d1e8e8db066f1f6594f137a7a1571dd5184aae363a9f919e44b634b782bba"
|
| 264 |
-
},
|
| 265 |
-
{
|
| 266 |
-
"path": "analysis/decision-benchmark-v3-baselines/Qwen3.5-9B/REPORT.json",
|
| 267 |
-
"sha256": "31b0a92bbbe7a0d259cb8e296a10b04b2bdd685b394eabdc5fbf6da1556ff12d"
|
| 268 |
-
},
|
| 269 |
-
{
|
| 270 |
-
"path": "analysis/decision-benchmark-v3-baselines/Qwen3.5-9B-projected/COMPLETE.json",
|
| 271 |
-
"sha256": "eaff79adb3a6a6bad61253a01b4c4e22cfa3d9969cd963cd82a3685755b77ec1"
|
| 272 |
-
}
|
| 273 |
-
]
|
| 274 |
},
|
| 275 |
"Decider": {
|
| 276 |
-
"
|
| 277 |
-
|
| 278 |
-
|
| 279 |
-
|
| 280 |
-
|
| 281 |
-
"
|
| 282 |
-
|
| 283 |
-
"sha256": "12a1f554decf1aff608743e7b4a44681289ea084ae25f383e519b1f34917d46f"
|
| 284 |
-
},
|
| 285 |
-
"v4": {
|
| 286 |
-
"path": "eval/heldout/v4/baselines/decider/core/normalized.jsonl",
|
| 287 |
-
"sha256": "626ef366ae471ec10dfb89ef2ff2f7b29a6c879976d3ff14fcbb2bafd7d9b045"
|
| 288 |
-
},
|
| 289 |
-
"v5": {
|
| 290 |
-
"path": "eval/heldout/v5/open-baselines-v1/decider/core/normalized.jsonl",
|
| 291 |
-
"sha256": "ae16b17ac8e4208940c0c3e9e04f25e94686b5d16bf46f64f53669baf41aeb71"
|
| 292 |
-
}
|
| 293 |
-
},
|
| 294 |
-
"transfer_rows": {
|
| 295 |
-
"path": "analysis/decision-benchmark-v3-baselines/decider/ROWS.json",
|
| 296 |
-
"sha256": "6da5b5692a8a0cea3dc8f6d02d911fafd9bbf62e286a9be8db27c89f77a897e4"
|
| 297 |
-
},
|
| 298 |
-
"evidence": [
|
| 299 |
-
{
|
| 300 |
-
"path": "results/accelerated-transfer-baselines-v1/decider/worker/COMPLETE.json",
|
| 301 |
-
"sha256": "e1ee0df8df59ec3eb2b49ba947a8508994a24bce226b7f2cccbe25f6835b99af"
|
| 302 |
-
},
|
| 303 |
-
{
|
| 304 |
-
"path": "results/accelerated-transfer-baselines-v1/decider/worker/METADATA.json",
|
| 305 |
-
"sha256": "f00061c1b0bb642bf1353aae867d19df988c01b0428b58206b31249521bcbff8"
|
| 306 |
-
},
|
| 307 |
-
{
|
| 308 |
-
"path": "results/accelerated-transfer-baselines-v1/decider/worker/predictions.jsonl",
|
| 309 |
-
"sha256": "f69e87fa65b09289e6861c1d789270aea5a92d23ac47817342647ed08f406df7"
|
| 310 |
-
},
|
| 311 |
-
{
|
| 312 |
-
"path": "analysis/decision-benchmark-v3-baselines/decider/REPORT.json",
|
| 313 |
-
"sha256": "5aae029f34524dfebc8e79128da8c39229c5747461dfce07ca59956da4416951"
|
| 314 |
-
},
|
| 315 |
-
{
|
| 316 |
-
"path": "eval/expanded-public-extension-v1/COMPARATOR-REGISTRY.json",
|
| 317 |
-
"sha256": "b80be4270721cbeeac9e28beb10c6f19f75c1cbf2553e38d3d59e3a42d56a91c"
|
| 318 |
-
},
|
| 319 |
-
{
|
| 320 |
-
"path": "analysis/decision-benchmark-v3-baselines/decider/COMPOSABLE-MANIFEST.json",
|
| 321 |
-
"sha256": "a49848e86feafaa4f36a33287910035e7605e344791aae139305b78cb304a77f"
|
| 322 |
-
}
|
| 323 |
-
]
|
| 324 |
},
|
| 325 |
"Qwen3.5-4B": {
|
| 326 |
-
"
|
| 327 |
-
|
| 328 |
-
|
| 329 |
-
|
| 330 |
-
|
| 331 |
-
"
|
| 332 |
-
|
| 333 |
-
"sha256": "7541893d2b2ec0ec2ffc513f50c568eaef860b65bcef9a1747996c116fff9661"
|
| 334 |
-
},
|
| 335 |
-
"v4": {
|
| 336 |
-
"path": "eval/heldout/v4/baselines/base-4b/core/normalized.jsonl",
|
| 337 |
-
"sha256": "7a6d252ccaa72bf54b94445190bbad023fb465a3eaf4cf657532e2cdd4f750f5"
|
| 338 |
-
},
|
| 339 |
-
"v5": {
|
| 340 |
-
"path": "eval/heldout/v5/open-baselines-v1/base-4b/core/normalized.jsonl",
|
| 341 |
-
"sha256": "c01ad5daed65267e9aec40ed9e97ece1b6511e4cab17ebff09339b41524fb7f1"
|
| 342 |
-
}
|
| 343 |
-
},
|
| 344 |
-
"transfer_rows": {
|
| 345 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-4b/ROWS.json",
|
| 346 |
-
"sha256": "37ab17d94d640ee46283e1f23723b9ea05133c22a8a9b62a97c756a89c5de785"
|
| 347 |
-
},
|
| 348 |
-
"evidence": [
|
| 349 |
-
{
|
| 350 |
-
"path": "results/remaining-transfer-baselines-v1/base-4b/worker/COMPLETE.json",
|
| 351 |
-
"sha256": "72abc4f87a9234f95eb3b0a72a9e7d66798779d5982c757faffb610003b1b242"
|
| 352 |
-
},
|
| 353 |
-
{
|
| 354 |
-
"path": "results/remaining-transfer-baselines-v1/base-4b/worker/METADATA.json",
|
| 355 |
-
"sha256": "871fed92cf762187015e59174a042808dbf3fc88f7791aa0ddd530ae923196ab"
|
| 356 |
-
},
|
| 357 |
-
{
|
| 358 |
-
"path": "results/remaining-transfer-baselines-v1/base-4b/worker/predictions.jsonl",
|
| 359 |
-
"sha256": "8c47edda469f8914eccc2ecafa91c245752812e830cb9f29445653102522854a"
|
| 360 |
-
},
|
| 361 |
-
{
|
| 362 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-4b/REPORT.json",
|
| 363 |
-
"sha256": "c2a0aafbe2b7c85b3c07a4b9843182a3361654f57883311155aa3b05cfc7002d"
|
| 364 |
-
},
|
| 365 |
-
{
|
| 366 |
-
"path": "eval/expanded-public-extension-v1/COMPARATOR-REGISTRY.json",
|
| 367 |
-
"sha256": "b80be4270721cbeeac9e28beb10c6f19f75c1cbf2553e38d3d59e3a42d56a91c"
|
| 368 |
-
},
|
| 369 |
-
{
|
| 370 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-4b/COMPOSABLE-MANIFEST.json",
|
| 371 |
-
"sha256": "c5966aab01f3d82ee26ed56dc26ed1277f4d3cbc976ecf28d2e8d3d190fd8fc0"
|
| 372 |
-
}
|
| 373 |
-
]
|
| 374 |
},
|
| 375 |
"Sol": {
|
| 376 |
-
"
|
| 377 |
-
|
| 378 |
-
|
| 379 |
-
|
| 380 |
-
|
| 381 |
-
"
|
| 382 |
-
|
| 383 |
-
"sha256": "b2c99fbd739b849a66a55929a658de4517a7ffaa4496255566e6972ba28d4f15"
|
| 384 |
-
},
|
| 385 |
-
"v4": {
|
| 386 |
-
"path": "results/expanded/sol-composition-v2/candidate/job-0/worker/v4/normalized.jsonl",
|
| 387 |
-
"sha256": "3c28302750e02d4cead0aacd66a323734e773f726c4c41db658dd6461251b28b"
|
| 388 |
-
},
|
| 389 |
-
"v5": {
|
| 390 |
-
"path": "results/expanded/sol-composition-v2/candidate/job-0/worker/v5/normalized.jsonl",
|
| 391 |
-
"sha256": "6ddb0496012b596ea122afa4db4c4158e57dfb934e9ae29eb006756b652e31e1"
|
| 392 |
-
}
|
| 393 |
-
},
|
| 394 |
-
"transfer_rows": {
|
| 395 |
-
"path": "analysis/kev-reciprocal-v1/reports/Decision-1.0-Sol-attempt02/v9-transfer-v9-test-published-temperature-rows.json",
|
| 396 |
-
"sha256": "b05799417cb3715de633135c58f713960b75bdb9ed30eb8912a197d33ee3d85b"
|
| 397 |
-
},
|
| 398 |
-
"evidence": [
|
| 399 |
-
{
|
| 400 |
-
"path": "results/expanded/sol-composition-v2/QUALITY-COLLECTION.json",
|
| 401 |
-
"sha256": "9b77aac87b85cbf3a74e86af4a2b9a03fd9e1ed854356a92ff54ccd0d47a16f4"
|
| 402 |
-
},
|
| 403 |
-
{
|
| 404 |
-
"path": "analysis/kev-reciprocal-v1/reports/Decision-1.0-Sol-attempt02/REPORT.json",
|
| 405 |
-
"sha256": "5d8ebb339293687cec9089f54fb96aabf3330f16111487a49489f5e52bbb6f29"
|
| 406 |
-
}
|
| 407 |
-
]
|
| 408 |
},
|
| 409 |
"kev-0.8b": {
|
| 410 |
-
"
|
| 411 |
-
|
| 412 |
-
|
| 413 |
-
|
| 414 |
-
|
| 415 |
-
"
|
| 416 |
-
|
| 417 |
-
"sha256": "70cbfc69dfc72ac0c5ae01305ec63a9f31129c115b7068c6824cd22ead013d68"
|
| 418 |
-
},
|
| 419 |
-
"v4": {
|
| 420 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-0.8b-v4-normalized.jsonl",
|
| 421 |
-
"sha256": "761dfd69acbface101f2c21fd27d2db684c5c8b8cc425890a075729ae1f7db24"
|
| 422 |
-
},
|
| 423 |
-
"v5": {
|
| 424 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/kev-0.8b-v5-normalized.jsonl",
|
| 425 |
-
"sha256": "51b6fc5f826ca8db44df43885f34d857ddd5b5a66453bbdc1a83fcf488928bdb"
|
| 426 |
-
}
|
| 427 |
-
},
|
| 428 |
-
"transfer_rows": {
|
| 429 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-0.8b/v9-transfer-v9-test-published-temperature-rows.json",
|
| 430 |
-
"sha256": "d6abed88001aeac630219466f02c1aad89fcbd354b3ae533507f3c2f7fe42898"
|
| 431 |
-
},
|
| 432 |
-
"evidence": [
|
| 433 |
-
{
|
| 434 |
-
"path": "analysis/kev-reciprocal-v1/reports/kev-0.8b/REPORT.json",
|
| 435 |
-
"sha256": "6760dcbaa7d85e6c2e893e4c80ab8e66e0b84d7cfa26ae9e25909f48a05f8de4"
|
| 436 |
-
},
|
| 437 |
-
{
|
| 438 |
-
"path": "analysis/kev-reciprocal-v1/our-panels/POINTS.json",
|
| 439 |
-
"sha256": "a3aa6d501a14df1cf5e7c98ecc6161c88483801310c56913465059d28c4d88ea"
|
| 440 |
-
}
|
| 441 |
-
]
|
| 442 |
},
|
| 443 |
"Qwen3.5-2B": {
|
| 444 |
-
"
|
| 445 |
-
|
| 446 |
-
|
| 447 |
-
|
| 448 |
-
|
| 449 |
-
"
|
| 450 |
-
|
| 451 |
-
"sha256": "cc51471716342ecd466b50d5dd9b4ba805f5a6873cd96075b56d7c2e8336175e"
|
| 452 |
-
},
|
| 453 |
-
"v4": {
|
| 454 |
-
"path": "eval/heldout/v4/baselines/base-2b/core/normalized.jsonl",
|
| 455 |
-
"sha256": "93d84cdbe607691e2cef35ece3aa2dd5900f6410b0d72796b86d74c9156f3cdd"
|
| 456 |
-
},
|
| 457 |
-
"v5": {
|
| 458 |
-
"path": "eval/heldout/v5/open-baselines-v1/base-2b/core/normalized.jsonl",
|
| 459 |
-
"sha256": "75740c4279eb8dcc9c90b55f155916b3748c887e8c9d9e1d3a47edf7f4dbc34b"
|
| 460 |
-
}
|
| 461 |
-
},
|
| 462 |
-
"transfer_rows": {
|
| 463 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-2b/ROWS.json",
|
| 464 |
-
"sha256": "62bf4f977d03eabd4f051aca5c8f7204f350eb995545aedeefa7bfa00bcbd115"
|
| 465 |
-
},
|
| 466 |
-
"evidence": [
|
| 467 |
-
{
|
| 468 |
-
"path": "results/accelerated-transfer-baselines-v1/base-2b/worker/COMPLETE.json",
|
| 469 |
-
"sha256": "0353d786b760d491c2a5ff3f8363ecd27a537f25a42dbeb2cea4db5d5eb77910"
|
| 470 |
-
},
|
| 471 |
-
{
|
| 472 |
-
"path": "results/accelerated-transfer-baselines-v1/base-2b/worker/METADATA.json",
|
| 473 |
-
"sha256": "745151846ade07431bc8ac964894bbd6ed234fb6a4665fcce223d91e09acaeeb"
|
| 474 |
-
},
|
| 475 |
-
{
|
| 476 |
-
"path": "results/accelerated-transfer-baselines-v1/base-2b/worker/predictions.jsonl",
|
| 477 |
-
"sha256": "347fc785ee59d5ca7694357b68a6c2e7bc7f863f1e26d94fc9b989391ed7d5fe"
|
| 478 |
-
},
|
| 479 |
-
{
|
| 480 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-2b/REPORT.json",
|
| 481 |
-
"sha256": "9eb1db12e826a75dcda322960478bb153c4c1c24e5f5853d29a761b82bea36e7"
|
| 482 |
-
},
|
| 483 |
-
{
|
| 484 |
-
"path": "eval/expanded-public-extension-v1/COMPARATOR-REGISTRY.json",
|
| 485 |
-
"sha256": "b80be4270721cbeeac9e28beb10c6f19f75c1cbf2553e38d3d59e3a42d56a91c"
|
| 486 |
-
},
|
| 487 |
-
{
|
| 488 |
-
"path": "analysis/decision-benchmark-v3-baselines/base-2b/COMPOSABLE-MANIFEST.json",
|
| 489 |
-
"sha256": "c3e6f28b600776686afc56c1386cabf0869e69d5ca79f3ae6a0b17189eee968e"
|
| 490 |
-
}
|
| 491 |
-
]
|
| 492 |
},
|
| 493 |
"Laya-base": {
|
| 494 |
-
"
|
| 495 |
-
|
| 496 |
-
|
| 497 |
-
|
| 498 |
-
|
| 499 |
-
"
|
| 500 |
-
|
| 501 |
-
"sha256": "159614ceede814b578e1e025836a595c8c71a9bec4896caf6454d7bf54c4c44d"
|
| 502 |
-
},
|
| 503 |
-
"old_core": {
|
| 504 |
-
"path": "eval/heldout/primary-v2/laya-english/core/normalized.jsonl",
|
| 505 |
-
"sha256": "850643d2daeb47a6909985003736cab0f1f0e5a87631a35e27192f0733efd756"
|
| 506 |
-
},
|
| 507 |
-
"v3_core": {
|
| 508 |
-
"path": "eval/heldout/v3/baselines/laya-english/core/normalized.jsonl",
|
| 509 |
-
"sha256": "dec19caeb8074e25289e43a367392172da21f0c2ccdb403a059e2e4a2364a64b"
|
| 510 |
-
}
|
| 511 |
-
},
|
| 512 |
-
"transfer_rows": {
|
| 513 |
-
"path": "analysis/decision-benchmark-v3-baselines/laya-english/ROWS.json",
|
| 514 |
-
"sha256": "19138909259400facb5d3d5402d0d736f870695fe3ee46019f2e438088c57275"
|
| 515 |
-
},
|
| 516 |
-
"evidence": [
|
| 517 |
-
{
|
| 518 |
-
"path": "results/public-roster-baselines-v1/laya-english/worker/COMPLETE.json",
|
| 519 |
-
"sha256": "cef0da797ae14cebaf1e4b9edf9b16c3cd9a412a30435d7d2517b8ba6dbe8b98"
|
| 520 |
-
},
|
| 521 |
-
{
|
| 522 |
-
"path": "analysis/public-roster-discovery-v1/ACTUAL-EVALUATION-COMPLETE.json",
|
| 523 |
-
"sha256": "686834a8289d3771908530a95c2ff85df71e61dbb1ce7ea95b7c9550a23ba82f"
|
| 524 |
-
},
|
| 525 |
-
{
|
| 526 |
-
"path": "analysis/decision-benchmark-v3-baselines/laya-english/REPORT.json",
|
| 527 |
-
"sha256": "cb5935bb399dcf23ffe24f855452b2e074bcbb3d023cfc7dd439342e54f02c6b"
|
| 528 |
-
},
|
| 529 |
-
{
|
| 530 |
-
"path": "analysis/public-roster-discovery-v1/LAYA-CORE-REUSE-VERIFIED.json",
|
| 531 |
-
"sha256": "01d2be9f23540f3fbe5d06e6707bb3d908dd76d2c509ee4af86c9e0c6129862f"
|
| 532 |
-
}
|
| 533 |
-
]
|
| 534 |
},
|
| 535 |
"Laya-multilingual": {
|
| 536 |
-
"
|
| 537 |
-
|
| 538 |
-
|
| 539 |
-
|
| 540 |
-
|
| 541 |
-
"
|
| 542 |
-
|
| 543 |
-
|
| 544 |
-
|
| 545 |
-
|
| 546 |
-
|
| 547 |
-
|
| 548 |
-
|
| 549 |
-
"
|
| 550 |
-
|
| 551 |
-
|
| 552 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 553 |
},
|
| 554 |
-
"
|
| 555 |
-
|
| 556 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 557 |
},
|
| 558 |
-
"
|
| 559 |
-
|
| 560 |
-
"path": "results/public-roster-baselines-v1/laya-multilingual/worker/COMPLETE.json",
|
| 561 |
-
"sha256": "b190d683fc4c4762c2805a03d52628a53e125d228a3105c8ac2ddda00a5364e4"
|
| 562 |
-
},
|
| 563 |
-
{
|
| 564 |
-
"path": "analysis/public-roster-discovery-v1/ACTUAL-EVALUATION-COMPLETE.json",
|
| 565 |
-
"sha256": "686834a8289d3771908530a95c2ff85df71e61dbb1ce7ea95b7c9550a23ba82f"
|
| 566 |
-
},
|
| 567 |
-
{
|
| 568 |
-
"path": "analysis/decision-benchmark-v3-baselines/laya-multilingual/REPORT.json",
|
| 569 |
-
"sha256": "d20d3d58f76ca2758545f83d358abd747f86d581c897d1fb521371d1000d6154"
|
| 570 |
-
},
|
| 571 |
-
{
|
| 572 |
-
"path": "analysis/public-roster-discovery-v1/LAYA-CORE-REUSE-VERIFIED.json",
|
| 573 |
-
"sha256": "01d2be9f23540f3fbe5d06e6707bb3d908dd76d2c509ee4af86c9e0c6129862f"
|
| 574 |
-
}
|
| 575 |
-
]
|
| 576 |
}
|
| 577 |
},
|
| 578 |
"task_rows": 54,
|
| 579 |
-
"
|
| 580 |
-
|
| 581 |
-
|
| 582 |
-
"
|
| 583 |
-
"
|
| 584 |
-
"
|
| 585 |
-
"
|
| 586 |
-
"
|
| 587 |
-
|
| 588 |
-
|
| 589 |
-
|
| 590 |
-
"
|
| 591 |
-
|
| 592 |
-
|
| 593 |
-
|
| 594 |
-
|
| 595 |
-
|
| 596 |
-
|
| 597 |
-
|
| 598 |
-
|
| 599 |
-
|
| 600 |
-
|
| 601 |
-
|
| 602 |
-
|
| 603 |
-
|
| 604 |
-
|
| 605 |
-
|
| 606 |
-
|
| 607 |
-
|
| 608 |
-
"
|
| 609 |
-
"
|
| 610 |
-
|
| 611 |
-
|
| 612 |
-
|
| 613 |
-
|
| 614 |
-
|
| 615 |
-
|
| 616 |
-
"
|
| 617 |
-
"
|
| 618 |
-
|
| 619 |
-
|
| 620 |
-
|
| 621 |
-
|
| 622 |
-
|
| 623 |
-
|
| 624 |
-
|
| 625 |
-
|
| 626 |
-
|
| 627 |
-
|
| 628 |
-
|
| 629 |
-
|
| 630 |
-
|
| 631 |
-
|
| 632 |
-
"
|
| 633 |
-
"
|
| 634 |
-
|
| 635 |
-
|
| 636 |
-
"
|
| 637 |
-
"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 638 |
},
|
| 639 |
-
"
|
| 640 |
-
"
|
| 641 |
-
"
|
| 642 |
-
"parameter_entries": 169,
|
| 643 |
-
"named_parameter_entries_with_duplicates": 169,
|
| 644 |
-
"tied_parameter_aliases": [],
|
| 645 |
-
"persistent_buffers": {
|
| 646 |
-
"temperature": [
|
| 647 |
-
3
|
| 648 |
-
]
|
| 649 |
-
},
|
| 650 |
-
"persistent_buffer_scalar_count": 3,
|
| 651 |
-
"nonpersistent_buffers": {
|
| 652 |
-
"encoder.rotary_emb.full_attention_inv_freq": [
|
| 653 |
-
32
|
| 654 |
-
],
|
| 655 |
-
"encoder.rotary_emb.full_attention_original_inv_freq": [
|
| 656 |
-
32
|
| 657 |
-
],
|
| 658 |
-
"encoder.rotary_emb.sliding_attention_inv_freq": [
|
| 659 |
-
32
|
| 660 |
-
],
|
| 661 |
-
"encoder.rotary_emb.sliding_attention_original_inv_freq": [
|
| 662 |
-
32
|
| 663 |
-
]
|
| 664 |
-
},
|
| 665 |
-
"exact_state_dict_shapes_match_header": true,
|
| 666 |
-
"header_sha256": "22ab94329063133fdd2944997b906bb6076d2cfd3e43eb05d7ea92187e9c3984",
|
| 667 |
-
"source_sha256": "8d83611d480c971d640a7b7d3aa2f2219c5e8455e9cc2329fd073681bd8be23e",
|
| 668 |
-
"encoder_config_sha256": "83f6916d13ef0f556ac461f28308dc2bffa7ebeadee8ec9e2db5812020ea5bb4",
|
| 669 |
-
"agent_config_sha256": "25061739243b617ad88d1219ba6f8a9c86c5881ca28df024fa2d9b3b2fcc30c6",
|
| 670 |
-
"architecture": "ModernBertModel",
|
| 671 |
-
"interpretation": "Complete decision model, including encoder and heads; temperature calibration buffer excluded. No tied parameter aliases in the instantiated published model."
|
| 672 |
}
|
| 673 |
-
}
|
| 674 |
-
|
| 675 |
-
|
| 676 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
|
| 3 |
+
"merged_statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc",
|
| 4 |
+
"source_per_model": {
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5 |
"Jev": {
|
| 6 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 7 |
+
"source_model_key": "Jev",
|
| 8 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 9 |
+
"qualified_runtime": {
|
| 10 |
+
"python": "3.12.13",
|
| 11 |
+
"numpy": "2.3.5"
|
| 12 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
},
|
| 14 |
"kev-9b": {
|
| 15 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 16 |
+
"source_model_key": "kev-9b",
|
| 17 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 18 |
+
"qualified_runtime": {
|
| 19 |
+
"python": "3.12.13",
|
| 20 |
+
"numpy": "2.3.5"
|
| 21 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 22 |
},
|
| 23 |
"kev-4b": {
|
| 24 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 25 |
+
"source_model_key": "kev-4b",
|
| 26 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 27 |
+
"qualified_runtime": {
|
| 28 |
+
"python": "3.12.13",
|
| 29 |
+
"numpy": "2.3.5"
|
| 30 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 31 |
},
|
| 32 |
"Qwen3.5-9B": {
|
| 33 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 34 |
+
"source_model_key": "Qwen3.5-9B",
|
| 35 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 36 |
+
"qualified_runtime": {
|
| 37 |
+
"python": "3.12.13",
|
| 38 |
+
"numpy": "2.3.5"
|
| 39 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 40 |
},
|
| 41 |
"Decider": {
|
| 42 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 43 |
+
"source_model_key": "Decider",
|
| 44 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 45 |
+
"qualified_runtime": {
|
| 46 |
+
"python": "3.12.13",
|
| 47 |
+
"numpy": "2.3.5"
|
| 48 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 49 |
},
|
| 50 |
"Qwen3.5-4B": {
|
| 51 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 52 |
+
"source_model_key": "Qwen3.5-4B",
|
| 53 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 54 |
+
"qualified_runtime": {
|
| 55 |
+
"python": "3.12.13",
|
| 56 |
+
"numpy": "2.3.5"
|
| 57 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 58 |
},
|
| 59 |
"Sol": {
|
| 60 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 61 |
+
"source_model_key": "Sol",
|
| 62 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 63 |
+
"qualified_runtime": {
|
| 64 |
+
"python": "3.12.13",
|
| 65 |
+
"numpy": "2.3.5"
|
| 66 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 67 |
},
|
| 68 |
"kev-0.8b": {
|
| 69 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 70 |
+
"source_model_key": "kev-0.8b",
|
| 71 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 72 |
+
"qualified_runtime": {
|
| 73 |
+
"python": "3.12.13",
|
| 74 |
+
"numpy": "2.3.5"
|
| 75 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 76 |
},
|
| 77 |
"Qwen3.5-2B": {
|
| 78 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 79 |
+
"source_model_key": "Qwen3.5-2B",
|
| 80 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 81 |
+
"qualified_runtime": {
|
| 82 |
+
"python": "3.12.13",
|
| 83 |
+
"numpy": "2.3.5"
|
| 84 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 85 |
},
|
| 86 |
"Laya-base": {
|
| 87 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 88 |
+
"source_model_key": "Laya-base",
|
| 89 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 90 |
+
"qualified_runtime": {
|
| 91 |
+
"python": "3.12.13",
|
| 92 |
+
"numpy": "2.3.5"
|
| 93 |
+
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 94 |
},
|
| 95 |
"Laya-multilingual": {
|
| 96 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 97 |
+
"source_model_key": "Laya-multilingual",
|
| 98 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 99 |
+
"qualified_runtime": {
|
| 100 |
+
"python": "3.12.13",
|
| 101 |
+
"numpy": "2.3.5"
|
| 102 |
+
}
|
| 103 |
+
},
|
| 104 |
+
"Nox": {
|
| 105 |
+
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 106 |
+
"source_model_key": "Nox-null-description-candidate",
|
| 107 |
+
"manifest_sha256": "cf455486f8eba2b7158bbb4b5bcdcd8a4bd4daee4f05184b1df61b5f70ef2b04",
|
| 108 |
+
"qualified_runtime": {
|
| 109 |
+
"python": "3.12.13",
|
| 110 |
+
"numpy": "2.3.5"
|
| 111 |
+
}
|
| 112 |
+
},
|
| 113 |
+
"Lux": {
|
| 114 |
+
"statistics_sha256": "bfc75a9080fb0fa22a45ac21c23048d5c8e5deac81dcc202792254565951d8e7",
|
| 115 |
+
"source_model_key": "Lux-retention-candidate",
|
| 116 |
+
"manifest_sha256": "469d9e05ff0283cb50efe45307c7e7bcefa9cb124cc5103b9aa319e2537801a0",
|
| 117 |
+
"qualified_runtime": {
|
| 118 |
+
"python": "3.12.13",
|
| 119 |
+
"numpy": "2.3.5"
|
| 120 |
+
}
|
| 121 |
+
},
|
| 122 |
+
"Kai": {
|
| 123 |
+
"statistics_sha256": "fc72b6c9f3d1594f5e4a6b8d33f9a438cbd7b762e6794e2d0ae6f47296b83d4d",
|
| 124 |
+
"source_model_key": "Kai-fixed-noul-restore-v2",
|
| 125 |
+
"manifest_sha256": "610a5a29f7619a5d4057c864c36cfa4958493a156fbe4dfdcb9185f89fb1e07d",
|
| 126 |
+
"qualified_runtime": {
|
| 127 |
+
"python": "3.12.13",
|
| 128 |
+
"numpy": "2.3.5"
|
| 129 |
},
|
| 130 |
+
"hub_revision": "db41e715facc312cfdf95f0f3ed763ef662db401",
|
| 131 |
+
"parameters": 571909635
|
| 132 |
+
},
|
| 133 |
+
"Eos": {
|
| 134 |
+
"statistics_sha256": "b25a7628e52247f69a165f8f1e71937bff814a12bf4b030817607320e9101645",
|
| 135 |
+
"source_model_key": "Eos-v1-candidate",
|
| 136 |
+
"manifest_sha256": "06d0e9f4980253b01ff43dbfd2dcbef53fe7c5c46cbc564dc3c6f7c1d73a7ef1",
|
| 137 |
+
"qualified_runtime": {
|
| 138 |
+
"python": "3.12.13",
|
| 139 |
+
"numpy": "2.3.5"
|
| 140 |
},
|
| 141 |
+
"hub_revision": "a121640a2332b14548453089cb77badcc9fb9a06",
|
| 142 |
+
"parameters": 753446208
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 143 |
}
|
| 144 |
},
|
| 145 |
"task_rows": 54,
|
| 146 |
+
"scored_questions": 3766,
|
| 147 |
+
"public_models": 15,
|
| 148 |
+
"weights": {
|
| 149 |
+
"old_core": 0.3,
|
| 150 |
+
"v3_core": 0.25,
|
| 151 |
+
"v4": 0.15,
|
| 152 |
+
"v5": 0.15,
|
| 153 |
+
"transfer_v9_test": 0.15
|
| 154 |
+
},
|
| 155 |
+
"method_note": "Exact rational integer correct/requested. Outcome-informed product-priority weights chosen by the user after earlier results; this display does not change the frozen scorer or constitute a training gain.",
|
| 156 |
+
"source_checks": {
|
| 157 |
+
"shared_reference_rows_exact": [
|
| 158 |
+
[
|
| 159 |
+
"Nox",
|
| 160 |
+
"kev-4b"
|
| 161 |
+
],
|
| 162 |
+
[
|
| 163 |
+
"Nox",
|
| 164 |
+
"Qwen3.5-4B"
|
| 165 |
+
],
|
| 166 |
+
[
|
| 167 |
+
"Nox",
|
| 168 |
+
"Nox-null-description-candidate"
|
| 169 |
+
],
|
| 170 |
+
[
|
| 171 |
+
"Nox",
|
| 172 |
+
"Qwen3.5-2B"
|
| 173 |
+
],
|
| 174 |
+
[
|
| 175 |
+
"Nox",
|
| 176 |
+
"llm2jev-4b"
|
| 177 |
+
],
|
| 178 |
+
[
|
| 179 |
+
"Nox",
|
| 180 |
+
"Qwen3.5-9B"
|
| 181 |
+
],
|
| 182 |
+
[
|
| 183 |
+
"Nox",
|
| 184 |
+
"Laya-base"
|
| 185 |
+
],
|
| 186 |
+
[
|
| 187 |
+
"Nox",
|
| 188 |
+
"llm2jev-2b"
|
| 189 |
+
],
|
| 190 |
+
[
|
| 191 |
+
"Nox",
|
| 192 |
+
"kev-0.8b"
|
| 193 |
+
],
|
| 194 |
+
[
|
| 195 |
+
"Nox",
|
| 196 |
+
"kev-9b"
|
| 197 |
+
],
|
| 198 |
+
[
|
| 199 |
+
"Nox",
|
| 200 |
+
"Sol"
|
| 201 |
+
],
|
| 202 |
+
[
|
| 203 |
+
"Nox",
|
| 204 |
+
"nimble-9b"
|
| 205 |
+
],
|
| 206 |
+
[
|
| 207 |
+
"Nox",
|
| 208 |
+
"Decider"
|
| 209 |
+
],
|
| 210 |
+
[
|
| 211 |
+
"Nox",
|
| 212 |
+
"Jev"
|
| 213 |
+
],
|
| 214 |
+
[
|
| 215 |
+
"Nox",
|
| 216 |
+
"Laya-multilingual"
|
| 217 |
+
],
|
| 218 |
+
[
|
| 219 |
+
"Lux",
|
| 220 |
+
"kev-4b"
|
| 221 |
+
],
|
| 222 |
+
[
|
| 223 |
+
"Lux",
|
| 224 |
+
"Qwen3.5-4B"
|
| 225 |
+
],
|
| 226 |
+
[
|
| 227 |
+
"Lux",
|
| 228 |
+
"Qwen3.5-2B"
|
| 229 |
+
],
|
| 230 |
+
[
|
| 231 |
+
"Lux",
|
| 232 |
+
"llm2jev-4b"
|
| 233 |
+
],
|
| 234 |
+
[
|
| 235 |
+
"Lux",
|
| 236 |
+
"Qwen3.5-9B"
|
| 237 |
+
],
|
| 238 |
+
[
|
| 239 |
+
"Lux",
|
| 240 |
+
"Laya-base"
|
| 241 |
+
],
|
| 242 |
+
[
|
| 243 |
+
"Lux",
|
| 244 |
+
"llm2jev-2b"
|
| 245 |
+
],
|
| 246 |
+
[
|
| 247 |
+
"Lux",
|
| 248 |
+
"kev-0.8b"
|
| 249 |
+
],
|
| 250 |
+
[
|
| 251 |
+
"Lux",
|
| 252 |
+
"kev-9b"
|
| 253 |
+
],
|
| 254 |
+
[
|
| 255 |
+
"Lux",
|
| 256 |
+
"Sol"
|
| 257 |
+
],
|
| 258 |
+
[
|
| 259 |
+
"Lux",
|
| 260 |
+
"nimble-9b"
|
| 261 |
+
],
|
| 262 |
+
[
|
| 263 |
+
"Lux",
|
| 264 |
+
"Decider"
|
| 265 |
+
],
|
| 266 |
+
[
|
| 267 |
+
"Lux",
|
| 268 |
+
"Jev"
|
| 269 |
+
],
|
| 270 |
+
[
|
| 271 |
+
"Lux",
|
| 272 |
+
"Laya-multilingual"
|
| 273 |
+
],
|
| 274 |
+
[
|
| 275 |
+
"Kai",
|
| 276 |
+
"Laya-multilingual"
|
| 277 |
+
],
|
| 278 |
+
[
|
| 279 |
+
"Kai",
|
| 280 |
+
"Laya-base"
|
| 281 |
+
],
|
| 282 |
+
[
|
| 283 |
+
"Kai",
|
| 284 |
+
"Qwen3.5-2B"
|
| 285 |
+
],
|
| 286 |
+
[
|
| 287 |
+
"Kai",
|
| 288 |
+
"kev-0.8b"
|
| 289 |
+
],
|
| 290 |
+
[
|
| 291 |
+
"Eos",
|
| 292 |
+
"Laya-multilingual"
|
| 293 |
+
],
|
| 294 |
+
[
|
| 295 |
+
"Eos",
|
| 296 |
+
"Laya-base"
|
| 297 |
+
],
|
| 298 |
+
[
|
| 299 |
+
"Eos",
|
| 300 |
+
"Qwen3.5-2B"
|
| 301 |
+
],
|
| 302 |
+
[
|
| 303 |
+
"Eos",
|
| 304 |
+
"kev-0.8b"
|
| 305 |
+
]
|
| 306 |
+
],
|
| 307 |
+
"published_matrices_exact": {
|
| 308 |
+
"Kai": {
|
| 309 |
+
"sha256": "5e50911352fdf356da6f4397a3868a2987cfdfb284ab693a8820dba58bb61f6d",
|
| 310 |
+
"cells_exact": 270
|
| 311 |
},
|
| 312 |
+
"Eos": {
|
| 313 |
+
"sha256": "e4c7e7c8f3a5a9ab0d57538ecba93141c839592b4e0e4cc50aaeb5782a1ca1de",
|
| 314 |
+
"cells_exact": 270
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 315 |
}
|
| 316 |
+
}
|
| 317 |
+
},
|
| 318 |
+
"source_builder_sha256": "6e4d6d175b15960c74b36a2df10c5a099c77dfdc0c3a5517acfbdab501172f36"
|
| 319 |
}
|
release-manifest.json
CHANGED
|
@@ -3,12 +3,17 @@
|
|
| 3 |
"status": "current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1",
|
| 5 |
"readiness_sha256": "8627fe4a471c8588bf17c0913da08692a31af01757ebfb8c5be400f13e21ed7f",
|
| 6 |
-
"model_card_sha256": "
|
| 7 |
-
"repo_id": "llm-semantic-router/Decision-1.0-Sol",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
| 10 |
"files_exclude_this_manifest": true,
|
| 11 |
"files": [
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
{
|
| 13 |
"file": "ATTRIBUTIONS.md",
|
| 14 |
"bytes": 6606,
|
|
@@ -16,8 +21,8 @@
|
|
| 16 |
},
|
| 17 |
{
|
| 18 |
"file": "DIAGNOSTICS.md",
|
| 19 |
-
"bytes":
|
| 20 |
-
"sha256": "
|
| 21 |
},
|
| 22 |
{
|
| 23 |
"file": "Dockerfile.runtime",
|
|
@@ -26,8 +31,8 @@
|
|
| 26 |
},
|
| 27 |
{
|
| 28 |
"file": "EVALUATION.md",
|
| 29 |
-
"bytes":
|
| 30 |
-
"sha256": "
|
| 31 |
},
|
| 32 |
{
|
| 33 |
"file": "LICENSE",
|
|
@@ -36,8 +41,8 @@
|
|
| 36 |
},
|
| 37 |
{
|
| 38 |
"file": "MATERIALS.json",
|
| 39 |
-
"bytes":
|
| 40 |
-
"sha256": "
|
| 41 |
},
|
| 42 |
{
|
| 43 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
@@ -56,8 +61,8 @@
|
|
| 56 |
},
|
| 57 |
{
|
| 58 |
"file": "README.md",
|
| 59 |
-
"bytes":
|
| 60 |
-
"sha256": "
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"file": "RUNTIME-RELEASE.json",
|
|
@@ -76,8 +81,8 @@
|
|
| 76 |
},
|
| 77 |
{
|
| 78 |
"file": "SENSITIVITY.md",
|
| 79 |
-
"bytes":
|
| 80 |
-
"sha256": "
|
| 81 |
},
|
| 82 |
{
|
| 83 |
"file": "SERVING_OPTIMIZATION.json",
|
|
@@ -91,18 +96,18 @@
|
|
| 91 |
},
|
| 92 |
{
|
| 93 |
"file": "TASKS.md",
|
| 94 |
-
"bytes":
|
| 95 |
-
"sha256": "
|
| 96 |
},
|
| 97 |
{
|
| 98 |
"file": "USAGE.md",
|
| 99 |
-
"bytes":
|
| 100 |
-
"sha256": "
|
| 101 |
},
|
| 102 |
{
|
| 103 |
"file": "WEIGHTING.md",
|
| 104 |
-
"bytes":
|
| 105 |
-
"sha256": "
|
| 106 |
},
|
| 107 |
{
|
| 108 |
"file": "assets/architecture.png",
|
|
@@ -241,18 +246,18 @@
|
|
| 241 |
},
|
| 242 |
{
|
| 243 |
"file": "assets/decision-matrix.pdf",
|
| 244 |
-
"bytes":
|
| 245 |
-
"sha256": "
|
| 246 |
},
|
| 247 |
{
|
| 248 |
"file": "assets/decision-matrix.png",
|
| 249 |
-
"bytes":
|
| 250 |
-
"sha256": "
|
| 251 |
},
|
| 252 |
{
|
| 253 |
"file": "assets/decision-matrix.svg",
|
| 254 |
-
"bytes":
|
| 255 |
-
"sha256": "
|
| 256 |
},
|
| 257 |
{
|
| 258 |
"file": "assets/decision-question-scaling-600px.png",
|
|
@@ -276,18 +281,23 @@
|
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"file": "assets/decision-ranking.pdf",
|
| 279 |
-
"bytes":
|
| 280 |
-
"sha256": "
|
| 281 |
},
|
| 282 |
{
|
| 283 |
"file": "assets/decision-ranking.png",
|
| 284 |
-
"bytes":
|
| 285 |
-
"sha256": "
|
| 286 |
},
|
| 287 |
{
|
| 288 |
"file": "assets/decision-ranking.svg",
|
| 289 |
-
"bytes":
|
| 290 |
-
"sha256": "
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 291 |
},
|
| 292 |
{
|
| 293 |
"file": "assets/readout.png",
|
|
@@ -351,8 +361,8 @@
|
|
| 351 |
},
|
| 352 |
{
|
| 353 |
"file": "metrics/benchmark.json",
|
| 354 |
-
"bytes":
|
| 355 |
-
"sha256": "
|
| 356 |
},
|
| 357 |
{
|
| 358 |
"file": "metrics/comparator-coverage.json",
|
|
@@ -361,8 +371,8 @@
|
|
| 361 |
},
|
| 362 |
{
|
| 363 |
"file": "metrics/evaluation-provenance.json",
|
| 364 |
-
"bytes":
|
| 365 |
-
"sha256": "
|
| 366 |
},
|
| 367 |
{
|
| 368 |
"file": "metrics/expanded-quality.json",
|
|
@@ -538,9 +548,11 @@
|
|
| 538 |
"MATERIALS.json": "copy",
|
| 539 |
"metrics/semantic-consistency.json": "copy"
|
| 540 |
},
|
| 541 |
-
"scope": "
|
| 542 |
"release_tag": "v1.3.1",
|
| 543 |
-
"change_kind": "current-
|
| 544 |
-
"previous_main_revision": "
|
| 545 |
-
"previous_release_manifest_sha256": "
|
|
|
|
|
|
|
| 546 |
}
|
|
|
|
| 3 |
"status": "current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1",
|
| 5 |
"readiness_sha256": "8627fe4a471c8588bf17c0913da08692a31af01757ebfb8c5be400f13e21ed7f",
|
| 6 |
+
"model_card_sha256": "cc5fdd26eb3b5b0e88107b4b84fc3e12fc96d30ac9e8fb35cddc32a92df7f4ab",
|
| 7 |
+
"repo_id": "llm-semantic-router/Decision-1.0-Sol-2B",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
| 10 |
"files_exclude_this_manifest": true,
|
| 11 |
"files": [
|
| 12 |
+
{
|
| 13 |
+
"file": ".gitattributes",
|
| 14 |
+
"bytes": 3132,
|
| 15 |
+
"sha256": "8ec513d4464879c383554c84c1acff60508f0748470daa16f475d05ac804e2ba"
|
| 16 |
+
},
|
| 17 |
{
|
| 18 |
"file": "ATTRIBUTIONS.md",
|
| 19 |
"bytes": 6606,
|
|
|
|
| 21 |
},
|
| 22 |
{
|
| 23 |
"file": "DIAGNOSTICS.md",
|
| 24 |
+
"bytes": 6327,
|
| 25 |
+
"sha256": "8db3cd79a98c9801ba237ede456f3337abca1990a02a3b5f755731ff78aead73"
|
| 26 |
},
|
| 27 |
{
|
| 28 |
"file": "Dockerfile.runtime",
|
|
|
|
| 31 |
},
|
| 32 |
{
|
| 33 |
"file": "EVALUATION.md",
|
| 34 |
+
"bytes": 4079,
|
| 35 |
+
"sha256": "ad152906fcd9902135869bee2e4684bf8018a5cc3add1ebf905bd57848bcd05e"
|
| 36 |
},
|
| 37 |
{
|
| 38 |
"file": "LICENSE",
|
|
|
|
| 41 |
},
|
| 42 |
{
|
| 43 |
"file": "MATERIALS.json",
|
| 44 |
+
"bytes": 1864,
|
| 45 |
+
"sha256": "9df7430135259b25d4ff77f082ecf0593bac7baa2359e441053fa8291be656b5"
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
|
|
| 61 |
},
|
| 62 |
{
|
| 63 |
"file": "README.md",
|
| 64 |
+
"bytes": 5871,
|
| 65 |
+
"sha256": "cc5fdd26eb3b5b0e88107b4b84fc3e12fc96d30ac9e8fb35cddc32a92df7f4ab"
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"file": "RUNTIME-RELEASE.json",
|
|
|
|
| 81 |
},
|
| 82 |
{
|
| 83 |
"file": "SENSITIVITY.md",
|
| 84 |
+
"bytes": 866,
|
| 85 |
+
"sha256": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4"
|
| 86 |
},
|
| 87 |
{
|
| 88 |
"file": "SERVING_OPTIMIZATION.json",
|
|
|
|
| 96 |
},
|
| 97 |
{
|
| 98 |
"file": "TASKS.md",
|
| 99 |
+
"bytes": 10386,
|
| 100 |
+
"sha256": "a667bf2e5f50d021f39c89158c84a504afed15bc72a1efc015bd5aa33a8e5476"
|
| 101 |
},
|
| 102 |
{
|
| 103 |
"file": "USAGE.md",
|
| 104 |
+
"bytes": 2886,
|
| 105 |
+
"sha256": "398c56cbb43047629551827fc5c0ba24f4dac2d7b01a5e6e3aeeb85b951706a0"
|
| 106 |
},
|
| 107 |
{
|
| 108 |
"file": "WEIGHTING.md",
|
| 109 |
+
"bytes": 866,
|
| 110 |
+
"sha256": "f8e312d4039c77085298312eaa96bbba413a754d12a55748267db59b7b3aece4"
|
| 111 |
},
|
| 112 |
{
|
| 113 |
"file": "assets/architecture.png",
|
|
|
|
| 246 |
},
|
| 247 |
{
|
| 248 |
"file": "assets/decision-matrix.pdf",
|
| 249 |
+
"bytes": 29409,
|
| 250 |
+
"sha256": "a7e091eb7b4df6e3dfda49b58be3ba046c3e471a9954e3ae6c023e6732dd4bcd"
|
| 251 |
},
|
| 252 |
{
|
| 253 |
"file": "assets/decision-matrix.png",
|
| 254 |
+
"bytes": 376179,
|
| 255 |
+
"sha256": "64e496d7692fe5404cff4664ab3fe93bec7396e19c31e7df3fdb097d50cc9e77"
|
| 256 |
},
|
| 257 |
{
|
| 258 |
"file": "assets/decision-matrix.svg",
|
| 259 |
+
"bytes": 50107,
|
| 260 |
+
"sha256": "5b83c10c82f0a7f0c23278fd91bde9fbfb0710e76ec0e88dc8843394dabea6fc"
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"file": "assets/decision-question-scaling-600px.png",
|
|
|
|
| 281 |
},
|
| 282 |
{
|
| 283 |
"file": "assets/decision-ranking.pdf",
|
| 284 |
+
"bytes": 25379,
|
| 285 |
+
"sha256": "118307de70f311bc6a0787610d76297fccbfd10cadd916ca386f765df084b060"
|
| 286 |
},
|
| 287 |
{
|
| 288 |
"file": "assets/decision-ranking.png",
|
| 289 |
+
"bytes": 248869,
|
| 290 |
+
"sha256": "ebf63842085424faecb8ee9a29e2f0f700d887fd40c499d23a2462b6834b64d6"
|
| 291 |
},
|
| 292 |
{
|
| 293 |
"file": "assets/decision-ranking.svg",
|
| 294 |
+
"bytes": 15728,
|
| 295 |
+
"sha256": "7ae0e61ff6d66073c271d6b0161b0dd9fa15dc2c481ec1162eb720b62aacbc62"
|
| 296 |
+
},
|
| 297 |
+
{
|
| 298 |
+
"file": "assets/decision-sol-2b-header.png",
|
| 299 |
+
"bytes": 2398842,
|
| 300 |
+
"sha256": "9616b7828774561692b642236d74b72de548e2dfc2ae83f1cae2b70bb3b69b08"
|
| 301 |
},
|
| 302 |
{
|
| 303 |
"file": "assets/readout.png",
|
|
|
|
| 361 |
},
|
| 362 |
{
|
| 363 |
"file": "metrics/benchmark.json",
|
| 364 |
+
"bytes": 2569381,
|
| 365 |
+
"sha256": "f25b1a00871e9a75143ae19ef420a80ee18983e07bd5cbafe690c8d61ff95e56"
|
| 366 |
},
|
| 367 |
{
|
| 368 |
"file": "metrics/comparator-coverage.json",
|
|
|
|
| 371 |
},
|
| 372 |
{
|
| 373 |
"file": "metrics/evaluation-provenance.json",
|
| 374 |
+
"bytes": 8432,
|
| 375 |
+
"sha256": "27bd646045e3eab660830507a8a2651ff3a543b93b7ceb651f90571f8f03e7d5"
|
| 376 |
},
|
| 377 |
{
|
| 378 |
"file": "metrics/expanded-quality.json",
|
|
|
|
| 548 |
"MATERIALS.json": "copy",
|
| 549 |
"metrics/semantic-consistency.json": "copy"
|
| 550 |
},
|
| 551 |
+
"scope": "Latest fifteen-model comparison and official TypeSafe SDK examples; weights, runtime, tokenizer and temperature unchanged.",
|
| 552 |
"release_tag": "v1.3.1",
|
| 553 |
+
"change_kind": "current-documents-and-official-sdk-examples-only",
|
| 554 |
+
"previous_main_revision": "3504a1033370fd927d83ba0c6445e8d851269e0e",
|
| 555 |
+
"previous_release_manifest_sha256": "de1d5d56dd93d3d15b4167bfae2772e0a17f190f1ef8e221ef893a5cc2d031de",
|
| 556 |
+
"presentation_amendment_sha256": "038f01d857b372a8236d0ca634e44d6d27095b5937218410b5050d5b9105a4bf",
|
| 557 |
+
"statistics_sha256": "169bdbb413aa3302369d56f814d35641a5b4e1de82ef3ae688c28ca740b3b0dc"
|
| 558 |
}
|