Publish clean Decision model repository
Browse filesSigned-off-by: Xunzhuo <Xunzhuo@users.noreply.huggingface.co>
This view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +0 -6
- ARCHITECTURE.md +1 -1
- BATCH_CAPACITY.json +0 -1015
- DISTRIBUTION_TERMS.md +1 -1
- FINETUNING.md +0 -32
- LICENSING_STATUS.md +3 -3
- PACKAGE_MANIFEST.json +0 -493
- PERFORMANCE.md +0 -66
- README.md +13 -7
- RUNTIME_UPDATE.json +0 -32
- SYSTEM_ONE.md +0 -134
- SYSTEM_ONE_VALIDATION.json +0 -66
- USAGE.md +0 -121
- assets/architecture.pdf +0 -3
- assets/attention-geglu.pdf +0 -3
- assets/attention-geglu.png +0 -3
- assets/candidate-readout.pdf +0 -3
- assets/candidate-readout.svg +0 -121
- assets/residual-layers.pdf +0 -3
- assets/residual-layers.png +0 -3
- config.json +21 -0
- decision_finetune/__init__.py +0 -1
- decision_finetune/__main__.py +0 -80
- decision_finetune/data.py +0 -139
- decision_finetune/metrics.py +0 -61
- decision_finetune/run.py +0 -237
- decision_finetune/state.py +0 -110
- decision_finetune/system_one.py +0 -34
- decision_inference/__init__.py +0 -5
- decision_inference/_auto.py +0 -34
- decision_inference/_grouped.py +0 -16
- decision_inference/_request.py +0 -58
- decision_inference/_system_one.py +0 -217
- decision_inference/profile.py +0 -51
- decision_runtime/__init__.py +0 -11
- decision_runtime/_compat.py +0 -20
- decision_runtime/native.py +0 -152
- decision_runtime/training.py +0 -310
- EVALUATION.md → evaluation/EVALUATION.md +0 -0
- METHODS.md → evaluation/METHODS.md +1 -1
- MIXED_QUESTION_SCALING.json → evaluation/MIXED_QUESTION_SCALING.json +0 -0
- evaluation/PERFORMANCE.md +17 -0
- TECHNICAL_VALIDATION.json → evaluation/TECHNICAL_VALIDATION.json +0 -0
- VALIDATION.md → evaluation/VALIDATION.md +2 -2
- examples/decisions.jsonl +0 -3
- examples/finetune.sh +0 -21
- examples/system-one.json +0 -29
- infer.py +0 -46
- native/INVENTORY.json +0 -1786
- native/MANIFEST.json +0 -125
.gitattributes
CHANGED
|
@@ -1,13 +1,7 @@
|
|
| 1 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 2 |
native/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 3 |
assets/architecture-atlas.pdf filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
assets/architecture.pdf filter=lfs diff=lfs merge=lfs -text
|
| 5 |
assets/architecture.png filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
assets/attention-geglu.pdf filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
assets/attention-geglu.png filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
assets/candidate-readout.pdf filter=lfs diff=lfs merge=lfs -text
|
| 9 |
assets/candidate-readout.png filter=lfs diff=lfs merge=lfs -text
|
| 10 |
assets/decision-lex-header.png filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
assets/residual-layers.pdf filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
assets/residual-layers.png filter=lfs diff=lfs merge=lfs -text
|
| 13 |
assets/mixed-question-scaling.png filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 1 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 2 |
native/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 3 |
assets/architecture-atlas.pdf filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 4 |
assets/architecture.png filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
| 5 |
assets/candidate-readout.png filter=lfs diff=lfs merge=lfs -text
|
| 6 |
assets/decision-lex-header.png filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
| 7 |
assets/mixed-question-scaling.png filter=lfs diff=lfs merge=lfs -text
|
ARCHITECTURE.md
CHANGED
|
@@ -10,6 +10,6 @@ A question and all of its candidate descriptions are encoded jointly with the co
|
|
| 10 |
|
| 11 |
Choice returns a candidate and probabilities; Noul returns the probability of yes; Score returns the ordered distribution and its expected value. Candidate IDs preserve associations outside the token stream. Multiple questions use batches, with repeated context encoding.
|
| 12 |
|
| 13 |
-
The
|
| 14 |
|
| 15 |
[Main SVG](assets/architecture.svg) · [Residual layers](assets/residual-layers.svg) · [Attention and GEGLU](assets/attention-geglu.svg) · [PDF atlas](assets/architecture-atlas.pdf)
|
|
|
|
| 10 |
|
| 11 |
Choice returns a candidate and probabilities; Noul returns the probability of yes; Score returns the ordered distribution and its expected value. Candidate IDs preserve associations outside the token stream. Multiple questions use batches, with repeated context encoding.
|
| 12 |
|
| 13 |
+
The evaluated Decision runtime accepted complete packed inputs of at most 1,024 tokens per question without truncation. A separately distributed runtime must enforce its own supported input limit.
|
| 14 |
|
| 15 |
[Main SVG](assets/architecture.svg) · [Residual layers](assets/residual-layers.svg) · [Attention and GEGLU](assets/attention-geglu.svg) · [PDF atlas](assets/architecture-atlas.pdf)
|
BATCH_CAPACITY.json
DELETED
|
@@ -1,1015 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"complete_input_tokens": 1024,
|
| 3 |
-
"default_entry": "predict_1k unchanged, integer1..8/default8",
|
| 4 |
-
"final_guard_separately_timed": false,
|
| 5 |
-
"limitations": [
|
| 6 |
-
"225 technical row-occurrences/model include fixed repetition; not independent quality support.",
|
| 7 |
-
"The optional padding-protected entry was not separately timed; timing benefits refer to the earlier paired B8/B32 experiment.",
|
| 8 |
-
"No claim of Studio speedup, universal batch invariance or reduced peak memory.",
|
| 9 |
-
"Published weights/runtime and Studio were not changed; this result does not automatically publish."
|
| 10 |
-
],
|
| 11 |
-
"model": "Decision-1.0-Lex",
|
| 12 |
-
"native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 13 |
-
"optional_entry": "predict_auto_1k",
|
| 14 |
-
"package_validation": {
|
| 15 |
-
"buffers_unchanged": 66,
|
| 16 |
-
"fallback_exact": true,
|
| 17 |
-
"forwards": 46,
|
| 18 |
-
"late_invalid_and_empty_calls": 0,
|
| 19 |
-
"numerics": {
|
| 20 |
-
"by_type": {
|
| 21 |
-
"Choice": {
|
| 22 |
-
"flips": 0,
|
| 23 |
-
"rows": 116
|
| 24 |
-
},
|
| 25 |
-
"Noul": {
|
| 26 |
-
"flips": 0,
|
| 27 |
-
"rows": 52
|
| 28 |
-
},
|
| 29 |
-
"Score": {
|
| 30 |
-
"flips": 0,
|
| 31 |
-
"rows": 57
|
| 32 |
-
}
|
| 33 |
-
},
|
| 34 |
-
"checks": {
|
| 35 |
-
"logits": true,
|
| 36 |
-
"no_hard_flips": true,
|
| 37 |
-
"probabilities": true,
|
| 38 |
-
"score_normalized": true
|
| 39 |
-
},
|
| 40 |
-
"full_rank_changes": 0,
|
| 41 |
-
"hard_flips": 0,
|
| 42 |
-
"max_logit_error": 1.3217329978942871e-05,
|
| 43 |
-
"max_probability_error": 1.4901161193847656e-06,
|
| 44 |
-
"rows": 225,
|
| 45 |
-
"strict_fp32_contract_pass": true
|
| 46 |
-
},
|
| 47 |
-
"parameters_unchanged": 489,
|
| 48 |
-
"row_occurrences": 225,
|
| 49 |
-
"source_closure_sha256": "d2b92dbf73abb3f80905e6c7e3445b48aa69df9b5a912d6a1de09ae7af56835c",
|
| 50 |
-
"source_complete_sha256": "1f93cd36e3ccd80950679712f05e193bf0cbe9d7fc31c2df30d0dcae0a7b6e10",
|
| 51 |
-
"source_plan_sha256": "dd43fd06ff147d4a2abd0c6c11df3dae9b7c73552365ae81f3cbc913f00d3586",
|
| 52 |
-
"source_result_sha256": "c60a00a2aa571f01dff8a71bfc658b23419d0b7141c642eb9ea43a1766f3de5b"
|
| 53 |
-
},
|
| 54 |
-
"performance": {
|
| 55 |
-
"bootstrap": {
|
| 56 |
-
"estimator": "median(cap32)/median(B8)",
|
| 57 |
-
"interval": "percentile95",
|
| 58 |
-
"order_strata": "five AB and five BA blocks resampled separately",
|
| 59 |
-
"replicates": 2000,
|
| 60 |
-
"seed": 20260922,
|
| 61 |
-
"unit": "whole paired five-request-per-mode block"
|
| 62 |
-
},
|
| 63 |
-
"forwards": 1605,
|
| 64 |
-
"numerics": {
|
| 65 |
-
"by_type": {
|
| 66 |
-
"Choice": {
|
| 67 |
-
"flips": 0,
|
| 68 |
-
"rows": 247
|
| 69 |
-
},
|
| 70 |
-
"Noul": {
|
| 71 |
-
"flips": 0,
|
| 72 |
-
"rows": 53
|
| 73 |
-
},
|
| 74 |
-
"Score": {
|
| 75 |
-
"flips": 0,
|
| 76 |
-
"rows": 57
|
| 77 |
-
}
|
| 78 |
-
},
|
| 79 |
-
"checks": {
|
| 80 |
-
"logits": true,
|
| 81 |
-
"no_hard_flips": true,
|
| 82 |
-
"probabilities": true,
|
| 83 |
-
"score_normalized": true
|
| 84 |
-
},
|
| 85 |
-
"full_rank_changes": 0,
|
| 86 |
-
"hard_flips": 0,
|
| 87 |
-
"max_logit_error": 1.5974044799804688e-05,
|
| 88 |
-
"max_probability_error": 1.9073486328125e-06,
|
| 89 |
-
"rows": 357,
|
| 90 |
-
"strict_fp32_contract_pass": true
|
| 91 |
-
},
|
| 92 |
-
"original_checks": {
|
| 93 |
-
"all_p95_no_more_5pct_regression": true,
|
| 94 |
-
"choice-l256-q16_p50_ratio_CI_below1": true,
|
| 95 |
-
"choice-l256-q16_p50_reduction_15pct": false,
|
| 96 |
-
"choice-l256-q32_p50_ratio_CI_below1": true,
|
| 97 |
-
"choice-l256-q32_p50_reduction_15pct": true,
|
| 98 |
-
"multicontext-choice-32_p50_ratio_CI_below1": true,
|
| 99 |
-
"multicontext-choice-32_p50_reduction_15pct": true,
|
| 100 |
-
"q1_q8_p50_no_more_5pct_regression": true
|
| 101 |
-
},
|
| 102 |
-
"original_utility_pass": false,
|
| 103 |
-
"points": {
|
| 104 |
-
"choice-l1024-q1": {
|
| 105 |
-
"blocks": 10,
|
| 106 |
-
"clocks": {
|
| 107 |
-
"device_forward_ms": {
|
| 108 |
-
"p50": {
|
| 109 |
-
"b8_ms": 10.556464672088623,
|
| 110 |
-
"cap32_ms": 10.566385269165039,
|
| 111 |
-
"difference_ms": 0.009920597076416016,
|
| 112 |
-
"difference_ms_percentile95": [
|
| 113 |
-
-0.012168836593627929,
|
| 114 |
-
0.029139995574951172
|
| 115 |
-
],
|
| 116 |
-
"ratio": 1.00093976509983,
|
| 117 |
-
"ratio_percentile95": [
|
| 118 |
-
0.9988491046005276,
|
| 119 |
-
1.0027634771344436
|
| 120 |
-
]
|
| 121 |
-
},
|
| 122 |
-
"p95": {
|
| 123 |
-
"b8_ms": 10.661027860641479,
|
| 124 |
-
"cap32_ms": 10.661649703979492,
|
| 125 |
-
"difference_ms": 0.0006218433380134059,
|
| 126 |
-
"difference_ms_percentile95": [
|
| 127 |
-
-0.03483018875122035,
|
| 128 |
-
0.0834893226623521
|
| 129 |
-
],
|
| 130 |
-
"ratio": 1.0000583286476823,
|
| 131 |
-
"ratio_percentile95": [
|
| 132 |
-
0.9967408769534367,
|
| 133 |
-
1.007839916995741
|
| 134 |
-
]
|
| 135 |
-
}
|
| 136 |
-
},
|
| 137 |
-
"predict_wall_ms": {
|
| 138 |
-
"p50": {
|
| 139 |
-
"b8_ms": 18.094859493430704,
|
| 140 |
-
"cap32_ms": 18.080359499435872,
|
| 141 |
-
"difference_ms": -0.014499993994832039,
|
| 142 |
-
"difference_ms_percentile95": [
|
| 143 |
-
-0.040014972910284996,
|
| 144 |
-
0.004804728087037776
|
| 145 |
-
],
|
| 146 |
-
"ratio": 0.9991986677763319,
|
| 147 |
-
"ratio_percentile95": [
|
| 148 |
-
0.99779205443173,
|
| 149 |
-
1.00026553617198
|
| 150 |
-
]
|
| 151 |
-
},
|
| 152 |
-
"p95": {
|
| 153 |
-
"b8_ms": 18.238159001339227,
|
| 154 |
-
"cap32_ms": 18.20715542708058,
|
| 155 |
-
"difference_ms": -0.03100357425864786,
|
| 156 |
-
"difference_ms_percentile95": [
|
| 157 |
-
-0.14951042539905757,
|
| 158 |
-
0.26860943471547216
|
| 159 |
-
],
|
| 160 |
-
"ratio": 0.998300071062196,
|
| 161 |
-
"ratio_percentile95": [
|
| 162 |
-
0.9918458693139535,
|
| 163 |
-
1.0147638224804787
|
| 164 |
-
]
|
| 165 |
-
}
|
| 166 |
-
},
|
| 167 |
-
"synchronized_forward_wall_ms": {
|
| 168 |
-
"p50": {
|
| 169 |
-
"b8_ms": 10.582302988041192,
|
| 170 |
-
"cap32_ms": 10.592097998596728,
|
| 171 |
-
"difference_ms": 0.009795010555535555,
|
| 172 |
-
"difference_ms_percentile95": [
|
| 173 |
-
-0.011939986143261194,
|
| 174 |
-
0.02943403160315936
|
| 175 |
-
],
|
| 176 |
-
"ratio": 1.0009256029208957,
|
| 177 |
-
"ratio_percentile95": [
|
| 178 |
-
0.9988732997077118,
|
| 179 |
-
1.0027851328214283
|
| 180 |
-
]
|
| 181 |
-
},
|
| 182 |
-
"p95": {
|
| 183 |
-
"b8_ms": 10.686860964051448,
|
| 184 |
-
"cap32_ms": 10.687752513331361,
|
| 185 |
-
"difference_ms": 0.0008915492799133062,
|
| 186 |
-
"difference_ms_percentile95": [
|
| 187 |
-
-0.034992871223948896,
|
| 188 |
-
0.0834054546430707
|
| 189 |
-
],
|
| 190 |
-
"ratio": 1.0000834248038701,
|
| 191 |
-
"ratio_percentile95": [
|
| 192 |
-
0.9967335955665599,
|
| 193 |
-
1.0078132144378036
|
| 194 |
-
]
|
| 195 |
-
}
|
| 196 |
-
}
|
| 197 |
-
},
|
| 198 |
-
"memory": {
|
| 199 |
-
"b8": {
|
| 200 |
-
"peak_allocated_bytes": 2481425408,
|
| 201 |
-
"peak_incremental_allocated_bytes": 53504512,
|
| 202 |
-
"peak_reserved_bytes_shared_cache": 2518679552
|
| 203 |
-
},
|
| 204 |
-
"cap32": {
|
| 205 |
-
"peak_allocated_bytes": 2481425408,
|
| 206 |
-
"peak_incremental_allocated_bytes": 53504512,
|
| 207 |
-
"peak_reserved_bytes_shared_cache": 2518679552
|
| 208 |
-
}
|
| 209 |
-
},
|
| 210 |
-
"order_strata": {
|
| 211 |
-
"AB": 5,
|
| 212 |
-
"BA": 5
|
| 213 |
-
},
|
| 214 |
-
"replicates": 2000,
|
| 215 |
-
"timed_requests_per_mode": 50
|
| 216 |
-
},
|
| 217 |
-
"choice-l1024-q32": {
|
| 218 |
-
"blocks": 10,
|
| 219 |
-
"clocks": {
|
| 220 |
-
"device_forward_ms": {
|
| 221 |
-
"p50": {
|
| 222 |
-
"b8_ms": 207.20293426513672,
|
| 223 |
-
"cap32_ms": 185.14539337158203,
|
| 224 |
-
"difference_ms": -22.057540893554688,
|
| 225 |
-
"difference_ms_percentile95": [
|
| 226 |
-
-22.2838134765625,
|
| 227 |
-
-21.868221282958984
|
| 228 |
-
],
|
| 229 |
-
"ratio": 0.8935461943539377,
|
| 230 |
-
"ratio_percentile95": [
|
| 231 |
-
0.8923934919811013,
|
| 232 |
-
0.8943251695490113
|
| 233 |
-
]
|
| 234 |
-
},
|
| 235 |
-
"p95": {
|
| 236 |
-
"b8_ms": 207.52946338653564,
|
| 237 |
-
"cap32_ms": 185.77925491333008,
|
| 238 |
-
"difference_ms": -21.75020847320556,
|
| 239 |
-
"difference_ms_percentile95": [
|
| 240 |
-
-22.083112335205072,
|
| 241 |
-
-20.28153457641602
|
| 242 |
-
],
|
| 243 |
-
"ratio": 0.8951945997533153,
|
| 244 |
-
"ratio_percentile95": [
|
| 245 |
-
0.8936076948402836,
|
| 246 |
-
0.9022578623509261
|
| 247 |
-
]
|
| 248 |
-
}
|
| 249 |
-
},
|
| 250 |
-
"predict_wall_ms": {
|
| 251 |
-
"p50": {
|
| 252 |
-
"b8_ms": 242.37409501802176,
|
| 253 |
-
"cap32_ms": 217.62683399720117,
|
| 254 |
-
"difference_ms": -24.747261020820588,
|
| 255 |
-
"difference_ms_percentile95": [
|
| 256 |
-
-25.16183662446565,
|
| 257 |
-
-24.575807009387063
|
| 258 |
-
],
|
| 259 |
-
"ratio": 0.8978964273431098,
|
| 260 |
-
"ratio_percentile95": [
|
| 261 |
-
0.8961426985350305,
|
| 262 |
-
0.8985948091719478
|
| 263 |
-
]
|
| 264 |
-
},
|
| 265 |
-
"p95": {
|
| 266 |
-
"b8_ms": 243.72325239528436,
|
| 267 |
-
"cap32_ms": 218.37091989000328,
|
| 268 |
-
"difference_ms": -25.352332505281083,
|
| 269 |
-
"difference_ms_percentile95": [
|
| 270 |
-
-26.005168532719836,
|
| 271 |
-
-23.27408672135789
|
| 272 |
-
],
|
| 273 |
-
"ratio": 0.8959790161335808,
|
| 274 |
-
"ratio_percentile95": [
|
| 275 |
-
0.893447528320472,
|
| 276 |
-
0.9042453228782044
|
| 277 |
-
]
|
| 278 |
-
}
|
| 279 |
-
},
|
| 280 |
-
"synchronized_forward_wall_ms": {
|
| 281 |
-
"p50": {
|
| 282 |
-
"b8_ms": 207.32908698846586,
|
| 283 |
-
"cap32_ms": 185.17211652942933,
|
| 284 |
-
"difference_ms": -22.15697045903653,
|
| 285 |
-
"difference_ms_percentile95": [
|
| 286 |
-
-22.36067791818641,
|
| 287 |
-
-21.93638199241832
|
| 288 |
-
],
|
| 289 |
-
"ratio": 0.8931313942444112,
|
| 290 |
-
"ratio_percentile95": [
|
| 291 |
-
0.8921173649815071,
|
| 292 |
-
0.8940271149855218
|
| 293 |
-
]
|
| 294 |
-
},
|
| 295 |
-
"p95": {
|
| 296 |
-
"b8_ms": 207.64006946410518,
|
| 297 |
-
"cap32_ms": 185.8059855090687,
|
| 298 |
-
"difference_ms": -21.834083955036476,
|
| 299 |
-
"difference_ms_percentile95": [
|
| 300 |
-
-22.15991159901023,
|
| 301 |
-
-20.358516700798646
|
| 302 |
-
],
|
| 303 |
-
"ratio": 0.8948464811662427,
|
| 304 |
-
"ratio_percentile95": [
|
| 305 |
-
0.8932978401930624,
|
| 306 |
-
0.9019359263090917
|
| 307 |
-
]
|
| 308 |
-
}
|
| 309 |
-
}
|
| 310 |
-
},
|
| 311 |
-
"memory": {
|
| 312 |
-
"b8": {
|
| 313 |
-
"peak_allocated_bytes": 2847486976,
|
| 314 |
-
"peak_incremental_allocated_bytes": 419566080,
|
| 315 |
-
"peak_reserved_bytes_shared_cache": 5786042368
|
| 316 |
-
},
|
| 317 |
-
"cap32": {
|
| 318 |
-
"peak_allocated_bytes": 4106146304,
|
| 319 |
-
"peak_incremental_allocated_bytes": 1678225408,
|
| 320 |
-
"peak_reserved_bytes_shared_cache": 5786042368
|
| 321 |
-
}
|
| 322 |
-
},
|
| 323 |
-
"order_strata": {
|
| 324 |
-
"AB": 5,
|
| 325 |
-
"BA": 5
|
| 326 |
-
},
|
| 327 |
-
"replicates": 2000,
|
| 328 |
-
"timed_requests_per_mode": 50
|
| 329 |
-
},
|
| 330 |
-
"choice-l1024-q8": {
|
| 331 |
-
"blocks": 10,
|
| 332 |
-
"clocks": {
|
| 333 |
-
"device_forward_ms": {
|
| 334 |
-
"p50": {
|
| 335 |
-
"b8_ms": 52.41438102722168,
|
| 336 |
-
"cap32_ms": 52.384565353393555,
|
| 337 |
-
"difference_ms": -0.029815673828125,
|
| 338 |
-
"difference_ms_percentile95": [
|
| 339 |
-
-0.058559417724609375,
|
| 340 |
-
0.017009687423706003
|
| 341 |
-
],
|
| 342 |
-
"ratio": 0.9994311547089979,
|
| 343 |
-
"ratio_percentile95": [
|
| 344 |
-
0.9988832372979973,
|
| 345 |
-
1.0003247540003477
|
| 346 |
-
]
|
| 347 |
-
},
|
| 348 |
-
"p95": {
|
| 349 |
-
"b8_ms": 52.496505928039554,
|
| 350 |
-
"cap32_ms": 52.48605175018311,
|
| 351 |
-
"difference_ms": -0.010454177856445312,
|
| 352 |
-
"difference_ms_percentile95": [
|
| 353 |
-
-0.0377529144287152,
|
| 354 |
-
0.2926578712463391
|
| 355 |
-
],
|
| 356 |
-
"ratio": 0.9998008595491903,
|
| 357 |
-
"ratio_percentile95": [
|
| 358 |
-
0.999281028941322,
|
| 359 |
-
1.0055735259439396
|
| 360 |
-
]
|
| 361 |
-
}
|
| 362 |
-
},
|
| 363 |
-
"predict_wall_ms": {
|
| 364 |
-
"p50": {
|
| 365 |
-
"b8_ms": 65.74378252844326,
|
| 366 |
-
"cap32_ms": 65.72207299177535,
|
| 367 |
-
"difference_ms": -0.0217095366679132,
|
| 368 |
-
"difference_ms_percentile95": [
|
| 369 |
-
-0.06555735380970873,
|
| 370 |
-
0.025685253785923123
|
| 371 |
-
],
|
| 372 |
-
"ratio": 0.9996697857069827,
|
| 373 |
-
"ratio_percentile95": [
|
| 374 |
-
0.9990025610937415,
|
| 375 |
-
1.0003905551675811
|
| 376 |
-
]
|
| 377 |
-
},
|
| 378 |
-
"p95": {
|
| 379 |
-
"b8_ms": 65.88881341158412,
|
| 380 |
-
"cap32_ms": 66.52213780616876,
|
| 381 |
-
"difference_ms": 0.6333243945846334,
|
| 382 |
-
"difference_ms_percentile95": [
|
| 383 |
-
0.0010406914952909609,
|
| 384 |
-
1.8839879310689867
|
| 385 |
-
],
|
| 386 |
-
"ratio": 1.0096120169994334,
|
| 387 |
-
"ratio_percentile95": [
|
| 388 |
-
1.0000158055055854,
|
| 389 |
-
1.0285920667545798
|
| 390 |
-
]
|
| 391 |
-
}
|
| 392 |
-
},
|
| 393 |
-
"synchronized_forward_wall_ms": {
|
| 394 |
-
"p50": {
|
| 395 |
-
"b8_ms": 52.44034499628469,
|
| 396 |
-
"cap32_ms": 52.414220495847985,
|
| 397 |
-
"difference_ms": -0.02612450043670833,
|
| 398 |
-
"difference_ms_percentile95": [
|
| 399 |
-
-0.0563944922760129,
|
| 400 |
-
0.019360042642802
|
| 401 |
-
],
|
| 402 |
-
"ratio": 0.9995018243980173,
|
| 403 |
-
"ratio_percentile95": [
|
| 404 |
-
0.9989250543350029,
|
| 405 |
-
1.0003694742542426
|
| 406 |
-
]
|
| 407 |
-
},
|
| 408 |
-
"p95": {
|
| 409 |
-
"b8_ms": 52.52235445950646,
|
| 410 |
-
"cap32_ms": 52.51286399143282,
|
| 411 |
-
"difference_ms": -0.009490468073636293,
|
| 412 |
-
"difference_ms_percentile95": [
|
| 413 |
-
-0.03923154145013541,
|
| 414 |
-
0.29656406215508296
|
| 415 |
-
],
|
| 416 |
-
"ratio": 0.999819306118865,
|
| 417 |
-
"ratio_percentile95": [
|
| 418 |
-
0.9992532410156931,
|
| 419 |
-
1.0056450304951374
|
| 420 |
-
]
|
| 421 |
-
}
|
| 422 |
-
}
|
| 423 |
-
},
|
| 424 |
-
"memory": {
|
| 425 |
-
"b8": {
|
| 426 |
-
"peak_allocated_bytes": 2847485952,
|
| 427 |
-
"peak_incremental_allocated_bytes": 419565056,
|
| 428 |
-
"peak_reserved_bytes_shared_cache": 3168796672
|
| 429 |
-
},
|
| 430 |
-
"cap32": {
|
| 431 |
-
"peak_allocated_bytes": 2847485952,
|
| 432 |
-
"peak_incremental_allocated_bytes": 419565056,
|
| 433 |
-
"peak_reserved_bytes_shared_cache": 3168796672
|
| 434 |
-
}
|
| 435 |
-
},
|
| 436 |
-
"order_strata": {
|
| 437 |
-
"AB": 5,
|
| 438 |
-
"BA": 5
|
| 439 |
-
},
|
| 440 |
-
"replicates": 2000,
|
| 441 |
-
"timed_requests_per_mode": 50
|
| 442 |
-
},
|
| 443 |
-
"choice-l256-q1": {
|
| 444 |
-
"blocks": 10,
|
| 445 |
-
"clocks": {
|
| 446 |
-
"device_forward_ms": {
|
| 447 |
-
"p50": {
|
| 448 |
-
"b8_ms": 6.985419034957886,
|
| 449 |
-
"cap32_ms": 6.974940061569214,
|
| 450 |
-
"difference_ms": -0.010478973388671875,
|
| 451 |
-
"difference_ms_percentile95": [
|
| 452 |
-
-0.022879636287689208,
|
| 453 |
-
0.010601043701171875
|
| 454 |
-
],
|
| 455 |
-
"ratio": 0.998499879057186,
|
| 456 |
-
"ratio_percentile95": [
|
| 457 |
-
0.9967288947675313,
|
| 458 |
-
1.0015206567387591
|
| 459 |
-
]
|
| 460 |
-
},
|
| 461 |
-
"p95": {
|
| 462 |
-
"b8_ms": 7.041934943199157,
|
| 463 |
-
"cap32_ms": 7.064635586738587,
|
| 464 |
-
"difference_ms": 0.022700643539429244,
|
| 465 |
-
"difference_ms_percentile95": [
|
| 466 |
-
-0.024966001510620117,
|
| 467 |
-
0.06541033685207343
|
| 468 |
-
],
|
| 469 |
-
"ratio": 1.003223637213711,
|
| 470 |
-
"ratio_percentile95": [
|
| 471 |
-
0.996459745978026,
|
| 472 |
-
1.0093022538220333
|
| 473 |
-
]
|
| 474 |
-
}
|
| 475 |
-
},
|
| 476 |
-
"predict_wall_ms": {
|
| 477 |
-
"p50": {
|
| 478 |
-
"b8_ms": 13.920774013968185,
|
| 479 |
-
"cap32_ms": 13.921903999289498,
|
| 480 |
-
"difference_ms": 0.0011299853213131428,
|
| 481 |
-
"difference_ms_percentile95": [
|
| 482 |
-
-0.034331005736021325,
|
| 483 |
-
0.02732866196311077
|
| 484 |
-
],
|
| 485 |
-
"ratio": 1.0000811725928587,
|
| 486 |
-
"ratio_percentile95": [
|
| 487 |
-
0.9975367080592278,
|
| 488 |
-
1.0019657993273317
|
| 489 |
-
]
|
| 490 |
-
},
|
| 491 |
-
"p95": {
|
| 492 |
-
"b8_ms": 14.007961520110257,
|
| 493 |
-
"cap32_ms": 14.045515967882238,
|
| 494 |
-
"difference_ms": 0.03755444777198136,
|
| 495 |
-
"difference_ms_percentile95": [
|
| 496 |
-
-0.008158999844454229,
|
| 497 |
-
0.12227892875671387
|
| 498 |
-
],
|
| 499 |
-
"ratio": 1.0026809359604585,
|
| 500 |
-
"ratio_percentile95": [
|
| 501 |
-
0.9994190895790983,
|
| 502 |
-
1.008731703172336
|
| 503 |
-
]
|
| 504 |
-
}
|
| 505 |
-
},
|
| 506 |
-
"synchronized_forward_wall_ms": {
|
| 507 |
-
"p50": {
|
| 508 |
-
"b8_ms": 7.020359509624541,
|
| 509 |
-
"cap32_ms": 7.008063985267654,
|
| 510 |
-
"difference_ms": -0.012295524356886744,
|
| 511 |
-
"difference_ms_percentile95": [
|
| 512 |
-
-0.02304499503225088,
|
| 513 |
-
0.007555005140602589
|
| 514 |
-
],
|
| 515 |
-
"ratio": 0.9982485904973911,
|
| 516 |
-
"ratio_percentile95": [
|
| 517 |
-
0.9967212883959398,
|
| 518 |
-
1.0010784238669923
|
| 519 |
-
]
|
| 520 |
-
},
|
| 521 |
-
"p95": {
|
| 522 |
-
"b8_ms": 7.073595060501248,
|
| 523 |
-
"cap32_ms": 7.098784536356106,
|
| 524 |
-
"difference_ms": 0.025189475854858756,
|
| 525 |
-
"difference_ms_percentile95": [
|
| 526 |
-
-0.021360546816140413,
|
| 527 |
-
0.06566304175066759
|
| 528 |
-
],
|
| 529 |
-
"ratio": 1.0035610570918196,
|
| 530 |
-
"ratio_percentile95": [
|
| 531 |
-
0.9969902183560572,
|
| 532 |
-
1.0092935466259783
|
| 533 |
-
]
|
| 534 |
-
}
|
| 535 |
-
}
|
| 536 |
-
},
|
| 537 |
-
"memory": {
|
| 538 |
-
"b8": {
|
| 539 |
-
"peak_allocated_bytes": 2438808064,
|
| 540 |
-
"peak_incremental_allocated_bytes": 10887168,
|
| 541 |
-
"peak_reserved_bytes_shared_cache": 2472542208
|
| 542 |
-
},
|
| 543 |
-
"cap32": {
|
| 544 |
-
"peak_allocated_bytes": 2438808064,
|
| 545 |
-
"peak_incremental_allocated_bytes": 10887168,
|
| 546 |
-
"peak_reserved_bytes_shared_cache": 2472542208
|
| 547 |
-
}
|
| 548 |
-
},
|
| 549 |
-
"order_strata": {
|
| 550 |
-
"AB": 5,
|
| 551 |
-
"BA": 5
|
| 552 |
-
},
|
| 553 |
-
"replicates": 2000,
|
| 554 |
-
"timed_requests_per_mode": 50
|
| 555 |
-
},
|
| 556 |
-
"choice-l256-q16": {
|
| 557 |
-
"blocks": 10,
|
| 558 |
-
"clocks": {
|
| 559 |
-
"device_forward_ms": {
|
| 560 |
-
"p50": {
|
| 561 |
-
"b8_ms": 26.544885635375977,
|
| 562 |
-
"cap32_ms": 21.927971839904785,
|
| 563 |
-
"difference_ms": -4.616913795471191,
|
| 564 |
-
"difference_ms_percentile95": [
|
| 565 |
-
-4.648672580718994,
|
| 566 |
-
-4.596253395080566
|
| 567 |
-
],
|
| 568 |
-
"ratio": 0.8260714376814531,
|
| 569 |
-
"ratio_percentile95": [
|
| 570 |
-
0.8248601346258421,
|
| 571 |
-
0.8267891393024169
|
| 572 |
-
]
|
| 573 |
-
},
|
| 574 |
-
"p95": {
|
| 575 |
-
"b8_ms": 26.64890079498291,
|
| 576 |
-
"cap32_ms": 21.992683506011964,
|
| 577 |
-
"difference_ms": -4.6562172889709466,
|
| 578 |
-
"difference_ms_percentile95": [
|
| 579 |
-
-5.2368529319763155,
|
| 580 |
-
-4.580370044708253
|
| 581 |
-
],
|
| 582 |
-
"ratio": 0.8252754466387764,
|
| 583 |
-
"ratio_percentile95": [
|
| 584 |
-
0.8076432393515695,
|
| 585 |
-
0.8278200215346474
|
| 586 |
-
]
|
| 587 |
-
}
|
| 588 |
-
},
|
| 589 |
-
"predict_wall_ms": {
|
| 590 |
-
"p50": {
|
| 591 |
-
"b8_ms": 38.862298999447376,
|
| 592 |
-
"cap32_ms": 33.38937502121553,
|
| 593 |
-
"difference_ms": -5.472923978231847,
|
| 594 |
-
"difference_ms_percentile95": [
|
| 595 |
-
-5.532156530534849,
|
| 596 |
-
-5.428209843375953
|
| 597 |
-
],
|
| 598 |
-
"ratio": 0.8591713789678354,
|
| 599 |
-
"ratio_percentile95": [
|
| 600 |
-
0.8578086140267409,
|
| 601 |
-
0.8603004537961789
|
| 602 |
-
]
|
| 603 |
-
},
|
| 604 |
-
"p95": {
|
| 605 |
-
"b8_ms": 39.418455920531414,
|
| 606 |
-
"cap32_ms": 33.553791453596205,
|
| 607 |
-
"difference_ms": -5.86466446693521,
|
| 608 |
-
"difference_ms_percentile95": [
|
| 609 |
-
-118.63445058697778,
|
| 610 |
-
-5.444934926345013
|
| 611 |
-
],
|
| 612 |
-
"ratio": 0.8512203400671371,
|
| 613 |
-
"ratio_percentile95": [
|
| 614 |
-
0.22047959173207704,
|
| 615 |
-
0.8605261779182554
|
| 616 |
-
]
|
| 617 |
-
}
|
| 618 |
-
},
|
| 619 |
-
"synchronized_forward_wall_ms": {
|
| 620 |
-
"p50": {
|
| 621 |
-
"b8_ms": 26.595044997520745,
|
| 622 |
-
"cap32_ms": 21.95358253084123,
|
| 623 |
-
"difference_ms": -4.6414624666795135,
|
| 624 |
-
"difference_ms_percentile95": [
|
| 625 |
-
-4.673314448882593,
|
| 626 |
-
-4.62049143970944
|
| 627 |
-
],
|
| 628 |
-
"ratio": 0.8254764198702351,
|
| 629 |
-
"ratio_percentile95": [
|
| 630 |
-
0.8242482464394296,
|
| 631 |
-
0.8262020510690964
|
| 632 |
-
]
|
| 633 |
-
},
|
| 634 |
-
"p95": {
|
| 635 |
-
"b8_ms": 26.699701527832076,
|
| 636 |
-
"cap32_ms": 22.01862498477567,
|
| 637 |
-
"difference_ms": -4.681076543056406,
|
| 638 |
-
"difference_ms_percentile95": [
|
| 639 |
-
-5.2700117637868935,
|
| 640 |
-
-4.603384018992074
|
| 641 |
-
],
|
| 642 |
-
"ratio": 0.8246768212679532,
|
| 643 |
-
"ratio_percentile95": [
|
| 644 |
-
0.8068452917646015,
|
| 645 |
-
0.8272821360169423
|
| 646 |
-
]
|
| 647 |
-
}
|
| 648 |
-
}
|
| 649 |
-
},
|
| 650 |
-
"memory": {
|
| 651 |
-
"b8": {
|
| 652 |
-
"peak_allocated_bytes": 2516791808,
|
| 653 |
-
"peak_incremental_allocated_bytes": 88870912,
|
| 654 |
-
"peak_reserved_bytes_shared_cache": 2791309312
|
| 655 |
-
},
|
| 656 |
-
"cap32": {
|
| 657 |
-
"peak_allocated_bytes": 2604008448,
|
| 658 |
-
"peak_incremental_allocated_bytes": 176087552,
|
| 659 |
-
"peak_reserved_bytes_shared_cache": 2791309312
|
| 660 |
-
}
|
| 661 |
-
},
|
| 662 |
-
"order_strata": {
|
| 663 |
-
"AB": 5,
|
| 664 |
-
"BA": 5
|
| 665 |
-
},
|
| 666 |
-
"replicates": 2000,
|
| 667 |
-
"timed_requests_per_mode": 50
|
| 668 |
-
},
|
| 669 |
-
"choice-l256-q32": {
|
| 670 |
-
"blocks": 10,
|
| 671 |
-
"clocks": {
|
| 672 |
-
"device_forward_ms": {
|
| 673 |
-
"p50": {
|
| 674 |
-
"b8_ms": 52.45026874542236,
|
| 675 |
-
"cap32_ms": 38.73085975646973,
|
| 676 |
-
"difference_ms": -13.719408988952637,
|
| 677 |
-
"difference_ms_percentile95": [
|
| 678 |
-
-13.757874083518981,
|
| 679 |
-
-13.673646450042725
|
| 680 |
-
],
|
| 681 |
-
"ratio": 0.7384301488417062,
|
| 682 |
-
"ratio_percentile95": [
|
| 683 |
-
0.737865882587114,
|
| 684 |
-
0.739298436956481
|
| 685 |
-
]
|
| 686 |
-
},
|
| 687 |
-
"p95": {
|
| 688 |
-
"b8_ms": 52.81323375701904,
|
| 689 |
-
"cap32_ms": 43.84300689697263,
|
| 690 |
-
"difference_ms": -8.97022686004641,
|
| 691 |
-
"difference_ms_percentile95": [
|
| 692 |
-
-25.975542593002217,
|
| 693 |
-
-5.001253128051758
|
| 694 |
-
],
|
| 695 |
-
"ratio": 0.8301519103845021,
|
| 696 |
-
"ratio_percentile95": [
|
| 697 |
-
0.7010933474185724,
|
| 698 |
-
0.9055233060373526
|
| 699 |
-
]
|
| 700 |
-
}
|
| 701 |
-
},
|
| 702 |
-
"predict_wall_ms": {
|
| 703 |
-
"p50": {
|
| 704 |
-
"b8_ms": 71.03448649286292,
|
| 705 |
-
"cap32_ms": 54.82690001372248,
|
| 706 |
-
"difference_ms": -16.207586479140446,
|
| 707 |
-
"difference_ms_percentile95": [
|
| 708 |
-
-16.303276085091056,
|
| 709 |
-
-16.163301013875753
|
| 710 |
-
],
|
| 711 |
-
"ratio": 0.771834959618257,
|
| 712 |
-
"ratio_percentile95": [
|
| 713 |
-
0.7706202855890361,
|
| 714 |
-
0.772370733763932
|
| 715 |
-
]
|
| 716 |
-
},
|
| 717 |
-
"p95": {
|
| 718 |
-
"b8_ms": 71.4557929430157,
|
| 719 |
-
"cap32_ms": 60.150451309164026,
|
| 720 |
-
"difference_ms": -11.305341633851668,
|
| 721 |
-
"difference_ms_percentile95": [
|
| 722 |
-
-28.958630622946544,
|
| 723 |
-
-8.23459104867652
|
| 724 |
-
],
|
| 725 |
-
"ratio": 0.8417855128573353,
|
| 726 |
-
"ratio_percentile95": [
|
| 727 |
-
0.728043189474086,
|
| 728 |
-
0.8878310636443568
|
| 729 |
-
]
|
| 730 |
-
}
|
| 731 |
-
},
|
| 732 |
-
"synchronized_forward_wall_ms": {
|
| 733 |
-
"p50": {
|
| 734 |
-
"b8_ms": 52.54925854387693,
|
| 735 |
-
"cap32_ms": 38.762504496844485,
|
| 736 |
-
"difference_ms": -13.786754047032446,
|
| 737 |
-
"difference_ms_percentile95": [
|
| 738 |
-
-13.831284875050187,
|
| 739 |
-
-13.745294068939984
|
| 740 |
-
],
|
| 741 |
-
"ratio": 0.7376413211326103,
|
| 742 |
-
"ratio_percentile95": [
|
| 743 |
-
0.7369902659433524,
|
| 744 |
-
0.738419261821391
|
| 745 |
-
]
|
| 746 |
-
},
|
| 747 |
-
"p95": {
|
| 748 |
-
"b8_ms": 52.91530743124895,
|
| 749 |
-
"cap32_ms": 43.87024541210846,
|
| 750 |
-
"difference_ms": -9.045062019140488,
|
| 751 |
-
"difference_ms_percentile95": [
|
| 752 |
-
-26.08439096948122,
|
| 753 |
-
-5.122330039739609
|
| 754 |
-
],
|
| 755 |
-
"ratio": 0.8290653034399842,
|
| 756 |
-
"ratio_percentile95": [
|
| 757 |
-
0.7003076207316908,
|
| 758 |
-
0.9035067818750958
|
| 759 |
-
]
|
| 760 |
-
}
|
| 761 |
-
}
|
| 762 |
-
},
|
| 763 |
-
"memory": {
|
| 764 |
-
"b8": {
|
| 765 |
-
"peak_allocated_bytes": 2516447744,
|
| 766 |
-
"peak_incremental_allocated_bytes": 88526848,
|
| 767 |
-
"peak_reserved_bytes_shared_cache": 3160408064
|
| 768 |
-
},
|
| 769 |
-
"cap32": {
|
| 770 |
-
"peak_allocated_bytes": 2776091136,
|
| 771 |
-
"peak_incremental_allocated_bytes": 348170240,
|
| 772 |
-
"peak_reserved_bytes_shared_cache": 3160408064
|
| 773 |
-
}
|
| 774 |
-
},
|
| 775 |
-
"order_strata": {
|
| 776 |
-
"AB": 5,
|
| 777 |
-
"BA": 5
|
| 778 |
-
},
|
| 779 |
-
"replicates": 2000,
|
| 780 |
-
"timed_requests_per_mode": 50
|
| 781 |
-
},
|
| 782 |
-
"choice-l256-q8": {
|
| 783 |
-
"blocks": 10,
|
| 784 |
-
"clocks": {
|
| 785 |
-
"device_forward_ms": {
|
| 786 |
-
"p50": {
|
| 787 |
-
"b8_ms": 13.475781917572021,
|
| 788 |
-
"cap32_ms": 13.454821586608887,
|
| 789 |
-
"difference_ms": -0.020960330963134766,
|
| 790 |
-
"difference_ms_percentile95": [
|
| 791 |
-
-0.03996642827987671,
|
| 792 |
-
0.0022802352905273438
|
| 793 |
-
],
|
| 794 |
-
"ratio": 0.9984445925964561,
|
| 795 |
-
"ratio_percentile95": [
|
| 796 |
-
0.9970365481188822,
|
| 797 |
-
1.0001694271325683
|
| 798 |
-
]
|
| 799 |
-
},
|
| 800 |
-
"p95": {
|
| 801 |
-
"b8_ms": 13.56228952407837,
|
| 802 |
-
"cap32_ms": 13.560133504867554,
|
| 803 |
-
"difference_ms": -0.0021560192108154297,
|
| 804 |
-
"difference_ms_percentile95": [
|
| 805 |
-
-0.0447998046875,
|
| 806 |
-
0.044429731369017844
|
| 807 |
-
],
|
| 808 |
-
"ratio": 0.9998410283745243,
|
| 809 |
-
"ratio_percentile95": [
|
| 810 |
-
0.9966977480632591,
|
| 811 |
-
1.0032788231911733
|
| 812 |
-
]
|
| 813 |
-
}
|
| 814 |
-
},
|
| 815 |
-
"predict_wall_ms": {
|
| 816 |
-
"p50": {
|
| 817 |
-
"b8_ms": 22.647707461146638,
|
| 818 |
-
"cap32_ms": 22.604158468311653,
|
| 819 |
-
"difference_ms": -0.043548992834985256,
|
| 820 |
-
"difference_ms_percentile95": [
|
| 821 |
-
-0.07484605885110795,
|
| 822 |
-
-0.013918994227424264
|
| 823 |
-
],
|
| 824 |
-
"ratio": 0.9980771125329265,
|
| 825 |
-
"ratio_percentile95": [
|
| 826 |
-
0.9966953876066275,
|
| 827 |
-
0.999385067302049
|
| 828 |
-
]
|
| 829 |
-
},
|
| 830 |
-
"p95": {
|
| 831 |
-
"b8_ms": 22.747458069352433,
|
| 832 |
-
"cap32_ms": 22.772549570072442,
|
| 833 |
-
"difference_ms": 0.025091500720009208,
|
| 834 |
-
"difference_ms_percentile95": [
|
| 835 |
-
-0.08666022906254511,
|
| 836 |
-
0.27172756381332874
|
| 837 |
-
],
|
| 838 |
-
"ratio": 1.0011030463554877,
|
| 839 |
-
"ratio_percentile95": [
|
| 840 |
-
0.9962034198121077,
|
| 841 |
-
1.011953229139044
|
| 842 |
-
]
|
| 843 |
-
}
|
| 844 |
-
},
|
| 845 |
-
"synchronized_forward_wall_ms": {
|
| 846 |
-
"p50": {
|
| 847 |
-
"b8_ms": 13.501687004463747,
|
| 848 |
-
"cap32_ms": 13.480431021889672,
|
| 849 |
-
"difference_ms": -0.02125598257407546,
|
| 850 |
-
"difference_ms_percentile95": [
|
| 851 |
-
-0.03631441213656217,
|
| 852 |
-
0.00041483872337264694
|
| 853 |
-
],
|
| 854 |
-
"ratio": 0.9984256795045651,
|
| 855 |
-
"ratio_percentile95": [
|
| 856 |
-
0.9973128364072793,
|
| 857 |
-
1.000030753064779
|
| 858 |
-
]
|
| 859 |
-
},
|
| 860 |
-
"p95": {
|
| 861 |
-
"b8_ms": 13.590919968555681,
|
| 862 |
-
"cap32_ms": 13.590195949655026,
|
| 863 |
-
"difference_ms": -0.00072401890065521,
|
| 864 |
-
"difference_ms_percentile95": [
|
| 865 |
-
-0.04874391604971606,
|
| 866 |
-
0.04526954144239426
|
| 867 |
-
],
|
| 868 |
-
"ratio": 0.9999467277489434,
|
| 869 |
-
"ratio_percentile95": [
|
| 870 |
-
0.9964148455762563,
|
| 871 |
-
1.0033364560799924
|
| 872 |
-
]
|
| 873 |
-
}
|
| 874 |
-
}
|
| 875 |
-
},
|
| 876 |
-
"memory": {
|
| 877 |
-
"b8": {
|
| 878 |
-
"peak_allocated_bytes": 2516446720,
|
| 879 |
-
"peak_incremental_allocated_bytes": 88525824,
|
| 880 |
-
"peak_reserved_bytes_shared_cache": 2594177024
|
| 881 |
-
},
|
| 882 |
-
"cap32": {
|
| 883 |
-
"peak_allocated_bytes": 2516446720,
|
| 884 |
-
"peak_incremental_allocated_bytes": 88525824,
|
| 885 |
-
"peak_reserved_bytes_shared_cache": 2594177024
|
| 886 |
-
}
|
| 887 |
-
},
|
| 888 |
-
"order_strata": {
|
| 889 |
-
"AB": 5,
|
| 890 |
-
"BA": 5
|
| 891 |
-
},
|
| 892 |
-
"replicates": 2000,
|
| 893 |
-
"timed_requests_per_mode": 50
|
| 894 |
-
},
|
| 895 |
-
"multicontext-choice-32": {
|
| 896 |
-
"blocks": 10,
|
| 897 |
-
"clocks": {
|
| 898 |
-
"device_forward_ms": {
|
| 899 |
-
"p50": {
|
| 900 |
-
"b8_ms": 28.839591026306152,
|
| 901 |
-
"cap32_ms": 15.876487731933594,
|
| 902 |
-
"difference_ms": -12.963103294372559,
|
| 903 |
-
"difference_ms_percentile95": [
|
| 904 |
-
-13.00510287284851,
|
| 905 |
-
-12.939247131347656
|
| 906 |
-
],
|
| 907 |
-
"ratio": 0.5505101552047597,
|
| 908 |
-
"ratio_percentile95": [
|
| 909 |
-
0.5495901472067748,
|
| 910 |
-
0.5512636790172465
|
| 911 |
-
]
|
| 912 |
-
},
|
| 913 |
-
"p95": {
|
| 914 |
-
"b8_ms": 28.963391041755678,
|
| 915 |
-
"cap32_ms": 15.983575963974,
|
| 916 |
-
"difference_ms": -12.979815077781678,
|
| 917 |
-
"difference_ms_percentile95": [
|
| 918 |
-
-13.066112565994263,
|
| 919 |
-
-12.911408408880234
|
| 920 |
-
],
|
| 921 |
-
"ratio": 0.5518544406948394,
|
| 922 |
-
"ratio_percentile95": [
|
| 923 |
-
0.5498390139393741,
|
| 924 |
-
0.5537668232778035
|
| 925 |
-
]
|
| 926 |
-
}
|
| 927 |
-
},
|
| 928 |
-
"predict_wall_ms": {
|
| 929 |
-
"p50": {
|
| 930 |
-
"b8_ms": 44.08348348806612,
|
| 931 |
-
"cap32_ms": 28.755783481756225,
|
| 932 |
-
"difference_ms": -15.327700006309897,
|
| 933 |
-
"difference_ms_percentile95": [
|
| 934 |
-
-15.348734974395484,
|
| 935 |
-
-15.245991060510278
|
| 936 |
-
],
|
| 937 |
-
"ratio": 0.6523028854909056,
|
| 938 |
-
"ratio_percentile95": [
|
| 939 |
-
0.6518971153285599,
|
| 940 |
-
0.6537317276429896
|
| 941 |
-
]
|
| 942 |
-
},
|
| 943 |
-
"p95": {
|
| 944 |
-
"b8_ms": 44.253829002263956,
|
| 945 |
-
"cap32_ms": 28.868743509519845,
|
| 946 |
-
"difference_ms": -15.38508549274411,
|
| 947 |
-
"difference_ms_percentile95": [
|
| 948 |
-
-15.46525440440746,
|
| 949 |
-
-15.249868051614612
|
| 950 |
-
],
|
| 951 |
-
"ratio": 0.6523445351597252,
|
| 952 |
-
"ratio_percentile95": [
|
| 953 |
-
0.6507656214790302,
|
| 954 |
-
0.655196504823658
|
| 955 |
-
]
|
| 956 |
-
}
|
| 957 |
-
},
|
| 958 |
-
"synchronized_forward_wall_ms": {
|
| 959 |
-
"p50": {
|
| 960 |
-
"b8_ms": 28.93799147568643,
|
| 961 |
-
"cap32_ms": 15.90245749684982,
|
| 962 |
-
"difference_ms": -13.035533978836611,
|
| 963 |
-
"difference_ms_percentile95": [
|
| 964 |
-
-13.076403469312936,
|
| 965 |
-
-13.010836963076144
|
| 966 |
-
],
|
| 967 |
-
"ratio": 0.5495356341579889,
|
| 968 |
-
"ratio_percentile95": [
|
| 969 |
-
0.5486493944100644,
|
| 970 |
-
0.5503878064714475
|
| 971 |
-
]
|
| 972 |
-
},
|
| 973 |
-
"p95": {
|
| 974 |
-
"b8_ms": 29.059543382027186,
|
| 975 |
-
"cap32_ms": 16.009619031683542,
|
| 976 |
-
"difference_ms": -13.049924350343645,
|
| 977 |
-
"difference_ms_percentile95": [
|
| 978 |
-
-13.136312982533127,
|
| 979 |
-
-12.980406277420116
|
| 980 |
-
],
|
| 981 |
-
"ratio": 0.5509246591116503,
|
| 982 |
-
"ratio_percentile95": [
|
| 983 |
-
0.5489527496966714,
|
| 984 |
-
0.5528477239894252
|
| 985 |
-
]
|
| 986 |
-
}
|
| 987 |
-
}
|
| 988 |
-
},
|
| 989 |
-
"memory": {
|
| 990 |
-
"b8": {
|
| 991 |
-
"peak_allocated_bytes": 2465764864,
|
| 992 |
-
"peak_incremental_allocated_bytes": 37843968,
|
| 993 |
-
"peak_reserved_bytes_shared_cache": 2713714688
|
| 994 |
-
},
|
| 995 |
-
"cap32": {
|
| 996 |
-
"peak_allocated_bytes": 2578972672,
|
| 997 |
-
"peak_incremental_allocated_bytes": 151051776,
|
| 998 |
-
"peak_reserved_bytes_shared_cache": 2713714688
|
| 999 |
-
}
|
| 1000 |
-
},
|
| 1001 |
-
"order_strata": {
|
| 1002 |
-
"AB": 5,
|
| 1003 |
-
"BA": 5
|
| 1004 |
-
},
|
| 1005 |
-
"replicates": 2000,
|
| 1006 |
-
"timed_requests_per_mode": 50
|
| 1007 |
-
}
|
| 1008 |
-
},
|
| 1009 |
-
"source_plan_sha256": "6df07d17e4481acd200b886cbb85112ba6b773320f6c542cc5d7ddcab8480a35",
|
| 1010 |
-
"source_public_aggregate_sha256": "0251e0ba6aeabafafbb759fefdeeec42b6e498f5fa51136adf3dfd78b5a57d88"
|
| 1011 |
-
},
|
| 1012 |
-
"precision": "fp32",
|
| 1013 |
-
"studio_performance_claim": false,
|
| 1014 |
-
"weights_unchanged": true
|
| 1015 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
DISTRIBUTION_TERMS.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
|
| 3 |
This repository is public and ungated. No Hugging Face approval or access form is required.
|
| 4 |
|
| 5 |
-
Decision's
|
| 6 |
|
| 7 |
The inherited tokenizer is distributed with the [Gemma Terms of Use](LICENSES/gemma/GEMMA_TERMS.html), including the incorporated [Gemma Prohibited Use Policy](LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html). By accessing, using or distributing the covered tokenizer material, you agree to comply with those terms and restrictions. Those documents are incorporated into this distribution agreement for that material. Redistributors must pass on the applicable agreement, restrictions and required notices, and identify their modifications as required by the upstream terms.
|
| 8 |
|
|
|
|
| 2 |
|
| 3 |
This repository is public and ungated. No Hugging Face approval or access form is required.
|
| 4 |
|
| 5 |
+
Decision's model-weight and documentation contributions are provided under the included [Apache License 2.0](LICENSE). Retained third-party material remains subject to its original license and notices; the project grant does not replace those conditions.
|
| 6 |
|
| 7 |
The inherited tokenizer is distributed with the [Gemma Terms of Use](LICENSES/gemma/GEMMA_TERMS.html), including the incorporated [Gemma Prohibited Use Policy](LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html). By accessing, using or distributing the covered tokenizer material, you agree to comply with those terms and restrictions. Those documents are incorporated into this distribution agreement for that material. Redistributors must pass on the applicable agreement, restrictions and required notices, and identify their modifications as required by the upstream terms.
|
| 8 |
|
FINETUNING.md
DELETED
|
@@ -1,32 +0,0 @@
|
|
| 1 |
-
# Fine-tune on your data
|
| 2 |
-
|
| 3 |
-
Already using System One requests? [Convert the same state/questions/criteria into training rows](SYSTEM_ONE.md#fine-tune-with-the-same-inputs) with separate hard or soft targets and shared source-component IDs.
|
| 4 |
-
|
| 5 |
-
The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
|
| 6 |
-
|
| 7 |
-
Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
|
| 8 |
-
|
| 9 |
-
| Type | Target |
|
| 10 |
-
|---|---|
|
| 11 |
-
| Choice | `{"choice_id":"candidate-id"}` or `{"probabilities":[0.2,0.8]}` |
|
| 12 |
-
| Noul | `{"probability":0.8}`; hard labels are 0 or 1 |
|
| 13 |
-
| Score | `{"probabilities":[0.1,0.2,0.7]}` in increasing-value level order; hard labels are one-hot |
|
| 14 |
-
|
| 15 |
-
Every row retains `id`, full `state_text` and a complete typed `question`. Probability vectors sum to one. A scalar Score is not silently converted into a distribution.
|
| 16 |
-
|
| 17 |
-
```bash
|
| 18 |
-
TRAIN=/data/train.jsonl DEV=/data/dev.jsonl OUTPUT=/runs/my-decision \
|
| 19 |
-
ROCR_VISIBLE_DEVICES=0 bash examples/finetune.sh
|
| 20 |
-
```
|
| 21 |
-
|
| 22 |
-
`examples/finetune.sh` is the editable example configuration; it uses actual supported flags, not a nonexistent `--config` option. Its four-epoch recipe uses logical batch 64 / microbatch 8, BF16 autocast with FP32 parameters/loss, AdamW, encoder LR 2.5e-5 / head LR 1e-4, clip 1, CE plus 0.1 ordinal RPS for Score, and a cosine floor of 1e-6 after 10% warmup. Logical tails use their actual row count. These are example defaults, not a guarantee of improvement.
|
| 23 |
-
|
| 24 |
-
The default selector minimizes equal-type macro soft NLL at step 0 and epoch ends. The separate `--selection hard-accuracy` policy requires an explicit `hard_target_id` on every DEV row, uses row hard accuracy then soft NLL then the earlier step, and never derives hard gold from soft targets. Noul hard selection uses `p_yes >= .5`. Record the selector and cadence used in any model comparison; these defaults are starting points rather than a guarantee of task quality.
|
| 25 |
-
|
| 26 |
-
Successful runs export the selected native bundle and report its manifest in `COMPLETE.json`. Selecting step 0 means no adopted adaptation. Immutable checkpoints include model, AdamW and RNG state; allow substantial disk space. For an interrupted run, use exactly the same original arguments and output plus `--resume <checkpoint> --resume-sha256 <hash>`. Read `LATEST.json` for the last completely saved boundary. `--stop-after-step N` is an optional clean pause; it does not restart or shorten the LR horizon.
|
| 27 |
-
|
| 28 |
-
The included source underwent a bounded AMD continuous-versus-resume check and independent native reload. That interface evidence is separate from dataset quality or long training stability. Validate the quality of your selected model on an appropriately isolated evaluation set; no automatic release or adoption is performed.
|
| 29 |
-
|
| 30 |
-
## Lex checkpoint provenance
|
| 31 |
-
|
| 32 |
-
The editable example above is a downstream fine-tuning recipe, not the exact Lex release recipe. Lex was refit from Kai on all 6,000 original typed-decisions TRAIN decisions using eight epochs, fixed final step 750 and blended original hard/soft cross-entropy. It used no final-refit DEV selector, no warmup and no RPS term. See [METHODS.md](METHODS.md) and [TRAINING_PROVENANCE.json](TRAINING_PROVENANCE.json). The public CLI does not provide a named `blend50` switch; explicitly form normalized blended targets if using that objective. Reusing Lex as an initializer is a new experiment.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
LICENSING_STATUS.md
CHANGED
|
@@ -1,14 +1,14 @@
|
|
| 1 |
# Decision-1.0-Lex licensing
|
| 2 |
|
| 3 |
-
**Decision's model-weight contributions
|
| 4 |
|
| 5 |
| Material | License and attribution |
|
| 6 |
|---|---|
|
| 7 |
| Decision contributions | [Apache License 2.0](LICENSE) |
|
| 8 |
| Upstream Vela/mmBERT contributions | Retained [MIT terms](LICENSES/Upstream-MIT.txt) and author/source attribution in [NOTICE](NOTICE) |
|
| 9 |
-
|
|
| 10 |
| Inherited Gemma-origin tokenizer material | Retained [tokenizer terms](LICENSES/gemma/TOKENIZER_TERMS.md), agreement, use policy and required Notice |
|
| 11 |
|
| 12 |
The Apache-2.0 project license does not replace third-party terms. In particular, the tokenizer inherited through Vela/mmBERT retains its upstream conditions; this does not describe Decision's independently trained encoder weights as Gemma weights. [Distribution terms](DISTRIBUTION_TERMS.md) preserve that component's requirements without a Hugging Face access gate.
|
| 13 |
|
| 14 |
-
|
|
|
|
| 1 |
# Decision-1.0-Lex licensing
|
| 2 |
|
| 3 |
+
**Decision's model-weight contributions and documentation are licensed under Apache 2.0.** The repository is public and ungated: no account approval or acceptance form is required to download it.
|
| 4 |
|
| 5 |
| Material | License and attribution |
|
| 6 |
|---|---|
|
| 7 |
| Decision contributions | [Apache License 2.0](LICENSE) |
|
| 8 |
| Upstream Vela/mmBERT contributions | Retained [MIT terms](LICENSES/Upstream-MIT.txt) and author/source attribution in [NOTICE](NOTICE) |
|
| 9 |
+
| Historical adapted ModernBERT/Transformers runtime code (no longer bundled) | Retained [Apache 2.0 terms](LICENSES/Transformers-Apache-2.0.txt) and modification notices |
|
| 10 |
| Inherited Gemma-origin tokenizer material | Retained [tokenizer terms](LICENSES/gemma/TOKENIZER_TERMS.md), agreement, use policy and required Notice |
|
| 11 |
|
| 12 |
The Apache-2.0 project license does not replace third-party terms. In particular, the tokenizer inherited through Vela/mmBERT retains its upstream conditions; this does not describe Decision's independently trained encoder weights as Gemma weights. [Distribution terms](DISTRIBUTION_TERMS.md) preserve that component's requirements without a Hugging Face access gate.
|
| 13 |
|
| 14 |
+
This model-only package retains the original weight and tokenizer objects but no longer bundles executable runtime code. Upstream sources and pinned revisions remain recorded in [NOTICE_SOURCES.json](NOTICE_SOURCES.json). Training/evaluation data, Laya weights and Laya SDK source are not redistributed; separately installed dependencies retain their own licenses.
|
PACKAGE_MANIFEST.json
DELETED
|
@@ -1,493 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"bytes_excluding_this_manifest": 2327567235,
|
| 3 |
-
"display_name": "Decision-1.0-Lex-0.6B",
|
| 4 |
-
"file_count_excluding_this_manifest": 107,
|
| 5 |
-
"files": {
|
| 6 |
-
".gitattributes": {
|
| 7 |
-
"bytes": 753,
|
| 8 |
-
"sha256": "d401c47d3da9ae93830e327b487cde895438378a1f90620a70b5c7b3e5d3ed92"
|
| 9 |
-
},
|
| 10 |
-
"ARCHITECTURE.md": {
|
| 11 |
-
"bytes": 1330,
|
| 12 |
-
"sha256": "82ceb4a1b1d6372ec15cce9b2902dc52477383f5bc52ba4206aad93efce097ff"
|
| 13 |
-
},
|
| 14 |
-
"BATCH_CAPACITY.json": {
|
| 15 |
-
"bytes": 33186,
|
| 16 |
-
"sha256": "ee74265a3232807f94115476847bf63194e4eb68d9d8532c25525ecd1f9c9e85"
|
| 17 |
-
},
|
| 18 |
-
"DISTRIBUTION_TERMS.md": {
|
| 19 |
-
"bytes": 1251,
|
| 20 |
-
"sha256": "54a5895871af33845ee939f2e8f37b7f821cee8e89ff25ec1fd78e9cf83e9c63"
|
| 21 |
-
},
|
| 22 |
-
"EVALUATION.md": {
|
| 23 |
-
"bytes": 2813,
|
| 24 |
-
"sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
|
| 25 |
-
},
|
| 26 |
-
"FINETUNING.md": {
|
| 27 |
-
"bytes": 3842,
|
| 28 |
-
"sha256": "2874eedec3b19462552a35e0dc99af29c456b3a2f8096cc61344b92e4cc6db41"
|
| 29 |
-
},
|
| 30 |
-
"LICENSE": {
|
| 31 |
-
"bytes": 11358,
|
| 32 |
-
"sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30"
|
| 33 |
-
},
|
| 34 |
-
"LICENSES/Transformers-Apache-2.0.txt": {
|
| 35 |
-
"bytes": 11418,
|
| 36 |
-
"sha256": "77fd4710def9ec3c0f6225800e0235f15a425abd4a8b03559127fcd782612049"
|
| 37 |
-
},
|
| 38 |
-
"LICENSES/Upstream-MIT.txt": {
|
| 39 |
-
"bytes": 1191,
|
| 40 |
-
"sha256": "3ec44d2f046b27986e28e3b4705110a33b139e6392ef3868ff594a0453311038"
|
| 41 |
-
},
|
| 42 |
-
"LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html": {
|
| 43 |
-
"bytes": 4636,
|
| 44 |
-
"sha256": "7e50ae0c7a6386ab51d36aa868e1d20a4fb28cbcab38778a5dbdac6a445f5a96"
|
| 45 |
-
},
|
| 46 |
-
"LICENSES/gemma/GEMMA_TERMS.html": {
|
| 47 |
-
"bytes": 13510,
|
| 48 |
-
"sha256": "01ccbe2f6504a6a1364db0e5d6d91ed37309435229c322a29b20ee5daebebc1a"
|
| 49 |
-
},
|
| 50 |
-
"LICENSES/gemma/Notice": {
|
| 51 |
-
"bytes": 1088,
|
| 52 |
-
"sha256": "a7b8b1b625d4a140b4dd55ee4479db7d64b5cda047f2cbc1be9d21516b470921"
|
| 53 |
-
},
|
| 54 |
-
"LICENSES/gemma/TOKENIZER_TERMS.md": {
|
| 55 |
-
"bytes": 1345,
|
| 56 |
-
"sha256": "1ca30bf8cc8d1054f78ad3481ada9cf39a137cec9e383f18c3a9f0f56b8d4241"
|
| 57 |
-
},
|
| 58 |
-
"LICENSING_STATUS.md": {
|
| 59 |
-
"bytes": 1530,
|
| 60 |
-
"sha256": "f82c305be9e8502c872efae0903b965ac1128109848dd2167e3fc59601cd9da0"
|
| 61 |
-
},
|
| 62 |
-
"METHODS.md": {
|
| 63 |
-
"bytes": 2543,
|
| 64 |
-
"sha256": "a7dd6dec55dfe463cc0211c3dffa114b88bb95db90816b70c18b62b5d7be7ad1"
|
| 65 |
-
},
|
| 66 |
-
"MIXED_QUESTION_SCALING.json": {
|
| 67 |
-
"bytes": 32763,
|
| 68 |
-
"sha256": "ed4a9e7a9aac0385f8a4516694a6d9ea843aacefde6c7189b9980bac86c9d773"
|
| 69 |
-
},
|
| 70 |
-
"NOTICE": {
|
| 71 |
-
"bytes": 4371,
|
| 72 |
-
"sha256": "e3f71eb2a262084fa572ccbf9b65110a4feb9d8c625dc26e792beba282cdf032"
|
| 73 |
-
},
|
| 74 |
-
"NOTICE_SOURCES.json": {
|
| 75 |
-
"bytes": 9279,
|
| 76 |
-
"sha256": "2e5936d3ea196e00968cc37b7662a6afbe8cfd6fd11d04b9fecba554b1e1e37b"
|
| 77 |
-
},
|
| 78 |
-
"PERFORMANCE.md": {
|
| 79 |
-
"bytes": 6351,
|
| 80 |
-
"sha256": "0f5c84a90b2c82be6146b1b7032a9f836e5858d59e29bd6c94f016be36d8b3b6"
|
| 81 |
-
},
|
| 82 |
-
"README.md": {
|
| 83 |
-
"bytes": 6230,
|
| 84 |
-
"sha256": "6ff9eb7a6ac0c6b90b017f497d75bd3366c9c598ce3c5f813dfa0f192d49b1ec"
|
| 85 |
-
},
|
| 86 |
-
"RUNTIME_UPDATE.json": {
|
| 87 |
-
"bytes": 1087,
|
| 88 |
-
"sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
|
| 89 |
-
},
|
| 90 |
-
"SYSTEM_ONE.md": {
|
| 91 |
-
"bytes": 6177,
|
| 92 |
-
"sha256": "c59c5d1fbab6c18033b1339701aadddc34653f9fb93f79aee93534bb34d66648"
|
| 93 |
-
},
|
| 94 |
-
"SYSTEM_ONE_VALIDATION.json": {
|
| 95 |
-
"bytes": 2583,
|
| 96 |
-
"sha256": "24c28a2b8c448507f26e116a7f6a6d4e40fb8eaae0384e6bee421a105e415e0a"
|
| 97 |
-
},
|
| 98 |
-
"TECHNICAL_VALIDATION.json": {
|
| 99 |
-
"bytes": 1547,
|
| 100 |
-
"sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
|
| 101 |
-
},
|
| 102 |
-
"TRAINING_ATTRIBUTION.md": {
|
| 103 |
-
"bytes": 5816,
|
| 104 |
-
"sha256": "305e5435ccb5abb6a6ad6deb30b762b508261a4179a0f9756605838d2c477961"
|
| 105 |
-
},
|
| 106 |
-
"TRAINING_PROVENANCE.json": {
|
| 107 |
-
"bytes": 2310,
|
| 108 |
-
"sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
|
| 109 |
-
},
|
| 110 |
-
"USAGE.md": {
|
| 111 |
-
"bytes": 8308,
|
| 112 |
-
"sha256": "f7b6fe124f34667480f324f125d92d72e35045ea043c8b45da7184b9c7a28dcb"
|
| 113 |
-
},
|
| 114 |
-
"VALIDATION.md": {
|
| 115 |
-
"bytes": 2120,
|
| 116 |
-
"sha256": "85e1f45ddf0d5e2650490eefdbd50f81ccccd7f414d78a8075206da49ef0043d"
|
| 117 |
-
},
|
| 118 |
-
"assets/architecture-atlas.pdf": {
|
| 119 |
-
"bytes": 635270,
|
| 120 |
-
"sha256": "ff64805a39d16be56d401166ef4d310da86305058bb2c532751754c8c34d49ee"
|
| 121 |
-
},
|
| 122 |
-
"assets/architecture.pdf": {
|
| 123 |
-
"bytes": 173099,
|
| 124 |
-
"sha256": "6aa6895f5708cda7d935991cf9c1179d9e54e22a32459968a052acf05f62a14a"
|
| 125 |
-
},
|
| 126 |
-
"assets/architecture.png": {
|
| 127 |
-
"bytes": 432419,
|
| 128 |
-
"sha256": "daeed79f92a66a2655eb0594f59af4e90424a84ba9519eeaf2ef5bb8b7949fa2"
|
| 129 |
-
},
|
| 130 |
-
"assets/architecture.svg": {
|
| 131 |
-
"bytes": 23430,
|
| 132 |
-
"sha256": "7c3ae44a25e7d9035e6a9635608f7e491a22d9be6991243b85d85f30874e563c"
|
| 133 |
-
},
|
| 134 |
-
"assets/attention-geglu.pdf": {
|
| 135 |
-
"bytes": 172797,
|
| 136 |
-
"sha256": "43b3485510971d29e66bc17b30b4bffe934ad66078598cc3e2efd63ae9b9bc7c"
|
| 137 |
-
},
|
| 138 |
-
"assets/attention-geglu.png": {
|
| 139 |
-
"bytes": 265587,
|
| 140 |
-
"sha256": "de7d88e3cf036d3b8cba38df47d5b3182faecb9233e675104cee7c75dd7fd8c5"
|
| 141 |
-
},
|
| 142 |
-
"assets/attention-geglu.svg": {
|
| 143 |
-
"bytes": 10256,
|
| 144 |
-
"sha256": "0ae02a36ebe97d0b6eed7b23e140c1fdc85e710c00015c518cae67ed5ec06cea"
|
| 145 |
-
},
|
| 146 |
-
"assets/candidate-readout.pdf": {
|
| 147 |
-
"bytes": 156814,
|
| 148 |
-
"sha256": "56b9e06a343e9044c38cfb8d1ba5ba121bdc7d763db5ce45b927aacdcb00c995"
|
| 149 |
-
},
|
| 150 |
-
"assets/candidate-readout.png": {
|
| 151 |
-
"bytes": 309736,
|
| 152 |
-
"sha256": "ad4be86a2aa0136cd7f528c317340c38988b932aab653f25a5a57a9a3cbad834"
|
| 153 |
-
},
|
| 154 |
-
"assets/candidate-readout.svg": {
|
| 155 |
-
"bytes": 9825,
|
| 156 |
-
"sha256": "1d3709818d5b08014ba63042e5287f84f1c97a735a85ec49108ec6e9e77151ee"
|
| 157 |
-
},
|
| 158 |
-
"assets/decision-lex-header.png": {
|
| 159 |
-
"bytes": 2188963,
|
| 160 |
-
"sha256": "9980c6b901aff89b69e9cc6de6de40408b95c7378f18e931401198f42d32e40b"
|
| 161 |
-
},
|
| 162 |
-
"assets/mixed-question-scaling.pdf": {
|
| 163 |
-
"bytes": 27984,
|
| 164 |
-
"sha256": "e1261c998cdcdf6b72f57b4df4cb56cbf47fa521210c4778a0577e2f3a47c1cf"
|
| 165 |
-
},
|
| 166 |
-
"assets/mixed-question-scaling.png": {
|
| 167 |
-
"bytes": 121576,
|
| 168 |
-
"sha256": "4efadb4888c43122d461ca432081f2b5bd67c350f9605940b96f156829d65a54"
|
| 169 |
-
},
|
| 170 |
-
"assets/mixed-question-scaling.svg": {
|
| 171 |
-
"bytes": 14623,
|
| 172 |
-
"sha256": "0fcea2c3e7fd1c649a1dd7acf206c0d41e4f6fcab5e51abd838d6bb18a2560ff"
|
| 173 |
-
},
|
| 174 |
-
"assets/residual-layers.pdf": {
|
| 175 |
-
"bytes": 131144,
|
| 176 |
-
"sha256": "2c5b50bb840ae115a1264b66d535a1048172e891a33d02d507c193525cc45630"
|
| 177 |
-
},
|
| 178 |
-
"assets/residual-layers.png": {
|
| 179 |
-
"bytes": 309369,
|
| 180 |
-
"sha256": "f3ee6441fe1016e2435256644675b08bff99cec72d521e49a78f6cbe5b2dc3bf"
|
| 181 |
-
},
|
| 182 |
-
"assets/residual-layers.svg": {
|
| 183 |
-
"bytes": 10990,
|
| 184 |
-
"sha256": "cdf5cfb263b50eef0171eafead3062ee75eb805961517f313490fdca03b390c3"
|
| 185 |
-
},
|
| 186 |
-
"decision_finetune/__init__.py": {
|
| 187 |
-
"bytes": 81,
|
| 188 |
-
"sha256": "51ab50ae9a598c7b09970161be3551f9dff0851a209ec692f2e296b1bd44566d"
|
| 189 |
-
},
|
| 190 |
-
"decision_finetune/__main__.py": {
|
| 191 |
-
"bytes": 5410,
|
| 192 |
-
"sha256": "d5cabebd24c823d5dc03a4b528dae8b8b70a63dd0692eb6516a59bd4bd64d6a7"
|
| 193 |
-
},
|
| 194 |
-
"decision_finetune/data.py": {
|
| 195 |
-
"bytes": 6679,
|
| 196 |
-
"sha256": "9fae31cb39512952a1dd78fb0b6735434da6f6b94606341fc0200ab9cdaf447e"
|
| 197 |
-
},
|
| 198 |
-
"decision_finetune/metrics.py": {
|
| 199 |
-
"bytes": 4046,
|
| 200 |
-
"sha256": "c3de487e9294d5278064b7e1768915bf68f593c9d7fb067addc52c4c9ac3b07e"
|
| 201 |
-
},
|
| 202 |
-
"decision_finetune/run.py": {
|
| 203 |
-
"bytes": 16750,
|
| 204 |
-
"sha256": "045ff99dd95c69e2ccdf4b1f8af84797c2929a27b16a9c1a3208bba10b41361b"
|
| 205 |
-
},
|
| 206 |
-
"decision_finetune/state.py": {
|
| 207 |
-
"bytes": 5751,
|
| 208 |
-
"sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
|
| 209 |
-
},
|
| 210 |
-
"decision_finetune/system_one.py": {
|
| 211 |
-
"bytes": 1784,
|
| 212 |
-
"sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
|
| 213 |
-
},
|
| 214 |
-
"decision_inference/__init__.py": {
|
| 215 |
-
"bytes": 282,
|
| 216 |
-
"sha256": "8199829090f1a3ca5fe59aec1a6dce2beb57026c301d2c87d8c5b1c960981e66"
|
| 217 |
-
},
|
| 218 |
-
"decision_inference/_auto.py": {
|
| 219 |
-
"bytes": 1325,
|
| 220 |
-
"sha256": "3602e3a06386daab5494305413a08adb581f23007ffb5ca6e5df62fa23d11bd6"
|
| 221 |
-
},
|
| 222 |
-
"decision_inference/_grouped.py": {
|
| 223 |
-
"bytes": 1016,
|
| 224 |
-
"sha256": "80551e1bebd10634abcbbfce1aecf6cafab93b7e178399c428dd97022c6f9d95"
|
| 225 |
-
},
|
| 226 |
-
"decision_inference/_request.py": {
|
| 227 |
-
"bytes": 3026,
|
| 228 |
-
"sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
|
| 229 |
-
},
|
| 230 |
-
"decision_inference/_system_one.py": {
|
| 231 |
-
"bytes": 10728,
|
| 232 |
-
"sha256": "0dc61f7558eed31c9ac186827b2050c0d391c69ea3684c02e8cd9c71400b1690"
|
| 233 |
-
},
|
| 234 |
-
"decision_inference/profile.py": {
|
| 235 |
-
"bytes": 2071,
|
| 236 |
-
"sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
|
| 237 |
-
},
|
| 238 |
-
"decision_runtime/__init__.py": {
|
| 239 |
-
"bytes": 723,
|
| 240 |
-
"sha256": "b0f9db94cfbcc73cab2f09e0269b8bd2eb87c0cc533b34ea5c9a8bfbb1ee48ca"
|
| 241 |
-
},
|
| 242 |
-
"decision_runtime/_compat.py": {
|
| 243 |
-
"bytes": 1847,
|
| 244 |
-
"sha256": "a0dd42a4e3b20eabbaf0d0d62ffc290ef4d9d20bc2fb2ba39d980771a8d0744c"
|
| 245 |
-
},
|
| 246 |
-
"decision_runtime/native.py": {
|
| 247 |
-
"bytes": 6903,
|
| 248 |
-
"sha256": "21c674a0d8a1406156504390974505ba5f981e7ecc9f0c3b29985876ef78511c"
|
| 249 |
-
},
|
| 250 |
-
"decision_runtime/training.py": {
|
| 251 |
-
"bytes": 15998,
|
| 252 |
-
"sha256": "166ea54629d5c90e0e0972b82c0b96e008c799f384400886ffd64312ab972177"
|
| 253 |
-
},
|
| 254 |
-
"examples/decisions.jsonl": {
|
| 255 |
-
"bytes": 867,
|
| 256 |
-
"sha256": "d4b89583ccc9aa35e74dd28c46e01884103ea7a0fa018de8c34d4f88a94287f1"
|
| 257 |
-
},
|
| 258 |
-
"examples/finetune.sh": {
|
| 259 |
-
"bytes": 1242,
|
| 260 |
-
"sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
|
| 261 |
-
},
|
| 262 |
-
"examples/system-one.json": {
|
| 263 |
-
"bytes": 692,
|
| 264 |
-
"sha256": "0b8320ecbeded8c3229dccc52f5045f00e8d1968183bc9d1371d24e949925dcf"
|
| 265 |
-
},
|
| 266 |
-
"infer.py": {
|
| 267 |
-
"bytes": 2373,
|
| 268 |
-
"sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
|
| 269 |
-
},
|
| 270 |
-
"native/INVENTORY.json": {
|
| 271 |
-
"bytes": 31007,
|
| 272 |
-
"sha256": "06aed00ac4571582b992a995a793875804a0546bdae1461c4161b30fba2143a9"
|
| 273 |
-
},
|
| 274 |
-
"native/MANIFEST.json": {
|
| 275 |
-
"bytes": 4326,
|
| 276 |
-
"sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6"
|
| 277 |
-
},
|
| 278 |
-
"native/STATE_LAYOUT.json": {
|
| 279 |
-
"bytes": 10971,
|
| 280 |
-
"sha256": "a6b24716e2b21240d327ba8763a73f1b54862bc86d5e1f92bb9e670be968e494"
|
| 281 |
-
},
|
| 282 |
-
"native/__init__.py": {
|
| 283 |
-
"bytes": 78,
|
| 284 |
-
"sha256": "1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010"
|
| 285 |
-
},
|
| 286 |
-
"native/artifacts.py": {
|
| 287 |
-
"bytes": 7958,
|
| 288 |
-
"sha256": "c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3"
|
| 289 |
-
},
|
| 290 |
-
"native/choice_encoder.safetensors": {
|
| 291 |
-
"bytes": 441337216,
|
| 292 |
-
"sha256": "9516cc841c485c98b27b4f63d2ea8e604fe8064121173064b113da8a6bf57ef6"
|
| 293 |
-
},
|
| 294 |
-
"native/contract.py": {
|
| 295 |
-
"bytes": 10886,
|
| 296 |
-
"sha256": "51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726"
|
| 297 |
-
},
|
| 298 |
-
"native/decision_config.json": {
|
| 299 |
-
"bytes": 14308,
|
| 300 |
-
"sha256": "1fefb4ad7dede00c4633bb2ce5fc44fb04237207b9e9ce90fcf8ea8b01a1328d"
|
| 301 |
-
},
|
| 302 |
-
"native/decision_heads.safetensors": {
|
| 303 |
-
"bytes": 177241884,
|
| 304 |
-
"sha256": "bce3ee658a978a19c48c605b921cff994f892e7fc17abd6f8645f07428fbc36f"
|
| 305 |
-
},
|
| 306 |
-
"native/encoder/config.json": {
|
| 307 |
-
"bytes": 2769,
|
| 308 |
-
"sha256": "7aff915e9f159305e0bef3eb0206416f99b8560260b8969b35f1dcd54aaad1a5"
|
| 309 |
-
},
|
| 310 |
-
"native/encoder/model.safetensors": {
|
| 311 |
-
"bytes": 1227771752,
|
| 312 |
-
"sha256": "daaafd81c4ed767d203e24226d78e792800068a7dac344aa043844b82a6306c7"
|
| 313 |
-
},
|
| 314 |
-
"native/infer.py": {
|
| 315 |
-
"bytes": 1095,
|
| 316 |
-
"sha256": "9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315"
|
| 317 |
-
},
|
| 318 |
-
"native/model.py": {
|
| 319 |
-
"bytes": 11098,
|
| 320 |
-
"sha256": "8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b"
|
| 321 |
-
},
|
| 322 |
-
"native/packing.py": {
|
| 323 |
-
"bytes": 6748,
|
| 324 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 325 |
-
},
|
| 326 |
-
"native/policy/__init__.py": {
|
| 327 |
-
"bytes": 131,
|
| 328 |
-
"sha256": "0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114"
|
| 329 |
-
},
|
| 330 |
-
"native/policy/artifacts.py": {
|
| 331 |
-
"bytes": 7877,
|
| 332 |
-
"sha256": "5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2"
|
| 333 |
-
},
|
| 334 |
-
"native/policy/contract.py": {
|
| 335 |
-
"bytes": 4334,
|
| 336 |
-
"sha256": "0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67"
|
| 337 |
-
},
|
| 338 |
-
"native/policy/infer.py": {
|
| 339 |
-
"bytes": 1545,
|
| 340 |
-
"sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
|
| 341 |
-
},
|
| 342 |
-
"native/policy/model.py": {
|
| 343 |
-
"bytes": 7678,
|
| 344 |
-
"sha256": "eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca"
|
| 345 |
-
},
|
| 346 |
-
"native/policy/packing.py": {
|
| 347 |
-
"bytes": 6748,
|
| 348 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 349 |
-
},
|
| 350 |
-
"native/policy/reference/__init__.py": {
|
| 351 |
-
"bytes": 192,
|
| 352 |
-
"sha256": "c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28"
|
| 353 |
-
},
|
| 354 |
-
"native/policy/reference/artifacts.py": {
|
| 355 |
-
"bytes": 9342,
|
| 356 |
-
"sha256": "3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159"
|
| 357 |
-
},
|
| 358 |
-
"native/policy/reference/infer.py": {
|
| 359 |
-
"bytes": 1545,
|
| 360 |
-
"sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
|
| 361 |
-
},
|
| 362 |
-
"native/policy/reference/model.py": {
|
| 363 |
-
"bytes": 5060,
|
| 364 |
-
"sha256": "733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26"
|
| 365 |
-
},
|
| 366 |
-
"native/policy/reference/modernbert_sdpa_layout.py": {
|
| 367 |
-
"bytes": 3365,
|
| 368 |
-
"sha256": "0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8"
|
| 369 |
-
},
|
| 370 |
-
"native/policy/reference/packing.py": {
|
| 371 |
-
"bytes": 6748,
|
| 372 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 373 |
-
},
|
| 374 |
-
"native/score_encoder.safetensors": {
|
| 375 |
-
"bytes": 441337080,
|
| 376 |
-
"sha256": "4f45795977846ef4c31b34e4f35bf95dabf5951d71358bab32c9a94814a84a53"
|
| 377 |
-
},
|
| 378 |
-
"native/tokenizer/special_tokens_map.json": {
|
| 379 |
-
"bytes": 1051,
|
| 380 |
-
"sha256": "6e3204c5a4004719c185007a745d1940471a6fb02521cfa0a00306dbb6bc5903"
|
| 381 |
-
},
|
| 382 |
-
"native/tokenizer/tokenizer.json": {
|
| 383 |
-
"bytes": 34363188,
|
| 384 |
-
"sha256": "609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f"
|
| 385 |
-
},
|
| 386 |
-
"native/tokenizer/tokenizer_config.json": {
|
| 387 |
-
"bytes": 46470,
|
| 388 |
-
"sha256": "74a259bb1a3811a7e3028adcd07a65765d866d5e66e0e883f0994ccfa67e8455"
|
| 389 |
-
},
|
| 390 |
-
"native/training_policy.py": {
|
| 391 |
-
"bytes": 4761,
|
| 392 |
-
"sha256": "6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5"
|
| 393 |
-
},
|
| 394 |
-
"requirements.txt": {
|
| 395 |
-
"bytes": 296,
|
| 396 |
-
"sha256": "001dd433614bcbacee15858a38f432b97f44b01ad404714f442738f4a8ce2428"
|
| 397 |
-
},
|
| 398 |
-
"source-metadata/ENCODER_README.md": {
|
| 399 |
-
"bytes": 2046,
|
| 400 |
-
"sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
|
| 401 |
-
},
|
| 402 |
-
"source-metadata/ENCODER_SOURCE.json": {
|
| 403 |
-
"bytes": 425,
|
| 404 |
-
"sha256": "9bb1a753d9979117520152860a599064d92a8eaf7f23a0e0d458c66007c8fdf9"
|
| 405 |
-
},
|
| 406 |
-
"source-metadata/MMBERT_LICENSE_SOURCE_README.md": {
|
| 407 |
-
"bytes": 29650,
|
| 408 |
-
"sha256": "35724415037ee3159aa733c16050620cc19a76219112a6ceeb3949572e0b37e9"
|
| 409 |
-
},
|
| 410 |
-
"source-metadata/MMBERT_TOKENIZER_LINEAGE.yaml": {
|
| 411 |
-
"bytes": 2768,
|
| 412 |
-
"sha256": "0ffa0f1c8388868a42cdbb8ec4b709db6eab4b3e33d8495f4e5f0c32bd15afd3"
|
| 413 |
-
},
|
| 414 |
-
"source-metadata/MODEL_LICENSE_METADATA.json": {
|
| 415 |
-
"bytes": 1538,
|
| 416 |
-
"sha256": "696395f97428b851f136d50d850978af8c2a3b278c82620f63246a9312b74a30"
|
| 417 |
-
},
|
| 418 |
-
"source-metadata/TRANSFORMERS_LICENSE_SOURCE.json": {
|
| 419 |
-
"bytes": 1178,
|
| 420 |
-
"sha256": "b60e54814b6c70791318475ebd4737c29fe5816f3141df392144dd31ac316d77"
|
| 421 |
-
},
|
| 422 |
-
"source-metadata/TRANSFORMERS_MODERNBERT_SOURCE_HEADER.txt": {
|
| 423 |
-
"bytes": 1497,
|
| 424 |
-
"sha256": "4577f1803f9dbe8b43f5da709f8750c798def2c33f7eb055d62baf286ed472c8"
|
| 425 |
-
},
|
| 426 |
-
"source-metadata/VELA_LICENSE_SOURCE_README.md": {
|
| 427 |
-
"bytes": 2046,
|
| 428 |
-
"sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
|
| 429 |
-
},
|
| 430 |
-
"systemone.py": {
|
| 431 |
-
"bytes": 2368,
|
| 432 |
-
"sha256": "8ce311c5651f1b0c1bc99d5a4ccaa6c7878b5c509256e4668a2191ad8c35e2e3"
|
| 433 |
-
}
|
| 434 |
-
},
|
| 435 |
-
"hub_repository": "llm-semantic-router/Decision-1.0-Lex-0.6B",
|
| 436 |
-
"language_scope": "English typed-decisions specialist",
|
| 437 |
-
"manifest_excludes_itself": true,
|
| 438 |
-
"name": "Decision-1.0-Lex",
|
| 439 |
-
"native_file_count": 31,
|
| 440 |
-
"native_manifest": "native/MANIFEST.json",
|
| 441 |
-
"optional_batch_capacity": {
|
| 442 |
-
"default_batch_limit": 8,
|
| 443 |
-
"entry": "predict_auto_1k",
|
| 444 |
-
"maximum_batch_size": 32,
|
| 445 |
-
"padding_guard": true,
|
| 446 |
-
"validation": "BATCH_CAPACITY.json",
|
| 447 |
-
"weights_unchanged": true
|
| 448 |
-
},
|
| 449 |
-
"parameter_tensors": 489,
|
| 450 |
-
"parameters": 571909635,
|
| 451 |
-
"public_non_native_bytes_excluding_this_manifest": 5308024,
|
| 452 |
-
"publication": {
|
| 453 |
-
"download_access": "public_ungated",
|
| 454 |
-
"new_contribution_license": "Apache-2.0",
|
| 455 |
-
"retained_third_party_terms": true
|
| 456 |
-
},
|
| 457 |
-
"runtime_update": {
|
| 458 |
-
"request_local_input_reuse": true,
|
| 459 |
-
"validation": "RUNTIME_UPDATE.json",
|
| 460 |
-
"weights_unchanged": true
|
| 461 |
-
},
|
| 462 |
-
"schema": "decision.public-distribution.v1",
|
| 463 |
-
"status": "READY_FOR_ROOT_PUBLICATION",
|
| 464 |
-
"subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 465 |
-
"system_one_api": {
|
| 466 |
-
"current_scheduling_validation": "MIXED_QUESTION_SCALING.json",
|
| 467 |
-
"default_physical_batch_limit": 8,
|
| 468 |
-
"default_scheduling": "stable typed groups",
|
| 469 |
-
"entry": "decision_inference.SystemOne",
|
| 470 |
-
"finetune_conversion": "decision_finetune.system_one.system_one_training_rows",
|
| 471 |
-
"max_batch_decisions": 512,
|
| 472 |
-
"max_questions": 128,
|
| 473 |
-
"max_request_bytes": 2097152,
|
| 474 |
-
"max_requests": 128,
|
| 475 |
-
"optional_auto_capacity": 32,
|
| 476 |
-
"request": "state/model/questions",
|
| 477 |
-
"response": "model/answers/usage",
|
| 478 |
-
"validation": "SYSTEM_ONE_VALIDATION.json",
|
| 479 |
-
"weights_unchanged": true
|
| 480 |
-
},
|
| 481 |
-
"technical_validation": "TECHNICAL_VALIDATION.json",
|
| 482 |
-
"training_provenance": "TRAINING_PROVENANCE.json",
|
| 483 |
-
"typed_scheduling_update": {
|
| 484 |
-
"complete_input_limit_unchanged": 1024,
|
| 485 |
-
"default_schedule": "Stable group by decision type; physicalB8; restore original caller order",
|
| 486 |
-
"entry": "decision_inference.SystemOne",
|
| 487 |
-
"native_weights_unchanged": true,
|
| 488 |
-
"new_chat_api": false,
|
| 489 |
-
"public_api_unchanged": true,
|
| 490 |
-
"subject_native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 491 |
-
"validation": "MIXED_QUESTION_SCALING.json"
|
| 492 |
-
}
|
| 493 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
PERFORMANCE.md
DELETED
|
@@ -1,66 +0,0 @@
|
|
| 1 |
-
# Lex: AMD SystemOne latency
|
| 2 |
-
|
| 3 |
-
## Mixed questions, faster decisions
|
| 4 |
-
|
| 5 |
-
**128 mixed questions: 154.41 ms median, 56.6% lower than the previous default runtime.**
|
| 6 |
-
|
| 7 |
-

|
| 8 |
-
|
| 9 |
-
| Questions | Previous p50 / p95 (ms) | Typed scheduling p50 / p95 (ms) |
|
| 10 |
-
|---:|---:|---:|
|
| 11 |
-
| 1 | 13.22 / 13.32 | 13.20 / 13.40 |
|
| 12 |
-
| 8 | 28.07 / 28.32 | 28.02 / 28.31 |
|
| 13 |
-
| 32 | 93.76 / 94.06 | 53.43 / 53.77 |
|
| 14 |
-
| 64 | 181.78 / 184.53 | 87.39 / 88.09 |
|
| 15 |
-
| 128 | 355.97 / 360.25 | 154.41 / 155.93 |
|
| 16 |
-
|
| 17 |
-
The request repeats three fixed questions—Choice, Noul and Score—over one unchanged context. Only question count and bookkeeping IDs change. The default SystemOne path groups admitted rows by type so each encoder path processes fuller batches, then restores the original answer order. Every question still has its own contextual computation. Weights, precision, complete-input limit and request schema are unchanged.
|
| 18 |
-
|
| 19 |
-
Measured using the exact release package on AMD ROCm, FP32, physical batch size eight: 10 warmup pairs and 30 alternating AB/BA measurement pairs per point, with GPU synchronization. Measurements include local request conversion, tokenization, model execution and answer assembly; they exclude transport and Studio. Models ran serially after our training and data jobs completed. Results describe this fixed workload, not a service latency guarantee.
|
| 20 |
-
|
| 21 |
-
Both models passed seven source panels: 4,160 admitted decisions without an argmax change and 57 identical whole-request refusals before any forward. Exact-package checks additionally cover 512 decisions, six current Studio examples, caller IDs/order, optional auto batching and input limits. Small floating-point probability differences remain possible.
|
| 22 |
-
|
| 23 |
-
[All samples, workload, checks and earlier concurrent measurements](MIXED_QUESTION_SCALING.json) · [SVG](assets/mixed-question-scaling.svg) · [PDF](assets/mixed-question-scaling.pdf)
|
| 24 |
-
|
| 25 |
-
## Earlier measurements
|
| 26 |
-
|
| 27 |
-
## Lex: AMD batch capacity
|
| 28 |
-
|
| 29 |
-
### Optional larger batches
|
| 30 |
-
|
| 31 |
-
```python
|
| 32 |
-
from decision_inference import predict_1k, predict_auto_1k
|
| 33 |
-
|
| 34 |
-
## native and records use the same objects as the Python usage examples.
|
| 35 |
-
outputs = predict_1k(native, records) # unchanged default: up to 8
|
| 36 |
-
outputs = predict_auto_1k(native, records) # opt in: up to 32 with the padding guard
|
| 37 |
-
```
|
| 38 |
-
|
| 39 |
-
The optional entry admits every complete input before the first forward. It uses consecutive batches up to 32 only when the request has one decision type and total padded tokens do not increase relative to B8. Mixed types or increased padding fall back to B8; requests of eight or fewer take the unchanged default path. Input order, candidates, full 1,024-token limit and FP32 weights remain unchanged. It does not share contextual activations across questions. Larger physical batches can change floating-point rounding and peak memory.
|
| 40 |
-
|
| 41 |
-
#### Paired B8/B32 study
|
| 42 |
-
|
| 43 |
-
These are synchronized resident **Python API measurements**, not Studio or network latency. This earlier study compared default B8 with a homogeneous cap32 policy before the final padding guard was added. All eight measured points satisfy that guard, but **the final `predict_auto_1k` entry was validated separately and was not timed**. Existing default-B8 measurements above, if present, remain their original separate run.
|
| 44 |
-
|
| 45 |
-
| Workload | B8 p50 / p95 (ms) | Cap32 p50 / p95 (ms) | p50 ratio · paired 95% CI |
|
| 46 |
-
|---|---:|---:|---:|
|
| 47 |
-
| 242 tokens × 1 question | 13.92 / 14.01 | 13.92 / 14.05 | 1.0001 · [0.9975, 1.0020] |
|
| 48 |
-
| 242 tokens × 8 questions | 22.65 / 22.75 | 22.60 / 22.77 | 0.9981 · [0.9967, 0.9994] |
|
| 49 |
-
| 242 tokens × 16 questions | 38.86 / 39.42 | 33.39 / 33.55 | 0.8592 · [0.8578, 0.8603] |
|
| 50 |
-
| 242 tokens × 32 questions | 71.03 / 71.46 | 54.83 / 60.15 | 0.7718 · [0.7706, 0.7724] |
|
| 51 |
-
| 1024 tokens × 1 question | 18.09 / 18.24 | 18.08 / 18.21 | 0.9992 · [0.9978, 1.0003] |
|
| 52 |
-
| 1024 tokens × 8 questions | 65.74 / 65.89 | 65.72 / 66.52 | 0.9997 · [0.9990, 1.0004] |
|
| 53 |
-
| 1024 tokens × 32 questions | 242.37 / 243.72 | 217.63 / 218.37 | 0.8979 · [0.8961, 0.8986] |
|
| 54 |
-
| Multi-context · 32 decisions | 44.08 / 44.25 | 28.76 / 28.87 | 0.6523 · [0.6519, 0.6537] |
|
| 55 |
-
|
| 56 |
-
AMD ROCm gfx942, FP32 SDPA, one resident model and one request at a time. Each point/mode had 10 warmups and 50 timed requests; the same validation, transfer and auditing hooks were present in both modes. Ten paired blocks (five AB, five BA; five observations per mode/block) were retained. The reported p50 ratio is median(cap32)/median(B8); 2,000 whole-block bootstrap resamples within order strata used fixed seed 20260922. p95 has only 50 observations/mode and correspondingly limited precision. Model loading is excluded.
|
| 57 |
-
|
| 58 |
-
Choice count curves repeat the same synthetic four-candidate question with opaque ID changes. The multi-context fixture repeats eight existing contexts with two related questions twice (32 decisions). These are throughput shapes, not quality examples. The fixed 15% utility gate remained **FAIL** for both models: Q16 improved approximately 14%, despite a confidence interval below 1. Q32 and multi-context improvements do not retroactively change that outcome. The option is an explicit engineering choice, not a universal speed guarantee.
|
| 59 |
-
|
| 60 |
-
At 1024×32, peak allocated memory increased from approximately 2.652 to 3.824 GiB. Shared allocator reserved-memory observations are not independent per-mode estimates. Equal total padding does not imply equal peak memory.
|
| 61 |
-
|
| 62 |
-
#### Public-entry validation
|
| 63 |
-
|
| 64 |
-
The actual imported package passed 46 forwards per model (92 total across Kai and Lex) over 225 technical row-occurrences/model, including all three types, dynamic candidate counts, enabled32, tail33, unequal-length fallback and mixed64. Late invalid and empty requests added zero forwards. The original strict bounds were 1e-4 logits, 2e-5 probability and normalized Score error, with zero hard-decision flips. Both models passed; fallback outputs were exact. All 489 parameters and 66 buffers retained identical contents and versions. No training, quality evaluation or timing was added in this integration check. Fixed repetition is not independent quality support.
|
| 65 |
-
|
| 66 |
-
[Exact paired measurements, numerical results and retained failure gates](BATCH_CAPACITY.json). No weights or default entry behavior changed, and no Studio speedup is claimed.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
README.md
CHANGED
|
@@ -46,9 +46,15 @@ Supply the context, question and candidate descriptions at runtime. **Choice** r
|
|
| 46 |
|
| 47 |
Lex is evaluated as an English specialist on these four workflows. For broader multilingual tasks, explore the general [Kai model](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B).
|
| 48 |
|
| 49 |
-
**
|
| 50 |
|
| 51 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 52 |
|
| 53 |
## Use
|
| 54 |
|
|
@@ -92,22 +98,22 @@ curl -X POST https://your-decision-endpoint.example/v1/systemone \
|
|
| 92 |
}'
|
| 93 |
```
|
| 94 |
|
| 95 |
-
[Official Python SDK](https://docs.typesafe.ai/sdk/python/usage) · [HTTP API](https://docs.typesafe.ai/api)
|
| 96 |
|
| 97 |
## Make it yours
|
| 98 |
|
| 99 |
-
**One state. Many decisions.**
|
| 100 |
|
| 101 |
-
|
| 102 |
|
| 103 |
The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
|
| 104 |
|
| 105 |
## Architecture
|
| 106 |
|
| 107 |
-
 · [Training and methods](METHODS.md)
|
| 112 |
|
| 113 |
Built on [Kai](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B) and [Vela Encoder](https://huggingface.co/llm-semantic-router/Vela-1.0-Encoder-307M). Probabilities are not calibrated confidence; candidate order and task wording can affect outputs. [Attribution and retained third-party terms](NOTICE) · [License scope](LICENSING_STATUS.md).
|
|
|
|
| 46 |
|
| 47 |
Lex is evaluated as an English specialist on these four workflows. For broader multilingual tasks, explore the general [Kai model](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B).
|
| 48 |
|
| 49 |
+
**128 mixed questions in 154 ms — 57% lower latency.** Automatic typed scheduling accelerates the measured SystemOne runtime with the same weights. Paired local AMD measurements on a fixed workload. [Latency and scaling](evaluation/PERFORMANCE.md).
|
| 50 |
|
| 51 |
+
## Download for local inference
|
| 52 |
+
|
| 53 |
+
```bash
|
| 54 |
+
hf download llm-semantic-router/Decision-1.0-Lex-0.6B --local-dir Decision-1.0-Lex-0.6B
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
This repository contains model files and provenance only. Local inference requires a compatible vLLM Semantic Router Decision runtime on AMD ROCm; the runtime is distributed separately. It must support `vllm-sr-decision` format version 1 and the file map in [`config.json`](config.json). `transformers.AutoModel.from_pretrained` does not load the complete decision model.
|
| 58 |
|
| 59 |
## Use
|
| 60 |
|
|
|
|
| 98 |
}'
|
| 99 |
```
|
| 100 |
|
| 101 |
+
[Official Python SDK](https://docs.typesafe.ai/sdk/python/usage) · [HTTP API](https://docs.typesafe.ai/api)
|
| 102 |
|
| 103 |
## Make it yours
|
| 104 |
|
| 105 |
+
**One state. Many decisions.** A compatible System One endpoint can submit typed questions and return results under their original question IDs. Request limits and batching depend on that deployment.
|
| 106 |
|
| 107 |
+
Use the [SDK and curl examples](#use) with a compatible endpoint. Lex's specialization recipe and source data are documented in [Methods](evaluation/METHODS.md) and [Training provenance](TRAINING_PROVENANCE.json).
|
| 108 |
|
| 109 |
The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
|
| 110 |
|
| 111 |
## Architecture
|
| 112 |
|
| 113 |
+

|
| 114 |
|
| 115 |
Three 22-layer bidirectional encoder paths share multilingual input embeddings, with separate interaction layers and candidate readouts for Choice, Noul and Score. Lex retains Kai's architecture and specializes its weights through supervised fine-tuning.
|
| 116 |
|
| 117 |
+
[Architecture details](ARCHITECTURE.md) · [Training and methods](evaluation/METHODS.md)
|
| 118 |
|
| 119 |
Built on [Kai](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B) and [Vela Encoder](https://huggingface.co/llm-semantic-router/Vela-1.0-Encoder-307M). Probabilities are not calibrated confidence; candidate order and task wording can affect outputs. [Attribution and retained third-party terms](NOTICE) · [License scope](LICENSING_STATUS.md).
|
RUNTIME_UPDATE.json
DELETED
|
@@ -1,32 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"schema": "decision.runtime-update.v1",
|
| 3 |
-
"change": "Reuse complete admitted input encodings within one synchronous request.",
|
| 4 |
-
"weights_unchanged": true,
|
| 5 |
-
"state_activations_cached": false,
|
| 6 |
-
"cross_request_cache": false,
|
| 7 |
-
"complete_input_tokens": 1024,
|
| 8 |
-
"maximum_batch_size": 8,
|
| 9 |
-
"native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 10 |
-
"verification": {
|
| 11 |
-
"AMD_FP32": true,
|
| 12 |
-
"per_model_calls": 24,
|
| 13 |
-
"all_output_fields_equal": true,
|
| 14 |
-
"all489_parameter_content_versions_unchanged": true,
|
| 15 |
-
"all_types": true,
|
| 16 |
-
"duplicate_external_ids": true,
|
| 17 |
-
"partial_final_batch": true,
|
| 18 |
-
"complete1024": true,
|
| 19 |
-
"invalid_later_row_rejected_before_forward": true,
|
| 20 |
-
"empty_request": true
|
| 21 |
-
},
|
| 22 |
-
"new_inference_sources": {
|
| 23 |
-
"decision_inference/profile.py": {
|
| 24 |
-
"bytes": 2071,
|
| 25 |
-
"sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
|
| 26 |
-
},
|
| 27 |
-
"decision_inference/_request.py": {
|
| 28 |
-
"bytes": 3026,
|
| 29 |
-
"sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
|
| 30 |
-
}
|
| 31 |
-
}
|
| 32 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
SYSTEM_ONE.md
DELETED
|
@@ -1,134 +0,0 @@
|
|
| 1 |
-
# One state. Many decisions.
|
| 2 |
-
|
| 3 |
-
Decision uses the [System One API format](https://docs.typesafe.ai/api): supply
|
| 4 |
-
`state`, `model` and a map of typed `questions`; receive an `answers` map under
|
| 5 |
-
the same question IDs. Noul returns the probability of yes, Choice selects from
|
| 6 |
-
your named options, and Score evaluates an ordered rubric.
|
| 7 |
-
|
| 8 |
-
```python
|
| 9 |
-
from decision_inference import SystemOne
|
| 10 |
-
|
| 11 |
-
# native is your loaded Decision checkpoint; see USAGE.md for loading.
|
| 12 |
-
client = SystemOne(native, batching="auto")
|
| 13 |
-
result = client.system_one(
|
| 14 |
-
state={"message": "Please refund the duplicate charge. I need this fixed today."},
|
| 15 |
-
questions={
|
| 16 |
-
"refund_requested": {
|
| 17 |
-
"type": "noul",
|
| 18 |
-
"instructions": "Does the customer explicitly request a refund?",
|
| 19 |
-
},
|
| 20 |
-
"team": {
|
| 21 |
-
"type": "choice",
|
| 22 |
-
"instructions": "Which team should handle this request?",
|
| 23 |
-
"criteria": {"Billing": "Charges and refunds", "Support": "Technical problems"},
|
| 24 |
-
},
|
| 25 |
-
"urgency": {
|
| 26 |
-
"type": "score",
|
| 27 |
-
"instructions": "How urgent is the request?",
|
| 28 |
-
"criteria": ["No deadline", "Needed soon", "Needed today"],
|
| 29 |
-
},
|
| 30 |
-
},
|
| 31 |
-
)
|
| 32 |
-
print(result["answers"])
|
| 33 |
-
```
|
| 34 |
-
|
| 35 |
-
`client.evaluate(request)` accepts the complete JSON request object, including
|
| 36 |
-
`model`. A client is bound to its loaded checkpoint; a mismatched model is an
|
| 37 |
-
error. Public Kai/Lex names are inferred from their verified native manifest.
|
| 38 |
-
For your own fine-tune, use `SystemOne(native, model="my-decision-model")`.
|
| 39 |
-
|
| 40 |
-
## Batch the questions or the contexts
|
| 41 |
-
|
| 42 |
-
One request accepts **up to 128 questions**. For one set of questions across many
|
| 43 |
-
independent contexts, use `client.batch`:
|
| 44 |
-
|
| 45 |
-
```python
|
| 46 |
-
questions = {
|
| 47 |
-
"refund_requested": {
|
| 48 |
-
"type": "noul",
|
| 49 |
-
"instructions": "Does the customer explicitly request a refund?",
|
| 50 |
-
}
|
| 51 |
-
}
|
| 52 |
-
results = client.batch([
|
| 53 |
-
{"model": client.model, "state": message, "questions": questions}
|
| 54 |
-
for message in messages
|
| 55 |
-
])
|
| 56 |
-
# results[i] corresponds to messages[i]; question IDs can repeat across requests.
|
| 57 |
-
```
|
| 58 |
-
|
| 59 |
-
`batch` is a Decision Python extension around ordinary System One requests. It
|
| 60 |
-
accepts up to 128 requests and 512 total decisions, with a combined 2 MiB input
|
| 61 |
-
limit. Responses preserve request order and question order. The entire batch
|
| 62 |
-
must pass validation and complete-input token admission before any forward.
|
| 63 |
-
An overlength or malformed input fails the call without partial answers.
|
| 64 |
-
|
| 65 |
-
The default uses physical batches of up to eight. `batching="auto"` uses the
|
| 66 |
-
released padding-aware scheduler: up to 32 consecutive same-type decisions when
|
| 67 |
-
that adds no padding; otherwise it keeps B8. FP32 rounding can vary with physical
|
| 68 |
-
batch shape. Each state/question pair still has its own encoder computation.
|
| 69 |
-
One API call does not imply one forward or a shared state activation cache.
|
| 70 |
-
|
| 71 |
-
## Typed fields
|
| 72 |
-
|
| 73 |
-
| Type | `criteria` | Answer |
|
| 74 |
-
|---|---|---|
|
| 75 |
-
| `noul` | Optional `true` / `false` descriptions | `type`, `noul` |
|
| 76 |
-
| `choice` | 2–255 named options; descriptions may be null | `type`, `choice`, `probabilities`, `confidence` |
|
| 77 |
-
| `score` | 2–10 ordered level descriptions | `type`, `score`, `legend`, `probabilities`, `confidence` |
|
| 78 |
-
|
| 79 |
-
State, instructions and descriptions accept strings, JSON objects or arrays.
|
| 80 |
-
Structured values become deterministic compact JSON with sorted object keys;
|
| 81 |
-
array order, Choice option order and Score level order are preserved. Question
|
| 82 |
-
IDs are bookkeeping only. Choice names are part of the semantic input, including
|
| 83 |
-
when their description is null. Score levels are indexed from zero; `score`
|
| 84 |
-
preserves the native probability-weighted FP32 expectation. Structured Score
|
| 85 |
-
descriptions appear as JSON strings in `legend`.
|
| 86 |
-
|
| 87 |
-
Every complete state/question/candidate sequence must fit **1,024 tokens**.
|
| 88 |
-
Nothing is truncated. The token count includes the repeated state for each
|
| 89 |
-
question; `usage.input_tokens` sums these complete sequences and
|
| 90 |
-
`usage.output_tokens` is zero because the model returns scores without generating
|
| 91 |
-
text. These are local computation counts, not TypeSafe billing counts.
|
| 92 |
-
|
| 93 |
-
Decision's `confidence` is the largest candidate probability, matching its native
|
| 94 |
-
runtime. It is not calibrated correctness. TypeSafe does not specify its own
|
| 95 |
-
formula in the [confidence documentation](https://docs.typesafe.ai/confidence),
|
| 96 |
-
so thresholds should not be transferred between models without validation.
|
| 97 |
-
The shared request/answer schema does not claim identical weights, confidence
|
| 98 |
-
values, hosted service limits or SDK behavior.
|
| 99 |
-
|
| 100 |
-
## Fine-tune with the same inputs
|
| 101 |
-
|
| 102 |
-
Use the same conversion for training so structured inputs, option names and
|
| 103 |
-
criteria have identical semantics at training and inference time:
|
| 104 |
-
|
| 105 |
-
```python
|
| 106 |
-
import json
|
| 107 |
-
from pathlib import Path
|
| 108 |
-
from decision_finetune.system_one import system_one_training_rows
|
| 109 |
-
|
| 110 |
-
request = json.loads(Path("examples/system-one.json").read_text())
|
| 111 |
-
rows = system_one_training_rows(
|
| 112 |
-
request, # the same model/state/questions object used for inference
|
| 113 |
-
targets={
|
| 114 |
-
"refund_requested": {"probability": 1.0},
|
| 115 |
-
"team": {"choice_id": "Billing"},
|
| 116 |
-
"urgency": {"probabilities": [0.0, 0.0, 1.0]},
|
| 117 |
-
},
|
| 118 |
-
request_id="ticket-1001",
|
| 119 |
-
source_id="support-tickets",
|
| 120 |
-
component_id="customer-42",
|
| 121 |
-
)
|
| 122 |
-
```
|
| 123 |
-
|
| 124 |
-
Save the rows as JSONL for the existing [fine-tuning CLI](FINETUNING.md). Targets
|
| 125 |
-
stay separate from state, instructions and criteria. Related examples share a
|
| 126 |
-
source/component ID and stay in one split; the CLI checks component and exact
|
| 127 |
-
input overlap. Optional `hard_target_ids` supplies separate evaluation labels
|
| 128 |
-
under the same question IDs. Noul supports soft yes probabilities; Choice and
|
| 129 |
-
Score support complete soft distributions in the original candidate order.
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
### Default typed scheduling
|
| 133 |
-
|
| 134 |
-
The default `SystemOne` path groups complete, admitted questions by decision type in physical batches of eight, then restores the original request and question order. This works for many questions over one state and questions across multiple contexts. No API changes or application-side sorting are needed. The optional `batching="auto"` policy is unchanged. [Measured mixed-question scaling](PERFORMANCE.md).
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
SYSTEM_ONE_VALIDATION.json
DELETED
|
@@ -1,66 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"all_reference_outputs_exact": true,
|
| 3 |
-
"api_source": {
|
| 4 |
-
"bytes": 10690,
|
| 5 |
-
"sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
|
| 6 |
-
},
|
| 7 |
-
"complete_input_tokens": 1024,
|
| 8 |
-
"coverage": [
|
| 9 |
-
"Noul, Choice and Score",
|
| 10 |
-
"128 questions",
|
| 11 |
-
"128 contexts and 512 decisions",
|
| 12 |
-
"question rename and reorder",
|
| 13 |
-
"irrelevant question insertion",
|
| 14 |
-
"single and multi-context agreement",
|
| 15 |
-
"default B8 and opt-in padding-aware B32",
|
| 16 |
-
"late invalid input before any forward",
|
| 17 |
-
"same training and inference input encoding"
|
| 18 |
-
],
|
| 19 |
-
"cross_batch_tolerance": 2e-05,
|
| 20 |
-
"device": "AMD GPU / ROCm",
|
| 21 |
-
"limits": "Technical compatibility validation on synthetic unlabeled cases; not evidence of accuracy, calibration or latency gains.",
|
| 22 |
-
"max_decisions_per_batch": 512,
|
| 23 |
-
"max_questions_per_request": 128,
|
| 24 |
-
"max_requests_per_batch": 128,
|
| 25 |
-
"models": {
|
| 26 |
-
"Kai": {
|
| 27 |
-
"all_489_parameters_and_66_buffers_unchanged": true,
|
| 28 |
-
"cross_case_checks": 713,
|
| 29 |
-
"forward_calls": 242,
|
| 30 |
-
"forward_rows": 2756,
|
| 31 |
-
"hard_flips": 0,
|
| 32 |
-
"invalid_input_checks": 7,
|
| 33 |
-
"invalid_input_forward_calls": 0,
|
| 34 |
-
"max_probability_error": 2.0563602447509766e-06,
|
| 35 |
-
"max_score_error": 2.980232238769531e-07,
|
| 36 |
-
"native_manifest_sha256": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
|
| 37 |
-
"result_sha256": "8cb3bc435e36b79d96b60058e9b4c57b9996dcb8a55a1cd3faf6c7324e7f917c",
|
| 38 |
-
"same_physical_reference_outputs_exact": true
|
| 39 |
-
},
|
| 40 |
-
"Lex": {
|
| 41 |
-
"all_489_parameters_and_66_buffers_unchanged": true,
|
| 42 |
-
"cross_case_checks": 713,
|
| 43 |
-
"forward_calls": 242,
|
| 44 |
-
"forward_rows": 2756,
|
| 45 |
-
"hard_flips": 0,
|
| 46 |
-
"invalid_input_checks": 7,
|
| 47 |
-
"invalid_input_forward_calls": 0,
|
| 48 |
-
"max_probability_error": 1.1324882507324219e-06,
|
| 49 |
-
"max_score_error": 7.152557373046875e-07,
|
| 50 |
-
"native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 51 |
-
"result_sha256": "e94c5f1109f15005415a48623602d5255b38bd33451b55c758a0db0c16f68191",
|
| 52 |
-
"same_physical_reference_outputs_exact": true
|
| 53 |
-
}
|
| 54 |
-
},
|
| 55 |
-
"optimizer_updates": 0,
|
| 56 |
-
"performance_benchmark": false,
|
| 57 |
-
"quality_evaluation": false,
|
| 58 |
-
"reference": "Independent System One mapping using the original Studio single-question conversion and native predictor; identical physical batches.",
|
| 59 |
-
"status": "PASS_AMD_SYSTEM_ONE_API",
|
| 60 |
-
"total_forward_calls": 484,
|
| 61 |
-
"training_conversion_source": {
|
| 62 |
-
"bytes": 1784,
|
| 63 |
-
"sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
|
| 64 |
-
},
|
| 65 |
-
"weights_unchanged": true
|
| 66 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
USAGE.md
DELETED
|
@@ -1,121 +0,0 @@
|
|
| 1 |
-
# Use Lex
|
| 2 |
-
|
| 3 |
-
Use the official TypeSafe SDK or curl to send a shared context and named typed questions to a SystemOne-compatible endpoint. The example asks a routing question and an urgency question about the same delivery request.
|
| 4 |
-
|
| 5 |
-
## Endpoint setup
|
| 6 |
-
|
| 7 |
-
Replace `https://your-decision-endpoint.example` with an endpoint configured to expose `Decision-1.0-Lex-0.6B`. Set the `DECISION_API_KEY` environment variable to the key issued by that endpoint's operator. The sized model name below is a deployment alias that the operator must configure.
|
| 8 |
-
|
| 9 |
-
The Hugging Face repository distributes weights and local inference code; it does not provision an HTTP service or issue API keys. TypeSafe does not host these Decision weights. Installing the official SDK supplies a client for a compatible service, not a model deployment.
|
| 10 |
-
|
| 11 |
-
These are request examples, not recorded model predictions. No example probabilities or performance results are implied.
|
| 12 |
-
|
| 13 |
-
## Official Python SDK
|
| 14 |
-
|
| 15 |
-
```bash
|
| 16 |
-
pip install typesafe-sdk
|
| 17 |
-
```
|
| 18 |
-
|
| 19 |
-
```python
|
| 20 |
-
import os
|
| 21 |
-
from typesafe_sdk import TypeSafeClient, Choice, Noul
|
| 22 |
-
|
| 23 |
-
client = TypeSafeClient(
|
| 24 |
-
api_key=os.environ["DECISION_API_KEY"],
|
| 25 |
-
base_url="https://your-decision-endpoint.example",
|
| 26 |
-
model="Decision-1.0-Lex-0.6B",
|
| 27 |
-
)
|
| 28 |
-
questions = {
|
| 29 |
-
"route": Choice(instructions="Which team should handle this request?",
|
| 30 |
-
criteria={"delivery": "Damaged or missing parcels", "billing": "Payments and invoices"}),
|
| 31 |
-
"urgent": Noul(instructions="Does the customer request action today?"),
|
| 32 |
-
}
|
| 33 |
-
response = client.system_one(state="The parcel arrived damaged. Please send a replacement today.", questions=questions)
|
| 34 |
-
print(response.choices["route"].choice, response.nouls["urgent"].noul)
|
| 35 |
-
```
|
| 36 |
-
|
| 37 |
-
`response.choices["route"].choice` is the selected candidate ID; `response.nouls["urgent"].noul` is the probability that the condition holds. The application decides how to act on the answers.
|
| 38 |
-
|
| 39 |
-
## Equivalent curl request
|
| 40 |
-
|
| 41 |
-
The model, state, question IDs, instructions and candidate order are identical to the SDK example.
|
| 42 |
-
|
| 43 |
-
```bash
|
| 44 |
-
curl -X POST https://your-decision-endpoint.example/v1/systemone \
|
| 45 |
-
-H "Authorization: Bearer $DECISION_API_KEY" \
|
| 46 |
-
-H "Content-Type: application/json" \
|
| 47 |
-
--data '{
|
| 48 |
-
"model": "Decision-1.0-Lex-0.6B",
|
| 49 |
-
"state": "The parcel arrived damaged. Please send a replacement today.",
|
| 50 |
-
"questions": {
|
| 51 |
-
"route": {"type": "choice", "instructions": "Which team should handle this request?", "criteria": {"delivery": "Damaged or missing parcels", "billing": "Payments and invoices"}},
|
| 52 |
-
"urgent": {"type": "noul", "instructions": "Does the customer request action today?"}
|
| 53 |
-
}
|
| 54 |
-
}'
|
| 55 |
-
```
|
| 56 |
-
|
| 57 |
-
See the [official Python usage guide](https://docs.typesafe.ai/sdk/python/usage) and [HTTP request/response reference](https://docs.typesafe.ai/api).
|
| 58 |
-
|
| 59 |
-
## Several contexts, the same questions
|
| 60 |
-
|
| 61 |
-
Keep the `questions` map and submit another `state` to `client.system_one`. Each request evaluates both questions against its own context. A larger question map expresses more decisions about that context; service concurrency, request limits and scheduling depend on the deployment.
|
| 62 |
-
|
| 63 |
-
The bundled native implementations also support multi-context batching. This is a local inference capability; it does not imply that a deployment exposes a batch HTTP route. All complete-input limits still apply to each rendered state/question/candidate sequence.
|
| 64 |
-
|
| 65 |
-
## Deployment names and native compatibility
|
| 66 |
-
|
| 67 |
-
The HTTP examples use the configured alias `Decision-1.0-Lex-0.6B`. The bundled local model retains the stable internal identifier `Decision-1.0-Lex`. A deployment maps its public alias to that native model; adding a size suffix to the Hub repository does not change the native identifier or its accepted aliases.
|
| 68 |
-
|
| 69 |
-
Existing local SystemOne CLI examples continue to use the unsized internal identifier. The native code and weights are unchanged.
|
| 70 |
-
|
| 71 |
-
## Local native installation
|
| 72 |
-
|
| 73 |
-
Download the complete public release with the Hugging Face CLI. No access approval or login is required:
|
| 74 |
-
|
| 75 |
-
```bash
|
| 76 |
-
hf download llm-semantic-router/Decision-1.0-Lex-0.6B --local-dir Decision-1.0-Lex-0.6B
|
| 77 |
-
cd Decision-1.0-Lex-0.6B
|
| 78 |
-
```
|
| 79 |
-
|
| 80 |
-
Use `--local-dir` to materialize ordinary files for the native loader. Use the compatible ROCm/Python environment described below.
|
| 81 |
-
|
| 82 |
-
This release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
|
| 83 |
-
|
| 84 |
-
Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
|
| 85 |
-
|
| 86 |
-
## System One: parallel typed questions
|
| 87 |
-
|
| 88 |
-
For local execution, the existing native CLI remains available from the downloaded repository:
|
| 89 |
-
|
| 90 |
-
```bash
|
| 91 |
-
ROCR_VISIBLE_DEVICES=0 python systemone.py \
|
| 92 |
-
--input examples/system-one.json --output answers.json --batching auto
|
| 93 |
-
```
|
| 94 |
-
|
| 95 |
-
The bundled wrapper accepts up to 128 questions per request and combines independent contexts with up to 512 total decisions. Each complete state/question/candidate sequence must fit 1,024 tokens; validation happens before the first forward. Default physical batching is B8; `auto` opts into the released homogeneous, padding-aware B32 scheduler. These local limits are not a promise about another endpoint's limits. [Native fields and fine-tuning conversion](SYSTEM_ONE.md).
|
| 96 |
-
|
| 97 |
-
## Native records
|
| 98 |
-
|
| 99 |
-
The original native JSONL interface remains available for applications that
|
| 100 |
-
need logits, explicit candidate IDs or arbitrary ordered Score values.
|
| 101 |
-
|
| 102 |
-
```bash
|
| 103 |
-
export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
|
| 104 |
-
export PYTHONDONTWRITEBYTECODE=1
|
| 105 |
-
export HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 TOKENIZERS_PARALLELISM=false
|
| 106 |
-
ROCR_VISIBLE_DEVICES=0 python infer.py \
|
| 107 |
-
--input examples/decisions.jsonl --output predictions.jsonl --batch-size 8
|
| 108 |
-
```
|
| 109 |
-
|
| 110 |
-
The three examples illustrate the Choice, Noul and Score interfaces. They are synthetic interface examples, not evaluated model predictions. Each input line has exactly `id`, `state_text`, and `question`; do not attach labels. Choice uses `options`, Score uses ordered `levels`, and Noul uses the fixed no/yes pair with optional `false_criterion`/`true_criterion`. IDs identify outputs and do not become model tokens.
|
| 111 |
-
|
| 112 |
-
The Python API returns original record/candidate identities, logits and probabilities in the supplied order. Choice includes the chosen candidate ID. Noul includes the yes probability; it does not emit a boolean decision. An application can choose first-argmax (an exact no/yes tie selects no) or `p_yes >= 0.5` (tie selects yes), but should declare that choice. Score includes `expected_value` over the supplied values and `score`, the expected ordinal index from 0 to K−1. Neither is automatically a calibrated business utility.
|
| 113 |
-
|
| 114 |
-
All state/question/candidate text and markers count toward the complete 1024 limit. `predict_1k` checks every input before the first forward and rejects overflow without truncation. Batch sizes 1–8 are supported by this wrapper. Use it rather than directly invoking the native research-capacity entrypoint.
|
| 115 |
-
|
| 116 |
-
A compatible fine-tuned export can be used with `--native <directory> --manifest-sha256 <trusted hash>`. Only the pinned runtime revision and three-path architecture are accepted. Downloaded HF symlinks must be materialized into real files before loading. The manifest hash is an integrity check, not a reason to execute arbitrary untrusted Python.
|
| 117 |
-
|
| 118 |
-
|
| 119 |
-
### Default typed scheduling
|
| 120 |
-
|
| 121 |
-
The default `SystemOne` path groups complete, admitted questions by decision type in physical batches of eight, then restores the original request and question order. This works for many questions over one state and questions across multiple contexts. No API changes or application-side sorting are needed. The optional `batching="auto"` policy is unchanged. [Measured mixed-question scaling](PERFORMANCE.md).
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
assets/architecture.pdf
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:6aa6895f5708cda7d935991cf9c1179d9e54e22a32459968a052acf05f62a14a
|
| 3 |
-
size 173099
|
|
|
|
|
|
|
|
|
|
|
|
assets/attention-geglu.pdf
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:43b3485510971d29e66bc17b30b4bffe934ad66078598cc3e2efd63ae9b9bc7c
|
| 3 |
-
size 172797
|
|
|
|
|
|
|
|
|
|
|
|
assets/attention-geglu.png
DELETED
Git LFS Details
|
assets/candidate-readout.pdf
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:56b9e06a343e9044c38cfb8d1ba5ba121bdc7d763db5ce45b927aacdcb00c995
|
| 3 |
-
size 156814
|
|
|
|
|
|
|
|
|
|
|
|
assets/candidate-readout.svg
DELETED
assets/residual-layers.pdf
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:2c5b50bb840ae115a1264b66d535a1048172e891a33d02d507c193525cc45630
|
| 3 |
-
size 131144
|
|
|
|
|
|
|
|
|
|
|
|
assets/residual-layers.png
DELETED
Git LFS Details
|
config.json
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"decision_format": "vllm-sr-decision",
|
| 3 |
+
"format_version": 1,
|
| 4 |
+
"model_name": "Decision-1.0-Lex-0.6B",
|
| 5 |
+
"runtime_family": "vela-encoder",
|
| 6 |
+
"model_config": "native/decision_config.json",
|
| 7 |
+
"backbone": {
|
| 8 |
+
"config": "native/encoder/config.json",
|
| 9 |
+
"weights": ["native/encoder/model.safetensors"]
|
| 10 |
+
},
|
| 11 |
+
"tokenizer": {
|
| 12 |
+
"json": "native/tokenizer/tokenizer.json",
|
| 13 |
+
"config": "native/tokenizer/tokenizer_config.json",
|
| 14 |
+
"special_tokens_map": "native/tokenizer/special_tokens_map.json"
|
| 15 |
+
},
|
| 16 |
+
"decision_weights": {
|
| 17 |
+
"choice_encoder": "native/choice_encoder.safetensors",
|
| 18 |
+
"score_encoder": "native/score_encoder.safetensors",
|
| 19 |
+
"decision_heads": "native/decision_heads.safetensors"
|
| 20 |
+
}
|
| 21 |
+
}
|
decision_finetune/__init__.py
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
"""Portable JSONL fine-tuning CLI; end-to-end AMD validation remains pending."""
|
|
|
|
|
|
decision_finetune/__main__.py
DELETED
|
@@ -1,80 +0,0 @@
|
|
| 1 |
-
"""Run with PYTHONPATH=<release directory> python -m decision_finetune."""
|
| 2 |
-
import argparse
|
| 3 |
-
import json
|
| 4 |
-
import math
|
| 5 |
-
import sys
|
| 6 |
-
from . import data
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
def parser():
|
| 10 |
-
p = argparse.ArgumentParser(description="Single-AMD full1K native Choice/Noul/Score fine-tuning")
|
| 11 |
-
commands = p.add_subparsers(dest="command", required=True)
|
| 12 |
-
v = commands.add_parser("validate-jsonl", help="Structural/split checks only; does NOT establish token support")
|
| 13 |
-
v.add_argument("--train", required=True); v.add_argument("--dev", required=True)
|
| 14 |
-
v.add_argument("--selection", choices=("macro-nll", "hard-accuracy"), default="macro-nll")
|
| 15 |
-
t = commands.add_parser("train", help="Loads a real AMD model; complete admission precedes every forward")
|
| 16 |
-
for name in ("native", "manifest-sha256", "train", "dev", "output"):
|
| 17 |
-
t.add_argument("--" + name, required=True)
|
| 18 |
-
t.add_argument("--epochs", type=int, default=4)
|
| 19 |
-
t.add_argument("--logical-batch-size", type=int, default=64)
|
| 20 |
-
t.add_argument("--micro-batch-size", type=int, default=8)
|
| 21 |
-
t.add_argument("--seed", type=int, default=20260921)
|
| 22 |
-
t.add_argument("--encoder-lr", type=float, default=2.5e-5)
|
| 23 |
-
t.add_argument("--head-lr", type=float, default=1e-4)
|
| 24 |
-
t.add_argument("--lr-min", type=float, default=1e-6)
|
| 25 |
-
t.add_argument("--weight-decay", type=float, default=.01)
|
| 26 |
-
t.add_argument("--clip-norm", type=float, default=1.)
|
| 27 |
-
t.add_argument("--score-rps-weight", type=float, default=.1)
|
| 28 |
-
t.add_argument("--warmup-steps", type=int)
|
| 29 |
-
t.add_argument("--warmup-ratio", type=float, default=.1)
|
| 30 |
-
t.add_argument("--max-steps", type=int)
|
| 31 |
-
t.add_argument("--eval-steps", help="Explicit comma-separated update steps including0/final; default epoch ends")
|
| 32 |
-
t.add_argument("--selection", choices=("macro-nll", "hard-accuracy"), default="macro-nll")
|
| 33 |
-
t.add_argument("--selection-tolerance", type=float, default=1e-8)
|
| 34 |
-
t.add_argument("--save-every", type=int, default=0, help="Additional update-boundary checkpoints;0 means evaluation points only")
|
| 35 |
-
t.add_argument("--cpu-threads", type=int, default=2)
|
| 36 |
-
t.add_argument("--max-reserved-gib", type=float, default=64.)
|
| 37 |
-
t.add_argument("--allow-nondeterministic-kernels", action="store_true", help="Explicitly opt out of PyTorch deterministic algorithm enforcement")
|
| 38 |
-
t.add_argument("--resume", help="Existing immutable update-boundary checkpoint directory")
|
| 39 |
-
t.add_argument("--resume-sha256", help="Trusted manifest SHA for --resume")
|
| 40 |
-
t.add_argument("--stop-after-step", type=int, help="Clean scheduling pause after this original update; does not change the LR horizon or training config")
|
| 41 |
-
return p
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
def configuration(args):
|
| 45 |
-
fields = ("epochs", "logical_batch_size", "micro_batch_size", "seed", "encoder_lr", "head_lr", "lr_min", "weight_decay",
|
| 46 |
-
"clip_norm", "score_rps_weight", "warmup_steps", "warmup_ratio", "max_steps", "selection", "selection_tolerance",
|
| 47 |
-
"save_every", "cpu_threads", "max_reserved_gib")
|
| 48 |
-
c = {k: getattr(args, k) for k in fields}
|
| 49 |
-
data.need(all(c[k] > 0 for k in ("epochs", "logical_batch_size", "micro_batch_size", "cpu_threads")), "Positive counts required")
|
| 50 |
-
data.need(c["micro_batch_size"] <= c["logical_batch_size"] and c["micro_batch_size"] <= 8, "Micro batch must fit logical batch and ≤8")
|
| 51 |
-
data.need(0 <= c["seed"] < 2**32 and c["save_every"] >= 0, "Invalid seed/save cadence")
|
| 52 |
-
data.need(all(math.isfinite(c[k]) and c[k] > 0 for k in ("encoder_lr", "head_lr", "clip_norm", "max_reserved_gib")), "Positive finite optimizer/memory values")
|
| 53 |
-
data.need(c["max_reserved_gib"] <= 64 and all(math.isfinite(c[k]) and c[k] >= 0 for k in ("weight_decay", "score_rps_weight", "selection_tolerance"))
|
| 54 |
-
and 0 <= c["warmup_ratio"] < 1, "Invalid budget/objective/warmup")
|
| 55 |
-
data.need(math.isfinite(c["lr_min"]) and 0 <= c["lr_min"] <= min(c["encoder_lr"], c["head_lr"]), "Invalid LR floor")
|
| 56 |
-
data.need(bool(args.resume) == bool(args.resume_sha256), "Resume directory AND manifest SHA required")
|
| 57 |
-
c.update(eval_steps=None if args.eval_steps is None else [int(v) for v in args.eval_steps.split(",")],
|
| 58 |
-
deterministic_algorithms=not args.allow_nondeterministic_kernels,
|
| 59 |
-
train_precision="bf16_autocast_fp32_parameters_loss", eval_precision="fp32", eval_batch_size=8,
|
| 60 |
-
lr_convention="linear warmup to base, then cosine to lr_min at final; with no warmup step1=base; one-step schedule uses base",
|
| 61 |
-
input_limit=1024, truncation="error", selection_weighting=("equal_present_types" if args.selection == "macro-nll" else "row"))
|
| 62 |
-
return c
|
| 63 |
-
|
| 64 |
-
|
| 65 |
-
def main():
|
| 66 |
-
sys.dont_write_bytecode = True
|
| 67 |
-
args = parser().parse_args()
|
| 68 |
-
if args.command == "validate-jsonl":
|
| 69 |
-
train, dev = data.load_splits(args.train, args.dev, args.selection)
|
| 70 |
-
print(json.dumps({"status": "PASS_JSONL_AND_EXACT_SPLIT_CHECKS", "train_rows": len(train), "dev_rows": len(dev),
|
| 71 |
-
"train_sha256": data.file_sha(args.train), "dev_sha256": data.file_sha(args.dev),
|
| 72 |
-
"token_admission_performed": False, "model_calls": 0}))
|
| 73 |
-
return
|
| 74 |
-
from .run import execute
|
| 75 |
-
result = execute(args, configuration(args))
|
| 76 |
-
print(json.dumps({k: result[k] for k in ("status", "native", "native_manifest_sha256", "selected0_no_adaptation", "checkpoint") if k in result}))
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
if __name__ == "__main__":
|
| 80 |
-
main()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_finetune/data.py
DELETED
|
@@ -1,139 +0,0 @@
|
|
| 1 |
-
"""JSONL, split and deterministic schedule contracts. No model imports."""
|
| 2 |
-
import copy
|
| 3 |
-
import hashlib
|
| 4 |
-
import json
|
| 5 |
-
import math
|
| 6 |
-
from pathlib import Path
|
| 7 |
-
import unicodedata
|
| 8 |
-
|
| 9 |
-
KINDS = ("choice", "noul", "score")
|
| 10 |
-
DEFAULT_NO = "No. The statement or question is not satisfied."
|
| 11 |
-
DEFAULT_YES = "Yes. The statement or question is satisfied."
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
def need(ok, message):
|
| 15 |
-
if not ok:
|
| 16 |
-
raise ValueError(message)
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
def digest(value):
|
| 20 |
-
return hashlib.sha256(json.dumps(value, ensure_ascii=False, sort_keys=True,
|
| 21 |
-
separators=(",", ":"), allow_nan=False).encode()).hexdigest()
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
def file_sha(path):
|
| 25 |
-
h = hashlib.sha256()
|
| 26 |
-
with Path(path).open("rb") as stream:
|
| 27 |
-
for chunk in iter(lambda: stream.read(1 << 20), b""):
|
| 28 |
-
h.update(chunk)
|
| 29 |
-
return h.hexdigest()
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
def text(value):
|
| 33 |
-
need(isinstance(value, str) and value.strip(), "Nonempty text required")
|
| 34 |
-
return " ".join(unicodedata.normalize("NFKC", value).casefold().split())
|
| 35 |
-
|
| 36 |
-
|
| 37 |
-
def semantics(row):
|
| 38 |
-
need(isinstance(row, dict) and isinstance(row.get("question"), dict), "Record/question objects required")
|
| 39 |
-
text(row.get("id")); text(row.get("state_text"))
|
| 40 |
-
q = row["question"]; text(q.get("text"))
|
| 41 |
-
kind = str(q.get("type", "")).lower()
|
| 42 |
-
need(kind in KINDS, "Question type must be Choice, Noul or Score")
|
| 43 |
-
if kind == "noul":
|
| 44 |
-
options = [{"id": "no", "text": q.get("false_criterion", DEFAULT_NO)},
|
| 45 |
-
{"id": "yes", "text": q.get("true_criterion", DEFAULT_YES)}]
|
| 46 |
-
else:
|
| 47 |
-
options = q.get("options" if kind == "choice" else "levels")
|
| 48 |
-
need(isinstance(options, list) and 2 <= len(options) <= 255, "Require 2..255 candidates")
|
| 49 |
-
for option in options:
|
| 50 |
-
need(isinstance(option, dict), "Candidate object required")
|
| 51 |
-
text(option.get("id")); text(option.get("text"))
|
| 52 |
-
ids = [o["id"] for o in options]
|
| 53 |
-
need(len(set(ids)) == len(ids), "Candidate IDs must be unique")
|
| 54 |
-
values = [float(o["value"]) for o in options] if kind == "score" else list(range(len(ids)))
|
| 55 |
-
need(all(math.isfinite(v) for v in values) and all(a < b for a, b in zip(values, values[1:])), "Increasing finite Score values")
|
| 56 |
-
gold = row.get("target")
|
| 57 |
-
need(isinstance(gold, dict), "Explicit target required")
|
| 58 |
-
if kind == "noul":
|
| 59 |
-
need(set(gold) == {"probability"}, "Noul target is probability of yes")
|
| 60 |
-
yes = float(gold["probability"]); target = [1 - yes, yes]
|
| 61 |
-
elif set(gold) == {"probabilities"}:
|
| 62 |
-
target = [float(p) for p in gold["probabilities"]]
|
| 63 |
-
else:
|
| 64 |
-
need(kind == "choice" and set(gold) == {"choice_id"} and gold["choice_id"] in ids,
|
| 65 |
-
"Choice needs choice_id/probabilities; Score needs probabilities (one-hot is hard)")
|
| 66 |
-
target = [float(cid == gold["choice_id"]) for cid in ids]
|
| 67 |
-
need(len(target) == len(ids) and all(math.isfinite(p) and 0 <= p <= 1 for p in target)
|
| 68 |
-
and abs(sum(target) - 1) <= 1e-6, "Normalized finite hard/soft target required")
|
| 69 |
-
hard = row.get("hard_target_id")
|
| 70 |
-
need(hard is None or hard in ids, "hard_target_id must be an original candidate ID")
|
| 71 |
-
return kind, options, values, target
|
| 72 |
-
|
| 73 |
-
|
| 74 |
-
def input_signature(row):
|
| 75 |
-
kind, options, values, _ = semantics(row)
|
| 76 |
-
descriptions = [text(o["text"]) for o in options]
|
| 77 |
-
if kind == "choice":
|
| 78 |
-
descriptions.sort() # Reordering and opaque IDs cannot hide exact split overlap.
|
| 79 |
-
return digest({"kind": kind, "state": text(row["state_text"]),
|
| 80 |
-
"question": text(row["question"]["text"]), "descriptions": descriptions,
|
| 81 |
-
"values": values if kind == "score" else None})
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
def model_row(row):
|
| 85 |
-
# External hard evaluation labels are never passed to the collator/model.
|
| 86 |
-
return copy.deepcopy({k: v for k, v in row.items() if k != "hard_target_id"})
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
def read_jsonl(path):
|
| 90 |
-
rows = []
|
| 91 |
-
with Path(path).open(encoding="utf-8") as stream:
|
| 92 |
-
for number, line in enumerate(stream, 1):
|
| 93 |
-
need(bool(line.strip()), f"Blank JSONL line {number}")
|
| 94 |
-
row = json.loads(line); semantics(row); rows.append(row)
|
| 95 |
-
need(bool(rows) and len({r["id"] for r in rows}) == len(rows), "Nonempty split with unique IDs required")
|
| 96 |
-
return rows
|
| 97 |
-
|
| 98 |
-
|
| 99 |
-
def load_splits(train_path, dev_path, selection):
|
| 100 |
-
train, dev = read_jsonl(train_path), read_jsonl(dev_path)
|
| 101 |
-
need(not ({r["id"] for r in train} & {r["id"] for r in dev}), "TRAIN/DEV IDs overlap")
|
| 102 |
-
need(not ({input_signature(r) for r in train} & {input_signature(r) for r in dev}),
|
| 103 |
-
"TRAIN/DEV normalized complete inputs overlap")
|
| 104 |
-
# When both fields are supplied, also reject same-source dependency components.
|
| 105 |
-
components = lambda rows: {(r["source_id"], r["component_id"]) for r in rows
|
| 106 |
-
if r.get("source_id") and r.get("component_id")}
|
| 107 |
-
need(not (components(train) & components(dev)), "TRAIN/DEV source components overlap")
|
| 108 |
-
if selection == "hard-accuracy":
|
| 109 |
-
need(all(r.get("hard_target_id") is not None for r in dev),
|
| 110 |
-
"Hard-accuracy selection requires explicit hard_target_id on EVERY DEV row")
|
| 111 |
-
return train, dev
|
| 112 |
-
|
| 113 |
-
|
| 114 |
-
def make_schedule(rows, epochs, logical_batch, seed, max_steps=None):
|
| 115 |
-
steps, epoch_ends = [], []
|
| 116 |
-
for epoch in range(epochs):
|
| 117 |
-
order = sorted(range(len(rows)), key=lambda i: (digest([seed, epoch, rows[i]["id"]]), rows[i]["id"]))
|
| 118 |
-
steps.extend([order[i:i + logical_batch] for i in range(0, len(order), logical_batch)])
|
| 119 |
-
epoch_ends.append(len(steps))
|
| 120 |
-
if max_steps is not None:
|
| 121 |
-
need(1 <= max_steps <= len(steps), "max-steps must fit the declared epochs; no implicit extra epochs")
|
| 122 |
-
steps = steps[:max_steps]
|
| 123 |
-
return steps, sorted(set([0, len(steps)] + [s for s in epoch_ends if s <= len(steps)]))
|
| 124 |
-
|
| 125 |
-
|
| 126 |
-
def admit(collator, rows):
|
| 127 |
-
need(collator.max_length == 1024 and collator.state_truncation == "error", "Full1K collator required")
|
| 128 |
-
evidence = {}
|
| 129 |
-
for row in rows:
|
| 130 |
-
e = collator.encode(model_row(row), labeled=True)
|
| 131 |
-
kind, options, values, target = semantics(row)
|
| 132 |
-
need(e["id"] == row["id"] and e["kind"] == kind and e["target"] == target
|
| 133 |
-
and e["candidate_ids"] == [o["id"] for o in options] and e["values"] == values,
|
| 134 |
-
"CLI schema and loaded native collator disagree")
|
| 135 |
-
need(e["state_tokens_original"] == e["state_tokens_kept"] and e["input_tokens"] == len(e["ids"]) <= 1024,
|
| 136 |
-
"Complete input exceeds1K or was truncated; no rows are silently dropped")
|
| 137 |
-
evidence[row["id"]] = {"encoded_sha256": digest(e), "input_tokens": len(e["ids"]),
|
| 138 |
-
"token_ids_sha256": digest(e["ids"]), "record_sha256": digest(row)}
|
| 139 |
-
return evidence
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_finetune/metrics.py
DELETED
|
@@ -1,61 +0,0 @@
|
|
| 1 |
-
"""Explicit DEV metrics/selectors; no external benchmark or hidden retention gates."""
|
| 2 |
-
import math
|
| 3 |
-
from collections import defaultdict
|
| 4 |
-
from .data import need, semantics
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
def summarize(rows, predictions):
|
| 8 |
-
need(len(rows) == len(predictions) > 0, "Complete DEV predictions required")
|
| 9 |
-
groups = defaultdict(list)
|
| 10 |
-
for row, p in zip(rows, predictions):
|
| 11 |
-
kind, options, values, target = semantics(row)
|
| 12 |
-
ids = [o["id"] for o in options]; z = p["logits"]; probs = p["probabilities"]
|
| 13 |
-
need(p["id"] == row["id"] and p["type"].lower() == kind and p["candidate_ids"] == ids,
|
| 14 |
-
"Prediction ID/type/order mismatch")
|
| 15 |
-
need(len(z) == len(probs) == len(ids) and all(math.isfinite(v) for v in z + probs)
|
| 16 |
-
and all(0 <= v <= 1 for v in probs) and abs(sum(probs) - 1) <= 1e-5, "Invalid probabilities/logits")
|
| 17 |
-
peak = max(z); logsum = peak + math.log(sum(math.exp(v - peak) for v in z))
|
| 18 |
-
need(max(abs(math.exp(v - logsum) - p) for v, p in zip(z, probs)) <= 2e-5,
|
| 19 |
-
"Probabilities disagree with saved logits")
|
| 20 |
-
# NLL from logits avoids an arbitrary probability floor.
|
| 21 |
-
nll = sum(t * (logsum - v) for t, v in zip(target, z))
|
| 22 |
-
native_hat = max(range(len(probs)), key=probs.__getitem__)
|
| 23 |
-
# Original typed evaluation convention, distinct from native first argmax at exact ties.
|
| 24 |
-
hat = int(probs[1] >= .5) if kind == "noul" else native_hat
|
| 25 |
-
item = {"nll": nll, "hard_accuracy": None if row.get("hard_target_id") is None else float(ids[hat] == row["hard_target_id"]),
|
| 26 |
-
"native_argmax_accuracy": None if row.get("hard_target_id") is None else float(ids[native_hat] == row["hard_target_id"])}
|
| 27 |
-
if kind == "score":
|
| 28 |
-
pc = tc = rps = 0.
|
| 29 |
-
for a, b in zip(probs[:-1], target[:-1]):
|
| 30 |
-
pc += a; tc += b; rps += (pc - tc) ** 2
|
| 31 |
-
item["rps"] = rps / (len(ids) - 1)
|
| 32 |
-
item["expected_value_squared_error"] = (sum(a * v for a, v in zip(probs, values)) - sum(a * v for a, v in zip(target, values))) ** 2
|
| 33 |
-
groups[kind].append(item)
|
| 34 |
-
def report(items):
|
| 35 |
-
result = {"rows": len(items), "soft_nll": sum(x["nll"] for x in items) / len(items)}
|
| 36 |
-
result["hard_accuracy"] = (sum(x["hard_accuracy"] for x in items) / len(items)
|
| 37 |
-
if all(x["hard_accuracy"] is not None for x in items) else None)
|
| 38 |
-
result["native_argmax_accuracy_diagnostic"] = (sum(x["native_argmax_accuracy"] for x in items) / len(items)
|
| 39 |
-
if all(x["native_argmax_accuracy"] is not None for x in items) else None)
|
| 40 |
-
if all("rps" in x for x in items):
|
| 41 |
-
result.update(rps=sum(x["rps"] for x in items) / len(items),
|
| 42 |
-
expected_value_rmse=math.sqrt(sum(x["expected_value_squared_error"] for x in items) / len(items)))
|
| 43 |
-
return result
|
| 44 |
-
by_type = {kind: report(items) for kind, items in groups.items()}
|
| 45 |
-
return {"by_type": by_type, "row": report([x for items in groups.values() for x in items]),
|
| 46 |
-
"macro_soft_nll": sum(x["soft_nll"] for x in by_type.values()) / len(by_type),
|
| 47 |
-
"macro_weighting": "equal weight for each type present in DEV; row mean within type",
|
| 48 |
-
"hard_rule": "Noul p_yes>=0.5; Choice/Score original-order first argmax",
|
| 49 |
-
"native_argmax_diagnostic": "original-order first argmax for every type; Noul exact tie no"}
|
| 50 |
-
|
| 51 |
-
|
| 52 |
-
def better(candidate, best, selection, tolerance=1e-8):
|
| 53 |
-
if best is None:
|
| 54 |
-
return True
|
| 55 |
-
a, b = candidate["metrics"], best["metrics"]
|
| 56 |
-
if selection == "macro-nll":
|
| 57 |
-
return a["macro_soft_nll"] < b["macro_soft_nll"] - tolerance
|
| 58 |
-
need(selection == "hard-accuracy" and a["row"]["hard_accuracy"] is not None
|
| 59 |
-
and b["row"]["hard_accuracy"] is not None, "Explicit complete hard labels required")
|
| 60 |
-
delta = a["row"]["hard_accuracy"] - b["row"]["hard_accuracy"]
|
| 61 |
-
return delta > tolerance or (abs(delta) <= tolerance and a["row"]["soft_nll"] < b["row"]["soft_nll"] - tolerance)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_finetune/run.py
DELETED
|
@@ -1,237 +0,0 @@
|
|
| 1 |
-
"""Single-device training orchestration over the public API, with no research paths."""
|
| 2 |
-
import gc
|
| 3 |
-
import importlib.metadata
|
| 4 |
-
import json
|
| 5 |
-
import os
|
| 6 |
-
from pathlib import Path
|
| 7 |
-
import random
|
| 8 |
-
import sys
|
| 9 |
-
import time
|
| 10 |
-
|
| 11 |
-
from . import data, metrics, state
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
def source_identity():
|
| 15 |
-
import decision_runtime
|
| 16 |
-
roots = {"decision_finetune": Path(__file__).parent,
|
| 17 |
-
"decision_runtime": Path(decision_runtime.__file__).parent}
|
| 18 |
-
return {package + "/" + p.name: data.file_sha(p) for package, root in roots.items()
|
| 19 |
-
for p in sorted(root.glob("*.py")) if p.name not in ("checks.py", "bridge_checks.py")}
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
def setup(config):
|
| 23 |
-
import numpy as np
|
| 24 |
-
import torch
|
| 25 |
-
data.need(torch.version.hip is not None and torch.cuda.is_available() and torch.cuda.device_count() == 1,
|
| 26 |
-
"Expose exactly one real AMD ROCm device; CPU/NVIDIA training is unsupported")
|
| 27 |
-
torch.cuda.set_device(0); torch.set_num_threads(config["cpu_threads"])
|
| 28 |
-
torch.backends.cuda.matmul.allow_tf32 = False; torch.backends.cudnn.allow_tf32 = False
|
| 29 |
-
torch.backends.mha.set_fastpath_enabled(False)
|
| 30 |
-
torch.use_deterministic_algorithms(config["deterministic_algorithms"])
|
| 31 |
-
torch.cuda.set_per_process_memory_fraction(min(1., config["max_reserved_gib"] * 2**30 / torch.cuda.get_device_properties(0).total_memory))
|
| 32 |
-
random.seed(config["seed"]); np.random.seed(config["seed"]); torch.manual_seed(config["seed"])
|
| 33 |
-
environment = {"python": sys.version.split()[0], "rocm": torch.version.hip,
|
| 34 |
-
"gpu": torch.cuda.get_device_name(0),
|
| 35 |
-
"versions": {n: importlib.metadata.version(n) for n in ("torch", "transformers", "tokenizers", "safetensors", "numpy")}}
|
| 36 |
-
return torch, environment
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
def check_memory(torch, config):
|
| 40 |
-
data.need(torch.cuda.max_memory_reserved() <= config["max_reserved_gib"] * 2**30, "Reserved GPU memory cap exceeded")
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
def evaluation(api, native, dev, admission, out, step, torch):
|
| 44 |
-
# Parameter objects and optimizer state survive policy restoration unchanged.
|
| 45 |
-
api.restore_native_policy_for_export(native)
|
| 46 |
-
predictions = api.predict(native, [data.model_row(r) for r in dev], batch_size=8)
|
| 47 |
-
for row, p in zip(dev, predictions):
|
| 48 |
-
data.need(p["input_tokens"] == admission[row["id"]]["input_tokens"] <= 1024
|
| 49 |
-
and p["state_tokens_original"] == p["state_tokens_kept"], "DEV admission mismatch")
|
| 50 |
-
point = {"step": step, "metrics": metrics.summarize(dev, predictions)}
|
| 51 |
-
state.write_json(out / f"evaluation-{step:06d}.json", point)
|
| 52 |
-
with (out / f"predictions-{step:06d}.jsonl").open("x", encoding="utf-8") as stream:
|
| 53 |
-
for p in predictions:
|
| 54 |
-
stream.write(json.dumps(p, ensure_ascii=False, allow_nan=False) + "\n")
|
| 55 |
-
api.configure_training(native, max_input_tokens=1024)
|
| 56 |
-
return point
|
| 57 |
-
|
| 58 |
-
|
| 59 |
-
def update(api, native, optimizer, rows, admission, config, step, total, torch):
|
| 60 |
-
api.set_training_mode(native, True); optimizer.zero_grad(set_to_none=True)
|
| 61 |
-
scale = state.learning_rate_scale(step, total, config["warmup_steps"])
|
| 62 |
-
for group in optimizer.param_groups:
|
| 63 |
-
base = config["encoder_lr"] if group["name"].endswith("encoder") else config["head_lr"]
|
| 64 |
-
group["lr"] = state.learning_rate(step, total, config["warmup_steps"], base, config["lr_min"])
|
| 65 |
-
sums = [0., 0., 0.]; tokens = padded = 0
|
| 66 |
-
for start in range(0, len(rows), config["micro_batch_size"]):
|
| 67 |
-
micro = rows[start:start + config["micro_batch_size"]]
|
| 68 |
-
batch, encoded = native.collator([data.model_row(r) for r in micro], labeled=True, device="cuda:0")
|
| 69 |
-
data.need(all(data.digest(e) == admission[r["id"]]["encoded_sha256"] for r, e in zip(micro, encoded)),
|
| 70 |
-
"Actual training tokens/labels changed after complete admission")
|
| 71 |
-
tokens += sum(e["input_tokens"] for e in encoded); padded += len(micro) * batch["input_ids"].shape[1]
|
| 72 |
-
with torch.autocast("cuda", dtype=torch.bfloat16):
|
| 73 |
-
logits = api.forward_for_training(native, batch)
|
| 74 |
-
losses = api.loss_for_training(native, logits, batch, score_rps_weight=config["score_rps_weight"])
|
| 75 |
-
(losses[0].sum() / len(rows)).backward()
|
| 76 |
-
for i, loss in enumerate(losses):
|
| 77 |
-
sums[i] += float(loss.detach().sum())
|
| 78 |
-
del logits, losses, batch
|
| 79 |
-
check_memory(torch, config)
|
| 80 |
-
present = {r["question"]["type"].lower() for r in rows}
|
| 81 |
-
proof = api.gradient_report(native, present_types=present) if step == 1 else None
|
| 82 |
-
# AdamW must never decay a path absent from the logical batch.
|
| 83 |
-
for group in optimizer.param_groups:
|
| 84 |
-
for p in group["params"]:
|
| 85 |
-
if group["name"].split(".")[0] not in present:
|
| 86 |
-
data.need(p.grad is None, "Absent type has stale gradient")
|
| 87 |
-
elif p.grad is not None:
|
| 88 |
-
data.need(torch.isfinite(p.grad).all().item(), "Nonfinite gradient")
|
| 89 |
-
active = [p for g in optimizer.param_groups for p in g["params"]]
|
| 90 |
-
norm = float(torch.nn.utils.clip_grad_norm_(active, config["clip_norm"], error_if_nonfinite=True))
|
| 91 |
-
optimizer.step(); optimizer.zero_grad(set_to_none=True)
|
| 92 |
-
check_memory(torch, config)
|
| 93 |
-
return {"step": step, "record_ids": [r["id"] for r in rows], "logical_rows": len(rows),
|
| 94 |
-
"denominator": len(rows), "micro_batches": (len(rows) + config["micro_batch_size"] - 1) // config["micro_batch_size"],
|
| 95 |
-
"mean_loss": sums[0] / len(rows), "mean_ce": sums[1] / len(rows), "mean_rps_all_types_diagnostic": sums[2] / len(rows),
|
| 96 |
-
"effective_tokens": tokens, "padded_tokens": padded, "gradient_norm_before_clip": norm,
|
| 97 |
-
"clipped": norm > config["clip_norm"], "lr_scale": scale,
|
| 98 |
-
"learning_rates": {g["name"]: g["lr"] for g in optimizer.param_groups}, "first_step_gradient_proof": proof}
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
def execute(args, config):
|
| 102 |
-
import decision_runtime as api
|
| 103 |
-
train, dev = data.load_splits(args.train, args.dev, config["selection"])
|
| 104 |
-
schedule, default_points = data.make_schedule(train, config["epochs"], config["logical_batch_size"], config["seed"], config["max_steps"])
|
| 105 |
-
points = default_points if config["eval_steps"] is None else config["eval_steps"]
|
| 106 |
-
data.need(points == sorted(set(points)) and points[0] == 0 and points[-1] == len(schedule)
|
| 107 |
-
and all(0 <= s <= len(schedule) for s in points), "Evaluation steps must be ordered, unique, and include0/final")
|
| 108 |
-
config = {**config, "eval_steps": points, "total_steps": len(schedule)}
|
| 109 |
-
data.need(args.stop_after_step is None or 1 <= args.stop_after_step < len(schedule),
|
| 110 |
-
"A scheduling pause must be an original update before final; it does not shorten training")
|
| 111 |
-
config["warmup_steps"] = (int(len(schedule) * config["warmup_ratio"]) if config["warmup_steps"] is None else config["warmup_steps"])
|
| 112 |
-
data.need(0 <= config["warmup_steps"] < len(schedule), "Warmup must be shorter than the original full schedule")
|
| 113 |
-
train_sha, dev_sha = data.file_sha(args.train), data.file_sha(args.dev)
|
| 114 |
-
torch, environment = setup(config)
|
| 115 |
-
identity = {"config": config, "sources": source_identity(), "native_manifest_sha256": args.manifest_sha256,
|
| 116 |
-
"train_sha256": train_sha, "dev_sha256": dev_sha, "schedule_sha256": data.digest(schedule), "environment": environment}
|
| 117 |
-
out = Path(args.output).resolve()
|
| 118 |
-
data.need(not (out / "COMPLETE.json").exists(), "Completed runs are immutable")
|
| 119 |
-
if args.resume:
|
| 120 |
-
data.need(out.is_dir() and state.read_json(out / "RUN.json")["identity"] == identity,
|
| 121 |
-
"Resume must use the original output/config/data/source/environment")
|
| 122 |
-
data.need(Path(args.resume).resolve().is_relative_to(out / "attempts"), "Resume checkpoint must belong to this run")
|
| 123 |
-
else:
|
| 124 |
-
data.need(not out.exists(), "Fresh output required; use explicit --resume for an interrupted run")
|
| 125 |
-
out.mkdir(parents=True); (out / "attempts").mkdir()
|
| 126 |
-
state.write_json(out / "RUN.json", {"identity": identity, "train_path": str(Path(args.train).resolve()),
|
| 127 |
-
"dev_path": str(Path(args.dev).resolve()), "native_path": str(Path(args.native).resolve()),
|
| 128 |
-
"model_validation": "CLI integration not yet externally validated"})
|
| 129 |
-
state.write_json(out / "SCHEDULE.json", {"train_ids": [r["id"] for r in train], "row_indices": schedule})
|
| 130 |
-
import fcntl
|
| 131 |
-
with (out / "LOCK").open("a") as lock:
|
| 132 |
-
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
| 133 |
-
attempt = out / "attempts" / f"{len(list((out / 'attempts').iterdir())) + 1:04d}"
|
| 134 |
-
attempt.mkdir(); started = time.monotonic()
|
| 135 |
-
try:
|
| 136 |
-
native = api.load_native(args.native, expected_manifest_sha256=args.manifest_sha256)
|
| 137 |
-
# Same native packing; the user profile lowers only the admission cap.
|
| 138 |
-
native.collator = type(native.collator)(native.collator.tokenizer, max_length=1024, state_truncation="error")
|
| 139 |
-
admitted_train, admitted_dev = data.admit(native.collator, train), data.admit(native.collator, dev)
|
| 140 |
-
admission = {"train": admitted_train, "dev": admitted_dev, "truncated": 0, "dropped": 0,
|
| 141 |
-
"full_input_limit": 1024, "identity_sha256": data.digest(identity)}
|
| 142 |
-
state.write_json(attempt / "ADMISSION.json", admission)
|
| 143 |
-
api.configure_training(native, max_input_tokens=1024)
|
| 144 |
-
frozen = api.frozen_snapshot(native)
|
| 145 |
-
optimizer = torch.optim.AdamW(api.optimizer_groups(native, encoder_lr=config["encoder_lr"], head_lr=config["head_lr"]),
|
| 146 |
-
weight_decay=config["weight_decay"], fused=False, foreach=False)
|
| 147 |
-
layout = state.optimizer_layout(native.model, optimizer)
|
| 148 |
-
parameter_ids = [id(p) for g in optimizer.param_groups for p in g["params"]]
|
| 149 |
-
curve, best, step = [], None, 0
|
| 150 |
-
if args.resume:
|
| 151 |
-
meta, rng = state.load_checkpoint(args.resume, args.resume_sha256, native, optimizer, identity, torch)
|
| 152 |
-
api.assert_frozen(native, frozen, content=True, versions=False)
|
| 153 |
-
data.need(meta["frozen_hashes"] == {n: x["sha256"] for n, x in frozen.items()}, "Checkpoint frozen content differs from parent")
|
| 154 |
-
frozen = api.frozen_snapshot(native)
|
| 155 |
-
curve, best, step = meta["curve"], meta["best"], meta["step"]
|
| 156 |
-
data.need(0 <= step <= len(schedule) and all(p["step"] <= step for p in curve), "Resume position invalid")
|
| 157 |
-
data.need(args.stop_after_step is None or args.stop_after_step > step, "Pause must follow the resumed checkpoint")
|
| 158 |
-
if best is not None and best.get("checkpoint_manifest_sha256") is None:
|
| 159 |
-
data.need(Path(best["checkpoint"]).resolve() == Path(args.resume).resolve()
|
| 160 |
-
and best["step"] == step, "Missing previous best checkpoint identity")
|
| 161 |
-
best = {**best, "checkpoint_manifest_sha256": args.resume_sha256}
|
| 162 |
-
state.restore_rng(torch, rng) # Last potentially RNG-sensitive setup operation.
|
| 163 |
-
def checkpoint(current):
|
| 164 |
-
nonlocal best
|
| 165 |
-
directory = attempt / f"checkpoint-{current:06d}"
|
| 166 |
-
if best is not None and best.get("checkpoint") is None:
|
| 167 |
-
best = {**best, "checkpoint": str(directory.resolve())}
|
| 168 |
-
api.assert_frozen(native, frozen, content=True)
|
| 169 |
-
receipt = state.save_checkpoint(directory, native, optimizer,
|
| 170 |
-
{"identity": identity, "step": current, "curve": curve, "best": best,
|
| 171 |
-
"frozen_hashes": {n: x["sha256"] for n, x in frozen.items()}}, torch)
|
| 172 |
-
if best is not None and best["checkpoint"] == str(directory.resolve()):
|
| 173 |
-
best = {**best, "checkpoint_manifest_sha256": receipt["manifest_sha256"]}
|
| 174 |
-
state.write_json(out / "LATEST.json", receipt)
|
| 175 |
-
return receipt
|
| 176 |
-
def evaluate(current):
|
| 177 |
-
nonlocal best
|
| 178 |
-
api.assert_frozen(native, frozen, content=True)
|
| 179 |
-
opt_states = {id(p): id(s) for p, s in optimizer.state.items()}
|
| 180 |
-
point = evaluation(api, native, dev, admitted_dev, attempt, current, torch)
|
| 181 |
-
data.need(parameter_ids == [id(p) for g in optimizer.param_groups for p in g["params"]]
|
| 182 |
-
and layout == state.optimizer_layout(native.model, optimizer)
|
| 183 |
-
and opt_states == {id(p): id(s) for p, s in optimizer.state.items()}, "Evaluation broke optimizer identity")
|
| 184 |
-
curve.append(point)
|
| 185 |
-
if metrics.better(point, best, config["selection"], config["selection_tolerance"]):
|
| 186 |
-
best = {**point, "checkpoint": None}
|
| 187 |
-
checkpoint(current)
|
| 188 |
-
if not args.resume:
|
| 189 |
-
evaluate(0)
|
| 190 |
-
with (attempt / "TRAIN_LOG.jsonl").open("x", encoding="utf-8") as log:
|
| 191 |
-
for current in range(step + 1, len(schedule) + 1):
|
| 192 |
-
report = update(api, native, optimizer, [train[i] for i in schedule[current - 1]], admitted_train, config, current, len(schedule), torch)
|
| 193 |
-
api.assert_frozen(native, frozen)
|
| 194 |
-
log.write(json.dumps(report, allow_nan=False) + "\n"); log.flush()
|
| 195 |
-
if current in points:
|
| 196 |
-
evaluate(current)
|
| 197 |
-
elif config["save_every"] and current % config["save_every"] == 0:
|
| 198 |
-
checkpoint(current)
|
| 199 |
-
if current == args.stop_after_step:
|
| 200 |
-
if current not in points and not (config["save_every"] and current % config["save_every"] == 0):
|
| 201 |
-
checkpoint(current)
|
| 202 |
-
paused = {"status": "PAUSED_DECISION_FINETUNE", "identity": identity, "step": current,
|
| 203 |
-
"checkpoint": state.read_json(out / "LATEST.json"), "total_steps": len(schedule),
|
| 204 |
-
"scheduling_pause_not_selection": True, "native_exported": False}
|
| 205 |
-
state.write_json(attempt / "PAUSED.json", paused)
|
| 206 |
-
return paused
|
| 207 |
-
data.need([p["step"] for p in curve] == points and best is not None, "Missing evaluation/selected checkpoint")
|
| 208 |
-
# A resume after final checkpoint can finish export without replaying training.
|
| 209 |
-
del optimizer; gc.collect(); torch.cuda.empty_cache()
|
| 210 |
-
from safetensors.torch import load_file
|
| 211 |
-
selected_dir = Path(best["checkpoint"])
|
| 212 |
-
selected_sha = best["checkpoint_manifest_sha256"]
|
| 213 |
-
_, selected_meta = state.verify_checkpoint(selected_dir, selected_sha)
|
| 214 |
-
data.need(selected_meta["identity"] == identity and selected_meta["step"] == best["step"], "Selected checkpoint identity")
|
| 215 |
-
native.model.load_state_dict(load_file(str(selected_dir / "model.safetensors"), device="cpu"), strict=True)
|
| 216 |
-
api.assert_frozen(native, frozen, content=True, versions=False)
|
| 217 |
-
api.restore_native_policy_for_export(native)
|
| 218 |
-
export_path = attempt / "selected-native"
|
| 219 |
-
api.export_native(native, export_path, provenance={"cli": "decision_finetune", "identity": identity,
|
| 220 |
-
"selected_step": best["step"], "selection": best["metrics"], "total_steps": len(schedule),
|
| 221 |
-
"selected0_no_adaptation": best["step"] == 0, "checkpoint_manifest_sha256": selected_sha,
|
| 222 |
-
"release_or_quality_qualification": False})
|
| 223 |
-
data.need(data.file_sha(args.train) == train_sha and data.file_sha(args.dev) == dev_sha
|
| 224 |
-
and source_identity() == identity["sources"], "Source/data changed during run")
|
| 225 |
-
check_memory(torch, config)
|
| 226 |
-
result = {"status": "COMPLETE_DECISION_FINETUNE", "identity": identity, "curve": curve,
|
| 227 |
-
"selected": best, "selected0_no_adaptation": best["step"] == 0,
|
| 228 |
-
"native": str(export_path), "native_manifest_sha256": data.file_sha(export_path / "MANIFEST.json"),
|
| 229 |
-
"attempt_elapsed_seconds": time.monotonic() - started,
|
| 230 |
-
"peak_reserved_bytes": torch.cuda.max_memory_reserved(), "resumed": bool(args.resume),
|
| 231 |
-
"independent_fresh_reload_performed": False, "release_qualified": False}
|
| 232 |
-
state.write_json(out / "COMPLETE.json", result)
|
| 233 |
-
return result
|
| 234 |
-
except BaseException as error:
|
| 235 |
-
state.write_json(attempt / "FAILURE.json", {"exception_type": type(error).__name__, "message": str(error),
|
| 236 |
-
"resume_requires_explicit_checkpoint": True})
|
| 237 |
-
raise
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_finetune/state.py
DELETED
|
@@ -1,110 +0,0 @@
|
|
| 1 |
-
"""Atomic update-boundary checkpoints with named optimizer mapping and RNG state."""
|
| 2 |
-
import json
|
| 3 |
-
import os
|
| 4 |
-
from pathlib import Path
|
| 5 |
-
import random
|
| 6 |
-
import tempfile
|
| 7 |
-
from .data import need, file_sha
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
def write_json(path, value):
|
| 11 |
-
path = Path(path)
|
| 12 |
-
fd, temporary = tempfile.mkstemp(prefix="." + path.name + ".", suffix=".tmp", dir=path.parent)
|
| 13 |
-
try:
|
| 14 |
-
with os.fdopen(fd, "w", encoding="utf-8") as stream:
|
| 15 |
-
json.dump(value, stream, ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
|
| 16 |
-
stream.write("\n"); stream.flush(); os.fsync(stream.fileno())
|
| 17 |
-
os.replace(temporary, path)
|
| 18 |
-
finally:
|
| 19 |
-
if os.path.exists(temporary):
|
| 20 |
-
os.unlink(temporary)
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
def read_json(path):
|
| 24 |
-
return json.loads(Path(path).read_text(encoding="utf-8"))
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
def optimizer_layout(model, optimizer):
|
| 28 |
-
names = {id(p): name for name, p in model.named_parameters()}
|
| 29 |
-
layout = [{"name": g["name"], "parameters": [names[id(p)] for p in g["params"]]}
|
| 30 |
-
for g in optimizer.param_groups]
|
| 31 |
-
flattened = [n for group in layout for n in group["parameters"]]
|
| 32 |
-
need(len(set(flattened)) == len(flattened), "Optimizer parameter alias")
|
| 33 |
-
return layout
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
def rng_state(torch):
|
| 37 |
-
import numpy as np
|
| 38 |
-
n = np.random.get_state()
|
| 39 |
-
return {"python": random.getstate(), "numpy": [n[0], n[1].tolist(), n[2], n[3], n[4]],
|
| 40 |
-
"torch_cpu": torch.get_rng_state(), "torch_cuda": torch.cuda.get_rng_state_all()}
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
def restore_rng(torch, saved):
|
| 44 |
-
import numpy as np
|
| 45 |
-
need(len(saved["torch_cuda"]) == torch.cuda.device_count(), "CUDA RNG device count changed")
|
| 46 |
-
random.setstate(saved["python"])
|
| 47 |
-
n = saved["numpy"]; np.random.set_state((n[0], np.asarray(n[1], dtype=np.uint32), n[2], n[3], n[4]))
|
| 48 |
-
torch.set_rng_state(saved["torch_cpu"]); torch.cuda.set_rng_state_all(saved["torch_cuda"])
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
def save_checkpoint(path, native, optimizer, metadata, torch):
|
| 52 |
-
from safetensors.torch import save_file
|
| 53 |
-
path = Path(path); temporary = path.with_name(path.name + ".tmp")
|
| 54 |
-
need(not path.exists() and not temporary.exists(), "Checkpoint destinations are immutable")
|
| 55 |
-
temporary.mkdir()
|
| 56 |
-
# CPU copies are serialization only; no CPU model or forward is constructed.
|
| 57 |
-
save_file({n: t.detach().cpu().contiguous() for n, t in native.model.state_dict().items()}, str(temporary / "model.safetensors"))
|
| 58 |
-
torch.save({"optimizer": optimizer.state_dict(), "rng": rng_state(torch)}, temporary / "state.pt")
|
| 59 |
-
write_json(temporary / "metadata.json", {**metadata, "optimizer_layout": optimizer_layout(native.model, optimizer)})
|
| 60 |
-
files = {p.name: {"sha256": file_sha(p), "bytes": p.stat().st_size} for p in sorted(temporary.iterdir())}
|
| 61 |
-
write_json(temporary / "MANIFEST.json", {"schema": "decision.finetune.checkpoint.v1", "files": files})
|
| 62 |
-
os.replace(temporary, path)
|
| 63 |
-
return {"path": str(path.resolve()), "manifest_sha256": file_sha(path / "MANIFEST.json")}
|
| 64 |
-
|
| 65 |
-
|
| 66 |
-
def verify_checkpoint(path, expected_sha):
|
| 67 |
-
path = Path(path).resolve(strict=True)
|
| 68 |
-
need(file_sha(path / "MANIFEST.json") == expected_sha, "Explicit checkpoint manifest SHA mismatch")
|
| 69 |
-
manifest = read_json(path / "MANIFEST.json")
|
| 70 |
-
need(manifest["schema"] == "decision.finetune.checkpoint.v1"
|
| 71 |
-
and set(manifest["files"]) == {"model.safetensors", "state.pt", "metadata.json"}, "Checkpoint roster mismatch")
|
| 72 |
-
need({p.name for p in path.iterdir()} == set(manifest["files"]) | {"MANIFEST.json"}, "Unexpected checkpoint file")
|
| 73 |
-
for name, ref in manifest["files"].items():
|
| 74 |
-
p = path / name
|
| 75 |
-
need(not p.is_symlink() and p.is_file() and p.stat().st_size == ref["bytes"] and file_sha(p) == ref["sha256"], "Checkpoint file changed")
|
| 76 |
-
return path, read_json(path / "metadata.json")
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
def load_checkpoint(path, expected_sha, native, optimizer, identity, torch):
|
| 80 |
-
from safetensors.torch import load_file
|
| 81 |
-
path, meta = verify_checkpoint(path, expected_sha)
|
| 82 |
-
need(meta["identity"] == identity, "Resume source/data/native/config/schedule/environment identity changed")
|
| 83 |
-
need(meta["optimizer_layout"] == optimizer_layout(native.model, optimizer), "Named optimizer group/order mismatch")
|
| 84 |
-
parameter_ids = [id(p) for p in native.model.parameters()]
|
| 85 |
-
native.model.load_state_dict(load_file(str(path / "model.safetensors"), device="cpu"), strict=True)
|
| 86 |
-
need([id(p) for p in native.model.parameters()] == parameter_ids, "Loading replaced parameters")
|
| 87 |
-
# Only trust locally produced, explicitly hash-verified checkpoints. No arbitrary pickle globals.
|
| 88 |
-
state = torch.load(path / "state.pt", map_location="cpu", weights_only=True)
|
| 89 |
-
optimizer.load_state_dict(state["optimizer"])
|
| 90 |
-
need(meta["optimizer_layout"] == optimizer_layout(native.model, optimizer), "Restored optimizer mapping drift")
|
| 91 |
-
for p, item in optimizer.state.items():
|
| 92 |
-
need(all(item[k].shape == p.shape and torch.isfinite(item[k]).all().item() for k in ("exp_avg", "exp_avg_sq")), "Invalid AdamW moments")
|
| 93 |
-
need(0 < float(item["step"]) <= meta["step"], "Invalid per-parameter AdamW step")
|
| 94 |
-
return meta, state["rng"]
|
| 95 |
-
|
| 96 |
-
|
| 97 |
-
def learning_rate_scale(step, total, warmup):
|
| 98 |
-
import math
|
| 99 |
-
need(1 <= step <= total and 0 <= warmup < total, "Invalid original LR horizon")
|
| 100 |
-
if warmup and step <= warmup:
|
| 101 |
-
return step / warmup
|
| 102 |
-
progress = ((step - warmup) / (total - warmup) if warmup
|
| 103 |
-
else (step - 1) / (total - 1) if total > 1 else 0.)
|
| 104 |
-
return .5 * (1 + math.cos(math.pi * progress))
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
def learning_rate(step, total, warmup, base, minimum):
|
| 108 |
-
need(0 <= minimum <= base, "LR floor must fit every optimizer group")
|
| 109 |
-
scale = learning_rate_scale(step, total, warmup)
|
| 110 |
-
return base * scale if warmup and step <= warmup else minimum + (base - minimum) * scale
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_finetune/system_one.py
DELETED
|
@@ -1,34 +0,0 @@
|
|
| 1 |
-
"""Keep System One inference and fine-tuning input semantics identical."""
|
| 2 |
-
import copy
|
| 3 |
-
import json
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
def system_one_training_rows(request, targets, *, request_id, source_id,
|
| 7 |
-
component_id, hard_target_ids=None):
|
| 8 |
-
"""Convert a typed request plus separate native targets to training rows.
|
| 9 |
-
|
| 10 |
-
targets is keyed by question ID. Noul uses {'probability': p_yes}, Choice
|
| 11 |
-
{'choice_id': label} or {'probabilities': [...]}, Score {'probabilities': [...]}.
|
| 12 |
-
The caller assigns shared source/component IDs across related examples so
|
| 13 |
-
existing TRAIN/DEV validation can reject component overlap.
|
| 14 |
-
"""
|
| 15 |
-
from decision_inference._system_one import system_one_records
|
| 16 |
-
from .data import semantics
|
| 17 |
-
for value in (request_id, source_id, component_id):
|
| 18 |
-
if not isinstance(value, str) or not value.strip():
|
| 19 |
-
raise ValueError("Explicit nonempty request/source/component IDs required")
|
| 20 |
-
rows = system_one_records(request)
|
| 21 |
-
qids = {row["question"]["id"] for row in rows}
|
| 22 |
-
if not isinstance(targets, dict) or set(targets) != qids:
|
| 23 |
-
raise ValueError("Exactly one target per question ID is required")
|
| 24 |
-
if hard_target_ids is not None and (not isinstance(hard_target_ids, dict) or set(hard_target_ids) != qids):
|
| 25 |
-
raise ValueError("Hard evaluation labels must cover exactly the question IDs")
|
| 26 |
-
for i, row in enumerate(rows):
|
| 27 |
-
qid = row["question"]["id"]
|
| 28 |
-
row.update(id=json.dumps([request_id, i], ensure_ascii=False, separators=(",", ":")),
|
| 29 |
-
source_id=source_id, component_id=component_id,
|
| 30 |
-
target=copy.deepcopy(targets[qid]))
|
| 31 |
-
if hard_target_ids is not None:
|
| 32 |
-
row["hard_target_id"] = hard_target_ids[qid]
|
| 33 |
-
semantics(row)
|
| 34 |
-
return rows
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/__init__.py
DELETED
|
@@ -1,5 +0,0 @@
|
|
| 1 |
-
from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
|
| 2 |
-
from ._auto import predict_auto_1k
|
| 3 |
-
from ._system_one import SystemOne, system_one_records
|
| 4 |
-
|
| 5 |
-
__all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k", "SystemOne", "system_one_records"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/_auto.py
DELETED
|
@@ -1,34 +0,0 @@
|
|
| 1 |
-
"""Optional homogeneous FP32 batching, with no increase in total padding."""
|
| 2 |
-
from dataclasses import replace
|
| 3 |
-
from ._request import request_collator, predict_1k
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
def _padded_tokens(entries, size):
|
| 7 |
-
return sum(len(chunk) * max(row['input_tokens'] for row in chunk)
|
| 8 |
-
for start in range(0, len(entries), size)
|
| 9 |
-
for chunk in [entries[start:start + size]])
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
def _capacity(entries):
|
| 13 |
-
if (len(entries) > 8 and len({row['kind'] for row in entries}) == 1
|
| 14 |
-
and _padded_tokens(entries, 32) <= _padded_tokens(entries, 8)):
|
| 15 |
-
return 32
|
| 16 |
-
return 8
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
def predict_auto_1k(native, records):
|
| 20 |
-
"""Opt in to at most32 consecutive rows; same complete1024 FP32 contract.
|
| 21 |
-
|
| 22 |
-
Default predict_1k remains unchanged. Mixed requests or extra padding retain
|
| 23 |
-
B8. Every input admits first; errors never trigger partial outputs/retries.
|
| 24 |
-
This changes physical shapes and can change FP32 rounding, not task semantics.
|
| 25 |
-
"""
|
| 26 |
-
records = list(records)
|
| 27 |
-
if len(records) <= 8:
|
| 28 |
-
return predict_1k(native, records, batch_size=8)
|
| 29 |
-
guard = request_collator(native.collator, records)
|
| 30 |
-
from decision_runtime import predict
|
| 31 |
-
result = predict(replace(native, collator=guard), records,
|
| 32 |
-
batch_size=_capacity(guard._encoded))
|
| 33 |
-
guard.finish()
|
| 34 |
-
return result
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/_grouped.py
DELETED
|
@@ -1,16 +0,0 @@
|
|
| 1 |
-
"""Stable typed scheduling behind SystemOne; preserve request and answer order."""
|
| 2 |
-
from dataclasses import replace
|
| 3 |
-
|
| 4 |
-
def predict_grouped_1k(native,records,*,batch_size=8):
|
| 5 |
-
from decision_inference._request import request_collator
|
| 6 |
-
from decision_runtime import predict
|
| 7 |
-
if type(batch_size) is not int or batch_size!=8:raise ValueError('SystemOne typed scheduling uses physical batch size 8')
|
| 8 |
-
records=list(records);guard=request_collator(native.collator,records)
|
| 9 |
-
# Admit every original row before scheduling or any forward.
|
| 10 |
-
order=sorted(range(len(records)),key=lambda i:records[i]['question']['type'])
|
| 11 |
-
sorted_rows=[records[i]for i in order]
|
| 12 |
-
guard._records=tuple(sorted_rows);guard._encoded=[guard._encoded[i]for i in order]
|
| 13 |
-
predictions=predict(replace(native,collator=guard),sorted_rows,batch_size=8);guard.finish();restored=[None]*len(records)
|
| 14 |
-
for i,result in zip(order,predictions):restored[i]=result
|
| 15 |
-
if any(x is None for x in restored):raise RuntimeError('Incomplete scheduled predictions')
|
| 16 |
-
return restored
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/_request.py
DELETED
|
@@ -1,58 +0,0 @@
|
|
| 1 |
-
"""Isolated complete-1K request-local encoding reuse; native arithmetic is untouched."""
|
| 2 |
-
from dataclasses import replace
|
| 3 |
-
|
| 4 |
-
MAX_INPUT_TOKENS=1024
|
| 5 |
-
|
| 6 |
-
def request_collator(native_collator,records):
|
| 7 |
-
"""Encode each occurrence once before any model call; never cache by external ID."""
|
| 8 |
-
class RequestEncodedCollator(type(native_collator)):
|
| 9 |
-
def __init__(self):
|
| 10 |
-
super().__init__(native_collator.tokenizer,max_length=MAX_INPUT_TOKENS,state_truncation='error')
|
| 11 |
-
self._records=tuple(records);self._encoded=[];self._cursor=0;self._active=None
|
| 12 |
-
for row in self._records:
|
| 13 |
-
encoded=super().encode(row,labeled=False)
|
| 14 |
-
if (encoded['input_tokens']>MAX_INPUT_TOKENS
|
| 15 |
-
or encoded['state_tokens_original']!=encoded['state_tokens_kept']):
|
| 16 |
-
raise ValueError('Native collator violated the complete-input 1K profile')
|
| 17 |
-
self._encoded.append(encoded)
|
| 18 |
-
|
| 19 |
-
def encode(self,record,labeled=False):
|
| 20 |
-
# Only the unchanged native tensor assembler consumes this request-local queue.
|
| 21 |
-
if labeled or self._active is None:
|
| 22 |
-
raise ValueError('Request-local collator supports only its admitted inference batch')
|
| 23 |
-
expected,encoded=next(self._active)
|
| 24 |
-
if record is not expected:
|
| 25 |
-
raise ValueError('Request-local record occurrence order changed')
|
| 26 |
-
return encoded
|
| 27 |
-
|
| 28 |
-
def __call__(self,records,labeled=False,device='cpu'):
|
| 29 |
-
records=list(records);end=self._cursor+len(records)
|
| 30 |
-
if labeled or end>len(self._records) or any(a is not b for a,b in zip(records,self._records[self._cursor:end])):
|
| 31 |
-
raise ValueError('Request-local record occurrence order changed')
|
| 32 |
-
self._active=iter(zip(records,self._encoded[self._cursor:end]))
|
| 33 |
-
try:
|
| 34 |
-
# Original allocation, padding, targets/values, kinds and device transfer.
|
| 35 |
-
result=super().__call__(records,labeled=False,device=device)
|
| 36 |
-
finally:
|
| 37 |
-
self._active=None
|
| 38 |
-
self._cursor=end
|
| 39 |
-
return result
|
| 40 |
-
|
| 41 |
-
def finish(self):
|
| 42 |
-
if self._cursor!=len(self._records):
|
| 43 |
-
raise ValueError('Native prediction did not consume the admitted request')
|
| 44 |
-
return RequestEncodedCollator()
|
| 45 |
-
|
| 46 |
-
def predict_1k(native,records,*,batch_size=8):
|
| 47 |
-
"""Same native prediction/output/errors; one encoding per occurrence per call.
|
| 48 |
-
|
| 49 |
-
No model mode/inventory check is skipped. No result, token array or record is
|
| 50 |
-
retained across API calls. Same IDs with different inputs remain distinct.
|
| 51 |
-
"""
|
| 52 |
-
if type(batch_size) is not int or not 1<=batch_size<=8:
|
| 53 |
-
raise ValueError('The product profile permits integer batch sizes 1..8')
|
| 54 |
-
records=list(records);guard=request_collator(native.collator,records)
|
| 55 |
-
from decision_runtime import predict
|
| 56 |
-
result=predict(replace(native,collator=guard),records,batch_size=batch_size)
|
| 57 |
-
guard.finish()
|
| 58 |
-
return result
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/_system_one.py
DELETED
|
@@ -1,217 +0,0 @@
|
|
| 1 |
-
"""System One request/answer schema over complete-input Decision inference.
|
| 2 |
-
|
| 3 |
-
Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping;
|
| 4 |
-
Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache.
|
| 5 |
-
"""
|
| 6 |
-
import copy
|
| 7 |
-
import json
|
| 8 |
-
import math
|
| 9 |
-
|
| 10 |
-
MAX_QUESTIONS = 128
|
| 11 |
-
MAX_REQUESTS = 128
|
| 12 |
-
MAX_DECISIONS = 512
|
| 13 |
-
MAX_REQUEST_BYTES = 2 * 1024 * 1024
|
| 14 |
-
PUBLIC_MODELS = {
|
| 15 |
-
"Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
|
| 16 |
-
"Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
|
| 17 |
-
}
|
| 18 |
-
|
| 19 |
-
|
| 20 |
-
def _identifier(value, label):
|
| 21 |
-
if not isinstance(value, str) or not value.strip() or len(value) > 128:
|
| 22 |
-
raise ValueError(label + " must be a nonempty string of at most 128 characters")
|
| 23 |
-
return value
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
def _json(value):
|
| 27 |
-
# JSON objects must have string keys: never silently coerce Python keys.
|
| 28 |
-
def check(item):
|
| 29 |
-
if isinstance(item, dict):
|
| 30 |
-
if not all(isinstance(k, str) for k in item):
|
| 31 |
-
raise ValueError("JSON object keys must be strings")
|
| 32 |
-
for v in item.values():
|
| 33 |
-
check(v)
|
| 34 |
-
elif isinstance(item, list):
|
| 35 |
-
for v in item:
|
| 36 |
-
check(v)
|
| 37 |
-
elif item is not None and not isinstance(item, (str, bool, int, float)):
|
| 38 |
-
raise ValueError("Only JSON values are supported")
|
| 39 |
-
try:
|
| 40 |
-
check(value)
|
| 41 |
-
return json.dumps(value, ensure_ascii=False, sort_keys=True,
|
| 42 |
-
separators=(",", ":"), allow_nan=False)
|
| 43 |
-
except (TypeError, RecursionError, UnicodeError) as exc:
|
| 44 |
-
raise ValueError("Invalid JSON content") from exc
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
def _content(value, label):
|
| 48 |
-
if isinstance(value, str):
|
| 49 |
-
if not value.strip():
|
| 50 |
-
raise ValueError(label + " must not be empty")
|
| 51 |
-
return value
|
| 52 |
-
if isinstance(value, (dict, list)):
|
| 53 |
-
return _json(value)
|
| 54 |
-
raise ValueError(label + " must be text, an object, or an array")
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
def system_one_records(request):
|
| 58 |
-
"""Validate one wire request and return native rows, without loading a model.
|
| 59 |
-
|
| 60 |
-
Full token admission occurs in predict_1k before the first model forward.
|
| 61 |
-
External record IDs should be made unique when combining training examples.
|
| 62 |
-
"""
|
| 63 |
-
if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}:
|
| 64 |
-
raise ValueError("A request contains exactly model, state and questions")
|
| 65 |
-
_identifier(request["model"], "Model")
|
| 66 |
-
if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES:
|
| 67 |
-
raise ValueError("Request exceeds 2 MiB; no input is truncated")
|
| 68 |
-
state = _content(request["state"], "State")
|
| 69 |
-
questions = request["questions"]
|
| 70 |
-
if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS:
|
| 71 |
-
raise ValueError("Provide 1..128 named questions")
|
| 72 |
-
rows = []
|
| 73 |
-
for index, (qid, item) in enumerate(questions.items()):
|
| 74 |
-
_identifier(qid, "Question ID")
|
| 75 |
-
if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"}
|
| 76 |
-
or not {"type", "instructions"} <= set(item)):
|
| 77 |
-
raise ValueError(qid + ": use type, instructions and optional criteria")
|
| 78 |
-
kind = item["type"]
|
| 79 |
-
if kind not in ("noul", "choice", "score"):
|
| 80 |
-
raise ValueError(qid + ": type must be noul, choice or score")
|
| 81 |
-
q = {"id": qid, "type": kind.capitalize(),
|
| 82 |
-
"text": _content(item["instructions"], qid + ".instructions")}
|
| 83 |
-
criteria = item.get("criteria")
|
| 84 |
-
if kind == "choice":
|
| 85 |
-
if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255:
|
| 86 |
-
raise ValueError(qid + ": Choice requires 2..255 named options")
|
| 87 |
-
q["options"] = []
|
| 88 |
-
for name, description in criteria.items():
|
| 89 |
-
_identifier(name, "Choice option")
|
| 90 |
-
text = name if description is None else name + ": " + _content(description, qid + ".criteria")
|
| 91 |
-
q["options"].append({"id": name, "text": text})
|
| 92 |
-
elif kind == "score":
|
| 93 |
-
if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
|
| 94 |
-
raise ValueError(qid + ": Score requires 2..10 ordered levels")
|
| 95 |
-
q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")}
|
| 96 |
-
for i, v in enumerate(criteria)]
|
| 97 |
-
elif "criteria" in item:
|
| 98 |
-
if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
|
| 99 |
-
raise ValueError(qid + ": Noul criteria accept false and true only")
|
| 100 |
-
for key in ("false", "true"):
|
| 101 |
-
if key in criteria:
|
| 102 |
-
q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key)
|
| 103 |
-
rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q})
|
| 104 |
-
return rows
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
def _answer(row, prediction):
|
| 108 |
-
q = row["question"]
|
| 109 |
-
kind = q["type"].lower()
|
| 110 |
-
ids = (["no", "yes"] if kind == "noul" else
|
| 111 |
-
[v["id"] for v in q["options" if kind == "choice" else "levels"]])
|
| 112 |
-
if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"]
|
| 113 |
-
or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids
|
| 114 |
-
or type(prediction.get("input_tokens")) is not int
|
| 115 |
-
or not 1 <= prediction["input_tokens"] <= 1024
|
| 116 |
-
or type(prediction.get("state_tokens_original")) is not int
|
| 117 |
-
or prediction["state_tokens_original"] < 0
|
| 118 |
-
or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")):
|
| 119 |
-
raise RuntimeError("Prediction identity or complete-input profile mismatch")
|
| 120 |
-
p = prediction.get("probabilities")
|
| 121 |
-
if (not isinstance(p, list) or len(p) != len(ids)
|
| 122 |
-
or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p)
|
| 123 |
-
or abs(sum(p) - 1) > 2e-5):
|
| 124 |
-
raise RuntimeError("Invalid prediction probabilities")
|
| 125 |
-
answer = {"type": kind}
|
| 126 |
-
if kind == "noul":
|
| 127 |
-
if prediction.get("probability") != p[1]:
|
| 128 |
-
raise RuntimeError("Native Noul probability mismatch")
|
| 129 |
-
answer["noul"] = p[1]
|
| 130 |
-
return answer
|
| 131 |
-
best = ids[max(range(len(p)), key=p.__getitem__)]
|
| 132 |
-
if prediction.get("choice_id") != best or prediction.get("confidence") != max(p):
|
| 133 |
-
raise RuntimeError("Native Choice/confidence mismatch")
|
| 134 |
-
answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"])
|
| 135 |
-
if kind == "choice":
|
| 136 |
-
answer["choice"] = best
|
| 137 |
-
else:
|
| 138 |
-
score = prediction.get("score")
|
| 139 |
-
if (type(score) not in (int, float) or not math.isfinite(score)
|
| 140 |
-
or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5):
|
| 141 |
-
raise RuntimeError("Native ordinal Score mismatch")
|
| 142 |
-
# Preserve native FP32 arithmetic, not a new CPU reduction.
|
| 143 |
-
answer["score"] = score
|
| 144 |
-
answer["legend"] = {v["id"]: v["text"] for v in q["levels"]}
|
| 145 |
-
return answer
|
| 146 |
-
|
| 147 |
-
|
| 148 |
-
class SystemOne:
|
| 149 |
-
"""Local System One API for a loaded Kai, Lex or compatible fine-tune.
|
| 150 |
-
|
| 151 |
-
evaluate(request) accepts the HTTP body shape; system_one(**request) is its
|
| 152 |
-
Python equivalent. batch(requests) flattens independent states into GPU
|
| 153 |
-
batches and restores the original request/question order. Default B8 groups rows by decision type;
|
| 154 |
-
batching='auto' opts into the published homogeneous padding-aware B32 path.
|
| 155 |
-
"""
|
| 156 |
-
def __init__(self, native, *, model=None, batching="default"):
|
| 157 |
-
if model is None:
|
| 158 |
-
model = next((name for name, sha in PUBLIC_MODELS.items()
|
| 159 |
-
if sha == native.manifest_sha256), None)
|
| 160 |
-
_identifier(model, "Model (required for a custom fine-tune)")
|
| 161 |
-
# Do not let a different loaded checkpoint claim a published identity.
|
| 162 |
-
if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]:
|
| 163 |
-
raise ValueError("Loaded checkpoint does not match the public model name")
|
| 164 |
-
if batching not in ("default", "auto"):
|
| 165 |
-
raise ValueError("batching must be default or auto")
|
| 166 |
-
self.native, self.model, self.batching = native, model, batching
|
| 167 |
-
|
| 168 |
-
def system_one(self, *, state, questions, model=None):
|
| 169 |
-
return self.evaluate({"model": self.model if model is None else model,
|
| 170 |
-
"state": state, "questions": questions})
|
| 171 |
-
|
| 172 |
-
def evaluate(self, request):
|
| 173 |
-
return self.batch([request])[0]
|
| 174 |
-
|
| 175 |
-
def batch(self, requests):
|
| 176 |
-
if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS:
|
| 177 |
-
raise ValueError("Provide 1..128 request objects")
|
| 178 |
-
if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES:
|
| 179 |
-
raise ValueError("Combined request exceeds 2 MiB")
|
| 180 |
-
# Detach mutable caller inputs before conversion/admission/inference.
|
| 181 |
-
requests = copy.deepcopy(requests)
|
| 182 |
-
groups = []
|
| 183 |
-
for request in requests:
|
| 184 |
-
rows = system_one_records(request)
|
| 185 |
-
if request["model"] != self.model:
|
| 186 |
-
raise ValueError("Request model does not match this loaded model")
|
| 187 |
-
groups.append(rows)
|
| 188 |
-
count = sum(map(len, groups))
|
| 189 |
-
if count > MAX_DECISIONS:
|
| 190 |
-
raise ValueError("Provide at most 512 decisions in one batch")
|
| 191 |
-
records, slots = [], []
|
| 192 |
-
# Question-major order permits the same question across many states to
|
| 193 |
-
# share a physical batch; external IDs never decide caching or grouping.
|
| 194 |
-
for qi in range(max(map(len, groups))):
|
| 195 |
-
for ri, group in enumerate(groups):
|
| 196 |
-
if qi < len(group):
|
| 197 |
-
row = group[qi]
|
| 198 |
-
row["id"] = f"systemone:{ri}:{qi}"
|
| 199 |
-
records.append(row)
|
| 200 |
-
slots.append((ri, qi))
|
| 201 |
-
if self.batching == "auto":
|
| 202 |
-
from ._auto import predict_auto_1k
|
| 203 |
-
predictions = predict_auto_1k(self.native, records)
|
| 204 |
-
else:
|
| 205 |
-
from ._grouped import predict_grouped_1k
|
| 206 |
-
predictions = predict_grouped_1k(self.native, records, batch_size=8)
|
| 207 |
-
if len(predictions) != len(records):
|
| 208 |
-
raise RuntimeError("Incomplete model result; no partial answers returned")
|
| 209 |
-
values = [[None] * len(group) for group in groups]
|
| 210 |
-
tokens = [0] * len(groups)
|
| 211 |
-
for row, prediction, (ri, qi) in zip(records, predictions, slots):
|
| 212 |
-
values[ri][qi] = _answer(row, prediction)
|
| 213 |
-
tokens[ri] += prediction["input_tokens"]
|
| 214 |
-
return [{"model": self.model,
|
| 215 |
-
"answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])},
|
| 216 |
-
"usage": {"input_tokens": tokens[ri], "output_tokens": 0}}
|
| 217 |
-
for ri, group in enumerate(groups)]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_inference/profile.py
DELETED
|
@@ -1,51 +0,0 @@
|
|
| 1 |
-
"""Complete-input 1K product profile around an unchanged native runtime."""
|
| 2 |
-
from dataclasses import replace
|
| 3 |
-
|
| 4 |
-
MAX_INPUT_TOKENS = 1024
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
class Complete1KCollator:
|
| 8 |
-
"""Use the native collator's actual length error, never truncate fields.
|
| 9 |
-
|
| 10 |
-
Constructible with a tokenizer before any model is loaded. A whole request
|
| 11 |
-
is admitted before predict, and every physical batch before tensor assembly.
|
| 12 |
-
"""
|
| 13 |
-
def __init__(self, native_collator):
|
| 14 |
-
self.base = type(native_collator)(native_collator.tokenizer,
|
| 15 |
-
max_length=MAX_INPUT_TOKENS,
|
| 16 |
-
state_truncation="error")
|
| 17 |
-
self.tokenizer = self.base.tokenizer
|
| 18 |
-
self.marker, self.pad = self.base.marker, self.base.pad
|
| 19 |
-
self.tensor_batches = 0
|
| 20 |
-
|
| 21 |
-
def tokens(self, text):
|
| 22 |
-
return self.base.tokens(text)
|
| 23 |
-
|
| 24 |
-
def encode(self, row, labeled=False):
|
| 25 |
-
encoded = self.base.encode(row, labeled=labeled)
|
| 26 |
-
if (encoded["input_tokens"] > MAX_INPUT_TOKENS
|
| 27 |
-
or encoded["state_tokens_original"] != encoded["state_tokens_kept"]):
|
| 28 |
-
raise ValueError("Native collator violated the complete-input 1K profile")
|
| 29 |
-
return encoded
|
| 30 |
-
|
| 31 |
-
def admit(self, records):
|
| 32 |
-
return [self.encode(row, labeled=False) for row in records]
|
| 33 |
-
|
| 34 |
-
def __call__(self, records, labeled=False, device="cpu"):
|
| 35 |
-
records = list(records)
|
| 36 |
-
# All encodes finish before delegating any tensor allocation.
|
| 37 |
-
for row in records:
|
| 38 |
-
self.encode(row, labeled=labeled)
|
| 39 |
-
self.tensor_batches += 1
|
| 40 |
-
return self.base(records, labeled=labeled, device=device)
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
def predict_1k(native, records, *, batch_size=8):
|
| 45 |
-
"""Admit every complete input, then reuse its encoding within this request.
|
| 46 |
-
|
| 47 |
-
Native prediction, batch boundaries and outputs are unchanged. Encodings
|
| 48 |
-
are released with the synchronous call; no state activations are cached.
|
| 49 |
-
"""
|
| 50 |
-
from ._request import predict_1k as predict_admitted
|
| 51 |
-
return predict_admitted(native, records, batch_size=batch_size)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_runtime/__init__.py
DELETED
|
@@ -1,11 +0,0 @@
|
|
| 1 |
-
"""Source-only portable training adapter; real AMD validation is still pending."""
|
| 2 |
-
from .native import load_native, export_native, predict
|
| 3 |
-
from .training import (configure_training, set_training_mode, forward_for_training,
|
| 4 |
-
loss_for_training, optimizer_groups, inventory,
|
| 5 |
-
frozen_snapshot, assert_frozen, gradient_report,
|
| 6 |
-
restore_native_policy_for_export)
|
| 7 |
-
|
| 8 |
-
__all__ = ["load_native", "export_native", "predict", "configure_training",
|
| 9 |
-
"set_training_mode", "forward_for_training", "loss_for_training",
|
| 10 |
-
"optimizer_groups", "inventory", "frozen_snapshot", "assert_frozen",
|
| 11 |
-
"gradient_report", "restore_native_policy_for_export"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_runtime/_compat.py
DELETED
|
@@ -1,20 +0,0 @@
|
|
| 1 |
-
"""Supported self-contained runtime revision; weights may differ. No path bindings."""
|
| 2 |
-
RUNTIME_SHA256 = {'__init__.py': '1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010',
|
| 3 |
-
'artifacts.py': 'c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3',
|
| 4 |
-
'contract.py': '51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726',
|
| 5 |
-
'infer.py': '9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315',
|
| 6 |
-
'model.py': '8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b',
|
| 7 |
-
'packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
|
| 8 |
-
'policy/__init__.py': '0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114',
|
| 9 |
-
'policy/artifacts.py': '5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2',
|
| 10 |
-
'policy/contract.py': '0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67',
|
| 11 |
-
'policy/infer.py': '672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896',
|
| 12 |
-
'policy/model.py': 'eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca',
|
| 13 |
-
'policy/packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
|
| 14 |
-
'policy/reference/__init__.py': 'c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28',
|
| 15 |
-
'policy/reference/artifacts.py': '3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159',
|
| 16 |
-
'policy/reference/infer.py': '672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896',
|
| 17 |
-
'policy/reference/model.py': '733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26',
|
| 18 |
-
'policy/reference/modernbert_sdpa_layout.py': '0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8',
|
| 19 |
-
'policy/reference/packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
|
| 20 |
-
'training_policy.py': '6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5'}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_runtime/native.py
DELETED
|
@@ -1,152 +0,0 @@
|
|
| 1 |
-
"""Load the export's own verified runtime; never import an experiment package."""
|
| 2 |
-
from dataclasses import dataclass, field
|
| 3 |
-
from pathlib import Path, PurePosixPath
|
| 4 |
-
import copy
|
| 5 |
-
import hashlib
|
| 6 |
-
import importlib
|
| 7 |
-
import importlib.util
|
| 8 |
-
import json
|
| 9 |
-
import re
|
| 10 |
-
import sys
|
| 11 |
-
import uuid
|
| 12 |
-
|
| 13 |
-
from ._compat import RUNTIME_SHA256
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
def sha256(path):
|
| 17 |
-
digest = hashlib.sha256()
|
| 18 |
-
with Path(path).open("rb") as stream:
|
| 19 |
-
for chunk in iter(lambda: stream.read(1 << 20), b""):
|
| 20 |
-
digest.update(chunk)
|
| 21 |
-
return digest.hexdigest()
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
def verify_files(directory, expected_manifest_sha256):
|
| 25 |
-
"""Stdlib pre-import verification, including every weight and runtime file."""
|
| 26 |
-
root = Path(directory).resolve(strict=True)
|
| 27 |
-
if not re.fullmatch(r"[0-9a-f]{64}", expected_manifest_sha256 or ""):
|
| 28 |
-
raise ValueError("An explicit trusted MANIFEST SHA256 is required")
|
| 29 |
-
if sha256(root / "MANIFEST.json") != expected_manifest_sha256:
|
| 30 |
-
raise ValueError("Native manifest identity mismatch")
|
| 31 |
-
manifest = json.loads((root / "MANIFEST.json").read_text())
|
| 32 |
-
files = manifest.get("files")
|
| 33 |
-
if manifest.get("schema") != "decision.files.v1" or not isinstance(files, dict):
|
| 34 |
-
raise ValueError("Unsupported native manifest")
|
| 35 |
-
for name, ref in files.items():
|
| 36 |
-
if not isinstance(name, str) or not isinstance(ref, dict):
|
| 37 |
-
raise ValueError("Invalid manifest entry")
|
| 38 |
-
rel = PurePosixPath(name)
|
| 39 |
-
if (rel.is_absolute() or ".." in rel.parts
|
| 40 |
-
or rel.as_posix() != name or "\\" in name or not rel.parts):
|
| 41 |
-
raise ValueError("Unsafe manifest path")
|
| 42 |
-
path = root / name
|
| 43 |
-
if (not path.is_file() or path.is_symlink()
|
| 44 |
-
or set(ref) != {"bytes", "sha256"}
|
| 45 |
-
or path.stat().st_size != ref["bytes"] or sha256(path) != ref["sha256"]):
|
| 46 |
-
raise ValueError("Native file mismatch: " + name)
|
| 47 |
-
actual = set()
|
| 48 |
-
for path in root.rglob("*"):
|
| 49 |
-
if path.is_symlink():
|
| 50 |
-
raise ValueError("Materialized native required; symlinks are not supported")
|
| 51 |
-
rel = path.relative_to(root)
|
| 52 |
-
if path.is_file() and rel.as_posix() != "MANIFEST.json":
|
| 53 |
-
if "__pycache__" in rel.parts and path.suffix == ".pyc":
|
| 54 |
-
continue
|
| 55 |
-
actual.add(rel.as_posix())
|
| 56 |
-
if actual != set(files):
|
| 57 |
-
raise ValueError("Native file roster mismatch")
|
| 58 |
-
runtime = {name: ref["sha256"] for name, ref in files.items() if name.endswith(".py")}
|
| 59 |
-
if runtime != RUNTIME_SHA256:
|
| 60 |
-
raise ValueError("Unsupported runtime revision; architecture name alone is insufficient")
|
| 61 |
-
return root, manifest
|
| 62 |
-
|
| 63 |
-
|
| 64 |
-
@dataclass
|
| 65 |
-
class Native:
|
| 66 |
-
directory: Path
|
| 67 |
-
manifest_sha256: str
|
| 68 |
-
model: object
|
| 69 |
-
collator: object
|
| 70 |
-
config: dict
|
| 71 |
-
model_api: object
|
| 72 |
-
contract: object
|
| 73 |
-
artifacts: object
|
| 74 |
-
training_state: dict | None = None
|
| 75 |
-
last_training_provenance: dict | None = None
|
| 76 |
-
package_name: str = field(default="", repr=False)
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
def _amd_device(device):
|
| 80 |
-
import torch
|
| 81 |
-
selected = torch.device(device)
|
| 82 |
-
if selected.type != "cuda" or torch.version.hip is None or not torch.cuda.is_available():
|
| 83 |
-
raise ValueError("This runtime requires a real ROCm CUDA device; no CPU fallback")
|
| 84 |
-
if selected.index is None:
|
| 85 |
-
selected = torch.device("cuda", torch.cuda.current_device())
|
| 86 |
-
return selected
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
def load_native(directory, *, expected_manifest_sha256, device="cuda:0"):
|
| 90 |
-
"""Load any compatible weight export with the pinned public runtime revision.
|
| 91 |
-
|
| 92 |
-
Verification precedes executing bundled Python. This function loads a model;
|
| 93 |
-
importing decision_runtime or running its stdlib checks does not.
|
| 94 |
-
"""
|
| 95 |
-
root, manifest = verify_files(directory, expected_manifest_sha256)
|
| 96 |
-
selected = _amd_device(device)
|
| 97 |
-
namespace = "_decision_native_" + uuid.uuid4().hex
|
| 98 |
-
spec = importlib.util.spec_from_file_location(namespace, root / "__init__.py",
|
| 99 |
-
submodule_search_locations=[str(root)])
|
| 100 |
-
package = importlib.util.module_from_spec(spec)
|
| 101 |
-
sys.modules[namespace] = package
|
| 102 |
-
try:
|
| 103 |
-
spec.loader.exec_module(package)
|
| 104 |
-
artifacts = importlib.import_module(namespace + ".artifacts")
|
| 105 |
-
contract = importlib.import_module(namespace + ".contract")
|
| 106 |
-
api = importlib.import_module(namespace + ".model")
|
| 107 |
-
declared = contract.validate_config(json.loads((root / "decision_config.json").read_text()))
|
| 108 |
-
if (declared["arm"], declared["training_arm"]) != ("all22", "S22"):
|
| 109 |
-
raise ValueError("Only the complete three-path all22/S22 architecture is supported")
|
| 110 |
-
model, collator, cfg = artifacts.load_export(root, device=str(selected))
|
| 111 |
-
if (cfg["arm"], cfg["training_arm"]) != ("all22", "S22"):
|
| 112 |
-
raise ValueError("Only the complete three-path all22/S22 architecture is supported")
|
| 113 |
-
if type(model) is not api.DecisionModel or artifacts.verify_native(root) != (manifest, cfg):
|
| 114 |
-
raise ValueError("Loaded class or native identity mismatch")
|
| 115 |
-
return Native(root, expected_manifest_sha256, model, collator, cfg, api,
|
| 116 |
-
contract, artifacts, package_name=namespace)
|
| 117 |
-
except BaseException:
|
| 118 |
-
for name in tuple(sys.modules):
|
| 119 |
-
if name == namespace or name.startswith(namespace + "."):
|
| 120 |
-
del sys.modules[name]
|
| 121 |
-
raise
|
| 122 |
-
|
| 123 |
-
|
| 124 |
-
def _native_mode(native):
|
| 125 |
-
if native.training_state is not None:
|
| 126 |
-
raise ValueError("Restore native policy before native inference or export")
|
| 127 |
-
native.model.verify_inventory(check_values=False)
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
def predict(native, records, *, batch_size=8):
|
| 131 |
-
"""Unchanged bundled inference, FP32; no training graph is used here."""
|
| 132 |
-
_native_mode(native)
|
| 133 |
-
device = next(native.model.parameters()).device
|
| 134 |
-
_amd_device(device)
|
| 135 |
-
return native.model_api.predict(native.model, native.collator, records,
|
| 136 |
-
batch_size=batch_size, device=str(device), precision="fp32")
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
def export_native(native, directory, *, provenance):
|
| 140 |
-
"""Use the original strict exporter after explicit policy restoration."""
|
| 141 |
-
_native_mode(native)
|
| 142 |
-
if native.last_training_provenance is None or not isinstance(provenance, dict):
|
| 143 |
-
raise ValueError("Restored training provenance and an explicit user provenance dict required")
|
| 144 |
-
if Path(directory).exists():
|
| 145 |
-
raise ValueError("Export destination must be fresh")
|
| 146 |
-
details = {"parent_native_manifest_sha256": native.manifest_sha256,
|
| 147 |
-
"parent_provenance": copy.deepcopy(native.config.get("provenance", {})),
|
| 148 |
-
"training_adapter": copy.deepcopy(native.last_training_provenance),
|
| 149 |
-
"user": copy.deepcopy(provenance)}
|
| 150 |
-
json.dumps(details, allow_nan=False)
|
| 151 |
-
return native.artifacts.export_model(native.model, native.collator.tokenizer,
|
| 152 |
-
directory, native.config, details)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
decision_runtime/training.py
DELETED
|
@@ -1,310 +0,0 @@
|
|
| 1 |
-
"""Explicit all-types training graph for the unchanged three-path native."""
|
| 2 |
-
import math
|
| 3 |
-
from pathlib import Path
|
| 4 |
-
from .native import sha256, _amd_device
|
| 5 |
-
|
| 6 |
-
KINDS = ("choice", "noul", "score")
|
| 7 |
-
FROZEN = ("encoder.embeddings.tok_embeddings.weight",
|
| 8 |
-
"encoder.embeddings.norm.weight", "type_embedding.weight")
|
| 9 |
-
COUNTS = {"tensors": 489, "parameters": 571909635, "active_tensors": 486,
|
| 10 |
-
"active_parameters": 375298563, "frozen_tensors": 3,
|
| 11 |
-
"frozen_parameters": 196611072}
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
def parameter_group(name):
|
| 15 |
-
if name in FROZEN:
|
| 16 |
-
return "frozen"
|
| 17 |
-
for kind, prefixes in {"choice": ("choice_blocks.", "choice_final_norm."),
|
| 18 |
-
"noul": ("encoder.layers.", "encoder.final_norm."),
|
| 19 |
-
"score": ("score_blocks.", "score_final_norm.")}.items():
|
| 20 |
-
if name.startswith(prefixes):
|
| 21 |
-
return kind + ".encoder"
|
| 22 |
-
for kind in KINDS:
|
| 23 |
-
for suffix, prefix in (("head", "heads."), ("scorer", "scorers.")):
|
| 24 |
-
if name.startswith(prefix + kind + "."):
|
| 25 |
-
return kind + "." + suffix
|
| 26 |
-
raise ValueError("Unsupported parameter name: " + name)
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
def _roots(model):
|
| 30 |
-
return {"choice": (model.choice_blocks, model.choice_final_norm, model.heads["choice"], model.scorers["choice"]),
|
| 31 |
-
"noul": (model.encoder.layers, model.encoder.final_norm, model.heads["noul"], model.scorers["noul"]),
|
| 32 |
-
"score": (model.score_blocks, model.score_final_norm, model.heads["score"], model.scorers["score"])}
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
def _identities(model):
|
| 36 |
-
return {n: (id(p), p.untyped_storage().data_ptr()) for n, p in model.named_parameters()}
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
def _api(native):
|
| 40 |
-
m, c, api = native.model, native.contract, native.model_api
|
| 41 |
-
if (type(m) is not api.DecisionModel or m.arm != "all22" or m.training_arm != "S22"
|
| 42 |
-
or m.trainability_mode != c.training_policy("S22")):
|
| 43 |
-
raise ValueError("Original all22/S22 native class and metadata required")
|
| 44 |
-
api.runtime_gate()
|
| 45 |
-
return api
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
def inventory(native):
|
| 49 |
-
import torch
|
| 50 |
-
_api(native)
|
| 51 |
-
m = native.model
|
| 52 |
-
shapes = native.contract.all_shapes(m.arm, m.training_arm)
|
| 53 |
-
raw = list(m.named_parameters(remove_duplicate=False))
|
| 54 |
-
names = {n for n, _ in raw}
|
| 55 |
-
if names != set(shapes) or set(m.state_dict()) != names or len(raw) != len(names):
|
| 56 |
-
raise ValueError("Exact native parameter/state roster required")
|
| 57 |
-
if len({id(p) for _, p in raw}) != len(raw) or len({p.untyped_storage().data_ptr() for _, p in raw}) != len(raw):
|
| 58 |
-
raise ValueError("Aliased parameters/storage are unsupported")
|
| 59 |
-
if any(tuple(p.shape) != tuple(shapes[n]) or p.dtype != torch.float32
|
| 60 |
-
or p.requires_grad != (parameter_group(n) != "frozen") for n, p in raw):
|
| 61 |
-
raise ValueError("Training shape, FP32 parameter dtype or policy drift")
|
| 62 |
-
active = [p for n, p in raw if parameter_group(n) != "frozen"]
|
| 63 |
-
counts = {"tensors": len(raw), "parameters": sum(p.numel() for _, p in raw),
|
| 64 |
-
"active_tensors": len(active), "active_parameters": sum(p.numel() for p in active),
|
| 65 |
-
"frozen_tensors": len(raw) - len(active),
|
| 66 |
-
"frozen_parameters": sum(p.numel() for n, p in raw if parameter_group(n) == "frozen")}
|
| 67 |
-
if counts != COUNTS:
|
| 68 |
-
raise ValueError("Unsupported native geometry")
|
| 69 |
-
groups = {g: [n for n, _ in raw if parameter_group(n) == g]
|
| 70 |
-
for g in ("frozen", *(k + "." + part for k in KINDS for part in ("encoder", "head", "scorer")))}
|
| 71 |
-
return {"counts": counts, "groups": groups}
|
| 72 |
-
|
| 73 |
-
|
| 74 |
-
def configure_training(native, *, max_input_tokens=1024):
|
| 75 |
-
"""Enable all three existing encoder/head/scorer paths; allocate no parameters."""
|
| 76 |
-
api = _api(native)
|
| 77 |
-
if native.training_state is not None:
|
| 78 |
-
raise ValueError("Restore before configuring again")
|
| 79 |
-
if type(max_input_tokens) is not int or not 1 <= max_input_tokens <= native.config["packing"]["max_length"]:
|
| 80 |
-
raise ValueError("Training token limit must fit the native complete-input cap")
|
| 81 |
-
m = native.model
|
| 82 |
-
m.verify_inventory(check_values=False)
|
| 83 |
-
native.contract.validate_encoder_config(m.encoder.config)
|
| 84 |
-
for blocks, _, _, _ in _roots(m).values():
|
| 85 |
-
if len(blocks) != 22:
|
| 86 |
-
raise ValueError("Three full 22-layer paths required")
|
| 87 |
-
for index, layer in enumerate(blocks):
|
| 88 |
-
api.layer_gate(layer, index)
|
| 89 |
-
for name, p in m.named_parameters():
|
| 90 |
-
p.requires_grad_(parameter_group(name) != "frozen")
|
| 91 |
-
p.grad = None
|
| 92 |
-
native.training_state = {"identities": _identities(m), "training": False,
|
| 93 |
-
"max_input_tokens": max_input_tokens}
|
| 94 |
-
set_training_mode(native, False)
|
| 95 |
-
return inventory(native)
|
| 96 |
-
|
| 97 |
-
|
| 98 |
-
def _assert_modes(native):
|
| 99 |
-
m, state = native.model, native.training_state
|
| 100 |
-
if state is None:
|
| 101 |
-
raise ValueError("Call configure_training first")
|
| 102 |
-
selected = {id(child) for roots in _roots(m).values() for root in roots for child in root.modules()}
|
| 103 |
-
if m.training != state["training"] or _identities(m) != state["identities"]:
|
| 104 |
-
raise ValueError("Mode or parameter/optimizer identity changed")
|
| 105 |
-
for module in m.modules():
|
| 106 |
-
if module is not m and module.training != (state["training"] and id(module) in selected):
|
| 107 |
-
raise ValueError("Active/frozen module mode drift; use set_training_mode")
|
| 108 |
-
if any(p.requires_grad != (parameter_group(n) != "frozen") for n, p in m.named_parameters()):
|
| 109 |
-
raise ValueError("Training requires_grad drift")
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
def set_training_mode(native, training=True):
|
| 113 |
-
if type(training) is not bool or native.training_state is None:
|
| 114 |
-
raise ValueError("Configured boolean training mode required")
|
| 115 |
-
m = native.model
|
| 116 |
-
m.eval()
|
| 117 |
-
m.training = training
|
| 118 |
-
for roots in _roots(m).values():
|
| 119 |
-
for module in roots:
|
| 120 |
-
module.train(training)
|
| 121 |
-
native.training_state["training"] = training
|
| 122 |
-
_assert_modes(native)
|
| 123 |
-
|
| 124 |
-
|
| 125 |
-
def optimizer_groups(native, *, encoder_lr, head_lr):
|
| 126 |
-
"""Return disjoint complete groups; no optimizer or scheduler is created."""
|
| 127 |
-
if any(not math.isfinite(x) or x <= 0 for x in (encoder_lr, head_lr)):
|
| 128 |
-
raise ValueError("Positive finite learning rates required")
|
| 129 |
-
info = inventory(native)
|
| 130 |
-
params = dict(native.model.named_parameters())
|
| 131 |
-
return [{"name": group, "params": [params[n] for n in names],
|
| 132 |
-
"lr": encoder_lr if group.endswith("encoder") else head_lr}
|
| 133 |
-
for group, names in info["groups"].items() if group != "frozen"]
|
| 134 |
-
|
| 135 |
-
|
| 136 |
-
def _validate_batch(native, batch):
|
| 137 |
-
import torch
|
| 138 |
-
required = {"input_ids", "attention_mask", "marker_positions", "valid_candidates", "targets", "values", "kind_ids"}
|
| 139 |
-
if set(batch) != required or any(not isinstance(t, torch.Tensor) for t in batch.values()):
|
| 140 |
-
raise ValueError("Use the native collator's complete batch dictionary")
|
| 141 |
-
ids, mask, kinds = batch["input_ids"], batch["attention_mask"], batch["kind_ids"]
|
| 142 |
-
pos, valid = batch["marker_positions"], batch["valid_candidates"]
|
| 143 |
-
_amd_device(ids.device)
|
| 144 |
-
if any(p.device != ids.device for p in native.model.parameters()) or any(t.device != ids.device for t in batch.values()):
|
| 145 |
-
raise ValueError("All parameters and batch tensors must share the actual AMD device")
|
| 146 |
-
if (ids.dtype != torch.long or ids.ndim != 2 or ids.shape[0] < 1
|
| 147 |
-
or not 1 <= ids.shape[1] <= native.training_state["max_input_tokens"]
|
| 148 |
-
or mask.dtype != torch.bool or mask.shape != ids.shape
|
| 149 |
-
or kinds.dtype != torch.long or kinds.shape != (ids.shape[0],)):
|
| 150 |
-
raise ValueError("Invalid complete B x L input")
|
| 151 |
-
if (pos.dtype != torch.long or valid.dtype != torch.bool or pos.ndim != 2
|
| 152 |
-
or pos.shape != valid.shape or pos.shape[0] != ids.shape[0]
|
| 153 |
-
or not 2 <= pos.shape[1] <= 255 or (pos < 0).any().item()
|
| 154 |
-
or (pos >= ids.shape[1]).any().item()):
|
| 155 |
-
raise ValueError("Invalid candidate markers")
|
| 156 |
-
counts = valid.sum(-1)
|
| 157 |
-
prefix = torch.arange(valid.shape[1], device=ids.device)[None] < counts[:, None]
|
| 158 |
-
if (counts < 2).any().item() or not torch.equal(prefix, valid):
|
| 159 |
-
raise ValueError("At least two candidates, then contiguous padding required")
|
| 160 |
-
if not mask.gather(1, pos)[valid].all().item() or not mask.any(-1).all().item():
|
| 161 |
-
raise ValueError("Candidate marker points into padding")
|
| 162 |
-
if not torch.equal(mask, torch.arange(ids.shape[1], device=ids.device)[None] < mask.sum(-1)[:, None]):
|
| 163 |
-
raise ValueError("Native right-padding layout required")
|
| 164 |
-
if ((pos[:, 1:] <= pos[:, :-1]) & valid[:, 1:]).any().item():
|
| 165 |
-
raise ValueError("Candidate markers must retain their original sequence order")
|
| 166 |
-
if ((kinds < 0) | (kinds > 2)).any().item() or (counts[kinds == 1] != 2).any().item():
|
| 167 |
-
raise ValueError("Unknown type or non-binary Noul")
|
| 168 |
-
if batch["targets"].shape != valid.shape or batch["values"].shape != valid.shape:
|
| 169 |
-
raise ValueError("Target/value shape mismatch")
|
| 170 |
-
values = batch["values"]
|
| 171 |
-
if values.dtype != torch.float32 or not torch.isfinite(values).all().item():
|
| 172 |
-
raise ValueError("Finite FP32 values required")
|
| 173 |
-
scored = (kinds == 2)[:, None] & valid[:, 1:]
|
| 174 |
-
if ((values[:, 1:] <= values[:, :-1]) & scored).any().item():
|
| 175 |
-
raise ValueError("Score values must remain strictly increasing")
|
| 176 |
-
|
| 177 |
-
|
| 178 |
-
def forward_for_training(native, batch):
|
| 179 |
-
"""Full-B×L suffixes, typed row gathering only at heads; no native monkeypatch."""
|
| 180 |
-
import torch
|
| 181 |
-
from torch.utils.checkpoint import checkpoint
|
| 182 |
-
api = _api(native)
|
| 183 |
-
_assert_modes(native)
|
| 184 |
-
_validate_batch(native, batch)
|
| 185 |
-
if native.training_state["training"] and not torch.is_grad_enabled():
|
| 186 |
-
raise ValueError("Training forward requires autograd")
|
| 187 |
-
m = native.model
|
| 188 |
-
ids, mask = batch["input_ids"], batch["attention_mask"]
|
| 189 |
-
kinds, valid, positions = batch["kind_ids"], batch["valid_candidates"], batch["marker_positions"]
|
| 190 |
-
routes = api.route_indices(kinds)
|
| 191 |
-
m.encoder._maybe_set_compile()
|
| 192 |
-
position_ids = torch.arange(ids.shape[1], device=ids.device).unsqueeze(0)
|
| 193 |
-
global_mask, local_mask = m.encoder._update_attention_mask(mask, output_attentions=False)
|
| 194 |
-
kwargs = dict(attention_mask=global_mask, sliding_window_mask=local_mask,
|
| 195 |
-
position_ids=position_ids, cu_seqlens=None, max_seqlen=None, output_attentions=False)
|
| 196 |
-
with torch.no_grad():
|
| 197 |
-
embedded = m.encoder.embeddings(input_ids=ids, inputs_embeds=None)
|
| 198 |
-
hidden = {}
|
| 199 |
-
roots = _roots(m)
|
| 200 |
-
for name, indices in routes:
|
| 201 |
-
if indices.numel() == 0:
|
| 202 |
-
continue
|
| 203 |
-
blocks, norm, _, _ = roots[name]
|
| 204 |
-
h = m._suffix(embedded, blocks, norm, kwargs)
|
| 205 |
-
with torch.no_grad():
|
| 206 |
-
type_value = m.type_embedding(kinds)[:, None, :]
|
| 207 |
-
hidden[name] = h + type_value.to(h.dtype)
|
| 208 |
-
pad = ~mask.bool()
|
| 209 |
-
out = torch.empty(positions.shape, device=ids.device, dtype=torch.float32)
|
| 210 |
-
for name, indices in routes:
|
| 211 |
-
if indices.numel() == 0:
|
| 212 |
-
continue
|
| 213 |
-
h = hidden[name].index_select(0, indices)
|
| 214 |
-
branch_pad = pad.index_select(0, indices)
|
| 215 |
-
for layer in m.heads[name]:
|
| 216 |
-
h = (checkpoint(layer, h, src_key_padding_mask=branch_pad, use_reentrant=False)
|
| 217 |
-
if layer.training and m.head_config["gradient_checkpointing"]
|
| 218 |
-
else layer(h, src_key_padding_mask=branch_pad))
|
| 219 |
-
loc = positions.index_select(0, indices)
|
| 220 |
-
markers = torch.gather(h, 1, loc[:, :, None].expand(-1, -1, h.shape[-1]))
|
| 221 |
-
logits = m.scorers[name](markers).squeeze(-1).float()
|
| 222 |
-
out = api.scatter_rows(out, indices, logits)
|
| 223 |
-
return out.masked_fill(~valid, torch.finfo(torch.float32).min)
|
| 224 |
-
|
| 225 |
-
|
| 226 |
-
def loss_for_training(native, logits, batch, *, score_rps_weight=0.0):
|
| 227 |
-
"""Return (per-row total, CE, RPS); caller chooses the logical-batch denominator."""
|
| 228 |
-
import torch
|
| 229 |
-
if not math.isfinite(score_rps_weight) or score_rps_weight < 0:
|
| 230 |
-
raise ValueError("Nonnegative finite Score RPS weight required")
|
| 231 |
-
target, valid = batch["targets"], batch["valid_candidates"]
|
| 232 |
-
if (logits.shape != target.shape or target.shape != valid.shape
|
| 233 |
-
or target.dtype != torch.float32 or not torch.isfinite(logits).all().item()
|
| 234 |
-
or not torch.isfinite(target).all().item() or ((target < 0) | (target > 1)).any().item()
|
| 235 |
-
or target[~valid].count_nonzero().item() or not torch.allclose(target.sum(-1), torch.ones_like(target[:, 0]), atol=1e-6, rtol=0)):
|
| 236 |
-
raise ValueError("Explicit normalized hard/soft targets and finite masked logits required")
|
| 237 |
-
return native.model_api.typed_loss(logits, batch, score_rps_weight=score_rps_weight)
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
def _tensor_sha(tensor):
|
| 241 |
-
import hashlib
|
| 242 |
-
return hashlib.sha256(tensor.detach().cpu().contiguous().numpy().tobytes()).hexdigest()
|
| 243 |
-
|
| 244 |
-
|
| 245 |
-
def frozen_snapshot(native):
|
| 246 |
-
inventory(native)
|
| 247 |
-
_assert_modes(native)
|
| 248 |
-
ps = dict(native.model.named_parameters())
|
| 249 |
-
return {n: {"version": int(ps[n]._version), "sha256": _tensor_sha(ps[n])} for n in FROZEN}
|
| 250 |
-
|
| 251 |
-
|
| 252 |
-
def assert_frozen(native, snapshot, *, content=False, versions=True):
|
| 253 |
-
_assert_modes(native)
|
| 254 |
-
ps = dict(native.model.named_parameters())
|
| 255 |
-
if set(snapshot) != set(FROZEN):
|
| 256 |
-
raise ValueError("Expected exactly the three shared frozen parameters")
|
| 257 |
-
for name, saved in snapshot.items():
|
| 258 |
-
p = ps[name]
|
| 259 |
-
if (p.requires_grad or p.grad is not None or (versions and int(p._version) != saved["version"])
|
| 260 |
-
or (content and _tensor_sha(p) != saved["sha256"])):
|
| 261 |
-
raise ValueError("Frozen parameter changed: " + name)
|
| 262 |
-
return True
|
| 263 |
-
|
| 264 |
-
|
| 265 |
-
def gradient_report(native, *, present_types):
|
| 266 |
-
"""Check groups actually present in the logical batch; absent grads must be None."""
|
| 267 |
-
import torch
|
| 268 |
-
if not present_types or not set(present_types) <= set(KINDS):
|
| 269 |
-
raise ValueError("Explicit nonempty logical-batch type set required")
|
| 270 |
-
info = inventory(native)
|
| 271 |
-
ps = dict(native.model.named_parameters())
|
| 272 |
-
report = {}
|
| 273 |
-
for group, names in info["groups"].items():
|
| 274 |
-
active = group != "frozen" and group.split(".")[0] in present_types
|
| 275 |
-
gradients = [ps[n].grad for n in names]
|
| 276 |
-
if not active:
|
| 277 |
-
if any(g is not None for g in gradients):
|
| 278 |
-
raise ValueError("Frozen/absent group has stale gradients: " + group)
|
| 279 |
-
continue
|
| 280 |
-
if any(g is None or not torch.isfinite(g).all().item() for g in gradients):
|
| 281 |
-
raise ValueError("Missing or nonfinite gradient: " + group)
|
| 282 |
-
norms = [float(g.float().norm().item()) for g in gradients]
|
| 283 |
-
if not any(n > 0 for n in norms):
|
| 284 |
-
raise ValueError("No nonzero gradient: " + group)
|
| 285 |
-
report[group] = {"tensors": len(names), "nonzero_tensors": sum(n > 0 for n in norms),
|
| 286 |
-
"l2": math.sqrt(sum(n * n for n in norms))}
|
| 287 |
-
return report
|
| 288 |
-
|
| 289 |
-
|
| 290 |
-
def restore_native_policy_for_export(native):
|
| 291 |
-
"""Clear gradients and restore the original strict metadata, without replacing parameters."""
|
| 292 |
-
_assert_modes(native)
|
| 293 |
-
inventory(native)
|
| 294 |
-
m, state, contract = native.model, native.training_state, native.contract
|
| 295 |
-
policy = contract.training_policy(m.training_arm)
|
| 296 |
-
sources = {p.name: sha256(p) for p in sorted(Path(__file__).parent.glob("*.py")) if p.name != "checks.py"}
|
| 297 |
-
provenance = {"actual_training_policy": "all_three_full22_paths_and_heads",
|
| 298 |
-
"counts": dict(COUNTS), "complete_input_limit": state["max_input_tokens"],
|
| 299 |
-
"native_policy_is_not_actual_training_provenance": True,
|
| 300 |
-
"restored_native_trainability_mode": policy, "adapter_sources_sha256": sources}
|
| 301 |
-
for n, p in m.named_parameters():
|
| 302 |
-
p.requires_grad_(contract.trainable_name(n, policy))
|
| 303 |
-
p.grad = None
|
| 304 |
-
m.eval()
|
| 305 |
-
m.verify_inventory(check_values=False)
|
| 306 |
-
if _identities(m) != state["identities"]:
|
| 307 |
-
raise ValueError("Restoration replaced parameter objects")
|
| 308 |
-
native.training_state = None
|
| 309 |
-
native.last_training_provenance = provenance
|
| 310 |
-
return provenance
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
EVALUATION.md → evaluation/EVALUATION.md
RENAMED
|
File without changes
|
METHODS.md → evaluation/METHODS.md
RENAMED
|
@@ -6,6 +6,6 @@ Development used a state-component split of the original official TRAIN: 4,800 d
|
|
| 6 |
|
| 7 |
For the original normalized soft label vector q and original semantic hard-label one-hot h, the target is 0.5q + 0.5h. The loss is cross-entropy summed over 64 decisions and divided by 64, with no RPS, consistency, distillation or calibration fitting. Candidate order and semantic descriptions are retained. Each logical batch uses eight physical batches of eight. Training uses BF16 autocast with FP32 parameters and loss, AdamW (weight decay 0.01), gradient clipping at 1, and seed 20260921. There is no warmup. Encoder/head starting learning rates are 2.5e-5/1e-4 and each follows `1e-6 + (base - 1e-6) * (1 + cos(pi * (step - 1) / 749)) / 2` for steps 1–750.
|
| 8 |
|
| 9 |
-
All complete packed inputs fit 1,024 tokens. The
|
| 10 |
|
| 11 |
The final TEST was opened only after this checkpoint and recipe were frozen. It was evaluated once against the official typed specialist on identical complete inputs. Neither TEST checkpoint selection nor temperature fitting was performed. [EVALUATION.md](EVALUATION.md) records the point improvement and its limits. This package does not transfer Kai's multilingual quality or latency measurements to Lex.
|
|
|
|
| 6 |
|
| 7 |
For the original normalized soft label vector q and original semantic hard-label one-hot h, the target is 0.5q + 0.5h. The loss is cross-entropy summed over 64 decisions and divided by 64, with no RPS, consistency, distillation or calibration fitting. Candidate order and semantic descriptions are retained. Each logical batch uses eight physical batches of eight. Training uses BF16 autocast with FP32 parameters and loss, AdamW (weight decay 0.01), gradient clipping at 1, and seed 20260921. There is no warmup. Encoder/head starting learning rates are 2.5e-5/1e-4 and each follows `1e-6 + (base - 1e-6) * (1 + cos(pi * (step - 1) / 749)) / 2` for steps 1–750.
|
| 8 |
|
| 9 |
+
All complete packed inputs fit 1,024 tokens. The original native manifest was `f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6`. That historical export included training-policy metadata; the model-only release retains the model weights and configuration. The actual all-types update is recorded in the export provenance and [TRAINING_PROVENANCE.json](../TRAINING_PROVENANCE.json).
|
| 10 |
|
| 11 |
The final TEST was opened only after this checkpoint and recipe were frozen. It was evaluated once against the official typed specialist on identical complete inputs. Neither TEST checkpoint selection nor temperature fitting was performed. [EVALUATION.md](EVALUATION.md) records the point improvement and its limits. This package does not transfer Kai's multilingual quality or latency measurements to Lex.
|
MIXED_QUESTION_SCALING.json → evaluation/MIXED_QUESTION_SCALING.json
RENAMED
|
File without changes
|
evaluation/PERFORMANCE.md
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Lex: measured Decision runtime latency
|
| 2 |
+
|
| 3 |
+
On the evaluated AMD ROCm runtime, 128 mixed Choice, Noul and Score questions took **154.41 ms median**, 56.6% less than the previous default runtime on the same fixed workload.
|
| 4 |
+
|
| 5 |
+

|
| 6 |
+
|
| 7 |
+
| Questions | Previous p50 / p95 (ms) | Typed scheduling p50 / p95 (ms) |
|
| 8 |
+
|---:|---:|---:|
|
| 9 |
+
| 1 | 13.22 / 13.32 | 13.20 / 13.40 |
|
| 10 |
+
| 8 | 28.07 / 28.32 | 28.02 / 28.31 |
|
| 11 |
+
| 32 | 93.76 / 94.06 | 53.43 / 53.77 |
|
| 12 |
+
| 64 | 181.78 / 184.53 | 87.39 / 88.09 |
|
| 13 |
+
| 128 | 355.97 / 360.25 | 154.41 / 155.93 |
|
| 14 |
+
|
| 15 |
+
The request repeats three fixed questions over one context; only question count and bookkeeping IDs change. Each question still receives its own contextual computation. Measurements used the exact released weights, FP32 inference, physical batch size eight, 10 warmup pairs and 30 alternating AB/BA pairs per point, with GPU synchronization. They include local request conversion, tokenization, model execution and answer assembly; transport and Studio are excluded.
|
| 16 |
+
|
| 17 |
+
These results describe the evaluated runtime and workload, not a latency guarantee for a separately distributed runtime. The model-only repository contains no serving code. [Samples and validation details](MIXED_QUESTION_SCALING.json) · [SVG](../assets/mixed-question-scaling.svg) · [PDF](../assets/mixed-question-scaling.pdf)
|
TECHNICAL_VALIDATION.json → evaluation/TECHNICAL_VALIDATION.json
RENAMED
|
File without changes
|
VALIDATION.md → evaluation/VALIDATION.md
RENAMED
|
@@ -4,6 +4,6 @@ The final refit completed 750 updates and exported a native bundle with 571,909,
|
|
| 4 |
|
| 5 |
After releasing the training model and optimizer, a separate process loaded the final native and repeated those 64 TRAIN decisions in eight FP32 batches. Logits and probabilities matched the saved final predictions exactly, with zero hard-label flips. This was a technical reload check, not another quality evaluation.
|
| 6 |
|
| 7 |
-
The
|
| 8 |
|
| 9 |
-
The
|
|
|
|
| 4 |
|
| 5 |
After releasing the training model and optimizer, a separate process loaded the final native and repeated those 64 TRAIN decisions in eight FP32 batches. Logits and probabilities matched the saved final predictions exactly, with zero hard-label flips. This was a technical reload check, not another quality evaluation.
|
| 6 |
|
| 7 |
+
The original release included Python inference and fine-tuning modules copied from the tested Kai distribution. Its package-isolation check loaded the bundled native and evaluated the same 64 TRAIN decisions in eight FP32 batches from an external working directory. Maximum logit/probability differences from the independent final-refit fresh reference were 0.0/0.0, with zero hard flips. All 489 parameter contents and versions were unchanged; no gradients, updates or TEST evaluation occurred. This is historical validation of those model objects and runtime; executable code is no longer part of the model-only repository. See [TECHNICAL_VALIDATION.json](TECHNICAL_VALIDATION.json).
|
| 8 |
|
| 9 |
+
The earlier fine-tuning CLI's bounded continuous-versus-resume and independent-reload checks are interface evidence, not new Lex training results. Lex's final training used the fixed research recipe described in [METHODS.md](METHODS.md). The evaluated runtime profile used complete packed length ≤1,024 and FP32 inference on compatible AMD ROCm. No browser end-to-end latency, CPU/NVIDIA inference, multilingual-specialist quality or larger-window quality is claimed.
|
examples/decisions.jsonl
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
{"id":"route","state_text":"The customer asks for a refund for a duplicate charge.","question":{"id":"route-q","type":"Choice","text":"Which team should handle this request?","options":[{"id":"billing","text":"Billing and payments"},{"id":"technical","text":"Technical support"},{"id":"sales","text":"Sales enquiries"}]}}
|
| 2 |
-
{"id":"evidence","state_text":"The package was delivered on Monday. A signed receipt is available.","question":{"id":"evidence-q","type":"Noul","text":"Does the evidence confirm that the package was delivered?"}}
|
| 3 |
-
{"id":"sentiment","state_text":"The replacement arrived quickly and works perfectly.","question":{"id":"sentiment-q","type":"Score","text":"How positive is the customer's sentiment?","levels":[{"id":"negative","text":"Negative","value":0},{"id":"neutral","text":"Neutral","value":1},{"id":"positive","text":"Positive","value":2}]}}
|
|
|
|
|
|
|
|
|
|
|
|
examples/finetune.sh
DELETED
|
@@ -1,21 +0,0 @@
|
|
| 1 |
-
#!/usr/bin/env bash
|
| 2 |
-
set -euo pipefail
|
| 3 |
-
# Run from the package root in an existing compatible AMD ROCm environment.
|
| 4 |
-
# Replace these paths with your own isolated TRAIN and DEV JSONL files.
|
| 5 |
-
: "${TRAIN:?Set TRAIN to your training JSONL}"
|
| 6 |
-
: "${DEV:?Set DEV to your development JSONL}"
|
| 7 |
-
: "${OUTPUT:?Set OUTPUT to a fresh run directory}"
|
| 8 |
-
NATIVE="${NATIVE:-./native}"
|
| 9 |
-
MANIFEST_SHA256="${MANIFEST_SHA256:-f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6}"
|
| 10 |
-
export PYTHONPATH="${PWD}${PYTHONPATH:+:${PYTHONPATH}}"
|
| 11 |
-
export PYTHONDONTWRITEBYTECODE=1 HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 TOKENIZERS_PARALLELISM=false
|
| 12 |
-
python -m decision_finetune validate-jsonl --train "$TRAIN" --dev "$DEV" --selection macro-nll
|
| 13 |
-
# Device visibility is supplied by the caller, e.g. ROCR_VISIBLE_DEVICES=0.
|
| 14 |
-
python -m decision_finetune train \
|
| 15 |
-
--native "$NATIVE" --manifest-sha256 "$MANIFEST_SHA256" \
|
| 16 |
-
--train "$TRAIN" --dev "$DEV" --output "$OUTPUT" \
|
| 17 |
-
--epochs 4 --logical-batch-size 64 --micro-batch-size 8 --seed 20260921 \
|
| 18 |
-
--encoder-lr 2.5e-5 --head-lr 1e-4 --lr-min 1e-6 \
|
| 19 |
-
--weight-decay 0.01 --clip-norm 1 --score-rps-weight 0.1 \
|
| 20 |
-
--warmup-ratio 0.1 --selection macro-nll --selection-tolerance 1e-8 \
|
| 21 |
-
--cpu-threads 2 --max-reserved-gib 64
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
examples/system-one.json
DELETED
|
@@ -1,29 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"model": "Decision-1.0-Lex",
|
| 3 |
-
"state": {
|
| 4 |
-
"message": "Please refund the duplicate charge. I need this fixed today."
|
| 5 |
-
},
|
| 6 |
-
"questions": {
|
| 7 |
-
"refund_requested": {
|
| 8 |
-
"type": "noul",
|
| 9 |
-
"instructions": "Does the customer explicitly request a refund?"
|
| 10 |
-
},
|
| 11 |
-
"team": {
|
| 12 |
-
"type": "choice",
|
| 13 |
-
"instructions": "Which team should handle this request?",
|
| 14 |
-
"criteria": {
|
| 15 |
-
"Billing": "Charges and refunds",
|
| 16 |
-
"Support": "Technical problems"
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
"urgency": {
|
| 20 |
-
"type": "score",
|
| 21 |
-
"instructions": "How urgent is the request?",
|
| 22 |
-
"criteria": [
|
| 23 |
-
"No deadline",
|
| 24 |
-
"Needed soon",
|
| 25 |
-
"Needed today"
|
| 26 |
-
]
|
| 27 |
-
}
|
| 28 |
-
}
|
| 29 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
infer.py
DELETED
|
@@ -1,46 +0,0 @@
|
|
| 1 |
-
"""Complete-input 1K inference using the bundled, unchanged native runtime."""
|
| 2 |
-
import argparse
|
| 3 |
-
import json
|
| 4 |
-
from pathlib import Path
|
| 5 |
-
|
| 6 |
-
NATIVE_MANIFEST_SHA256 = 'f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6'
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
def main():
|
| 10 |
-
p = argparse.ArgumentParser(description='Decision Lex: complete1024, FP32 AMD inference')
|
| 11 |
-
p.add_argument('--native', default=str(Path(__file__).resolve().parent / 'native'))
|
| 12 |
-
p.add_argument('--manifest-sha256', default=NATIVE_MANIFEST_SHA256)
|
| 13 |
-
p.add_argument('--input', required=True, help='JSONL with id, state_text, question')
|
| 14 |
-
p.add_argument('--output', required=True, help='Fresh prediction JSONL; existing files are not overwritten')
|
| 15 |
-
p.add_argument('--batch-size', type=int, choices=range(1, 9), default=8)
|
| 16 |
-
args = p.parse_args()
|
| 17 |
-
records = []
|
| 18 |
-
with Path(args.input).open(encoding='utf-8') as stream:
|
| 19 |
-
for line in stream:
|
| 20 |
-
row = json.loads(line)
|
| 21 |
-
if set(row) != {'id', 'state_text', 'question'}:
|
| 22 |
-
raise ValueError('Inference rows must contain exactly id, state_text and question')
|
| 23 |
-
records.append(row)
|
| 24 |
-
if not records or len({r['id'] for r in records}) != len(records):
|
| 25 |
-
raise ValueError('Nonempty input with unique record IDs required')
|
| 26 |
-
if Path(args.output).exists():
|
| 27 |
-
raise ValueError('Prediction output must be fresh')
|
| 28 |
-
import torch
|
| 29 |
-
from decision_runtime import load_native
|
| 30 |
-
from decision_inference import predict_1k
|
| 31 |
-
if torch.version.hip is None or not torch.cuda.is_available() or torch.cuda.device_count() != 1:
|
| 32 |
-
raise RuntimeError('Expose exactly one AMD ROCm GPU; this package has no CPU/NVIDIA fallback')
|
| 33 |
-
torch.cuda.set_device(0); torch.set_num_threads(2)
|
| 34 |
-
torch.backends.cuda.matmul.allow_tf32 = False
|
| 35 |
-
torch.backends.cudnn.allow_tf32 = False
|
| 36 |
-
torch.backends.mha.set_fastpath_enabled(False)
|
| 37 |
-
native = load_native(args.native, expected_manifest_sha256=args.manifest_sha256, device='cuda:0')
|
| 38 |
-
answers = predict_1k(native, records, batch_size=args.batch_size)
|
| 39 |
-
if len(answers) != len(records):
|
| 40 |
-
raise RuntimeError('Incomplete inference result')
|
| 41 |
-
with Path(args.output).open('x', encoding='utf-8') as stream:
|
| 42 |
-
for answer in answers:
|
| 43 |
-
stream.write(json.dumps(answer, ensure_ascii=False, allow_nan=False) + '\n')
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
if __name__ == '__main__': main()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
native/INVENTORY.json
DELETED
|
@@ -1,1786 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"choice_suffix_shapes": {
|
| 3 |
-
"choice_blocks.0.attn.Wo.weight": [
|
| 4 |
-
768,
|
| 5 |
-
768
|
| 6 |
-
],
|
| 7 |
-
"choice_blocks.0.attn.Wqkv.weight": [
|
| 8 |
-
2304,
|
| 9 |
-
768
|
| 10 |
-
],
|
| 11 |
-
"choice_blocks.0.mlp.Wi.weight": [
|
| 12 |
-
2304,
|
| 13 |
-
768
|
| 14 |
-
],
|
| 15 |
-
"choice_blocks.0.mlp.Wo.weight": [
|
| 16 |
-
768,
|
| 17 |
-
1152
|
| 18 |
-
],
|
| 19 |
-
"choice_blocks.0.mlp_norm.weight": [
|
| 20 |
-
768
|
| 21 |
-
],
|
| 22 |
-
"choice_blocks.1.attn.Wo.weight": [
|
| 23 |
-
768,
|
| 24 |
-
768
|
| 25 |
-
],
|
| 26 |
-
"choice_blocks.1.attn.Wqkv.weight": [
|
| 27 |
-
2304,
|
| 28 |
-
768
|
| 29 |
-
],
|
| 30 |
-
"choice_blocks.1.attn_norm.weight": [
|
| 31 |
-
768
|
| 32 |
-
],
|
| 33 |
-
"choice_blocks.1.mlp.Wi.weight": [
|
| 34 |
-
2304,
|
| 35 |
-
768
|
| 36 |
-
],
|
| 37 |
-
"choice_blocks.1.mlp.Wo.weight": [
|
| 38 |
-
768,
|
| 39 |
-
1152
|
| 40 |
-
],
|
| 41 |
-
"choice_blocks.1.mlp_norm.weight": [
|
| 42 |
-
768
|
| 43 |
-
],
|
| 44 |
-
"choice_blocks.10.attn.Wo.weight": [
|
| 45 |
-
768,
|
| 46 |
-
768
|
| 47 |
-
],
|
| 48 |
-
"choice_blocks.10.attn.Wqkv.weight": [
|
| 49 |
-
2304,
|
| 50 |
-
768
|
| 51 |
-
],
|
| 52 |
-
"choice_blocks.10.attn_norm.weight": [
|
| 53 |
-
768
|
| 54 |
-
],
|
| 55 |
-
"choice_blocks.10.mlp.Wi.weight": [
|
| 56 |
-
2304,
|
| 57 |
-
768
|
| 58 |
-
],
|
| 59 |
-
"choice_blocks.10.mlp.Wo.weight": [
|
| 60 |
-
768,
|
| 61 |
-
1152
|
| 62 |
-
],
|
| 63 |
-
"choice_blocks.10.mlp_norm.weight": [
|
| 64 |
-
768
|
| 65 |
-
],
|
| 66 |
-
"choice_blocks.11.attn.Wo.weight": [
|
| 67 |
-
768,
|
| 68 |
-
768
|
| 69 |
-
],
|
| 70 |
-
"choice_blocks.11.attn.Wqkv.weight": [
|
| 71 |
-
2304,
|
| 72 |
-
768
|
| 73 |
-
],
|
| 74 |
-
"choice_blocks.11.attn_norm.weight": [
|
| 75 |
-
768
|
| 76 |
-
],
|
| 77 |
-
"choice_blocks.11.mlp.Wi.weight": [
|
| 78 |
-
2304,
|
| 79 |
-
768
|
| 80 |
-
],
|
| 81 |
-
"choice_blocks.11.mlp.Wo.weight": [
|
| 82 |
-
768,
|
| 83 |
-
1152
|
| 84 |
-
],
|
| 85 |
-
"choice_blocks.11.mlp_norm.weight": [
|
| 86 |
-
768
|
| 87 |
-
],
|
| 88 |
-
"choice_blocks.12.attn.Wo.weight": [
|
| 89 |
-
768,
|
| 90 |
-
768
|
| 91 |
-
],
|
| 92 |
-
"choice_blocks.12.attn.Wqkv.weight": [
|
| 93 |
-
2304,
|
| 94 |
-
768
|
| 95 |
-
],
|
| 96 |
-
"choice_blocks.12.attn_norm.weight": [
|
| 97 |
-
768
|
| 98 |
-
],
|
| 99 |
-
"choice_blocks.12.mlp.Wi.weight": [
|
| 100 |
-
2304,
|
| 101 |
-
768
|
| 102 |
-
],
|
| 103 |
-
"choice_blocks.12.mlp.Wo.weight": [
|
| 104 |
-
768,
|
| 105 |
-
1152
|
| 106 |
-
],
|
| 107 |
-
"choice_blocks.12.mlp_norm.weight": [
|
| 108 |
-
768
|
| 109 |
-
],
|
| 110 |
-
"choice_blocks.13.attn.Wo.weight": [
|
| 111 |
-
768,
|
| 112 |
-
768
|
| 113 |
-
],
|
| 114 |
-
"choice_blocks.13.attn.Wqkv.weight": [
|
| 115 |
-
2304,
|
| 116 |
-
768
|
| 117 |
-
],
|
| 118 |
-
"choice_blocks.13.attn_norm.weight": [
|
| 119 |
-
768
|
| 120 |
-
],
|
| 121 |
-
"choice_blocks.13.mlp.Wi.weight": [
|
| 122 |
-
2304,
|
| 123 |
-
768
|
| 124 |
-
],
|
| 125 |
-
"choice_blocks.13.mlp.Wo.weight": [
|
| 126 |
-
768,
|
| 127 |
-
1152
|
| 128 |
-
],
|
| 129 |
-
"choice_blocks.13.mlp_norm.weight": [
|
| 130 |
-
768
|
| 131 |
-
],
|
| 132 |
-
"choice_blocks.14.attn.Wo.weight": [
|
| 133 |
-
768,
|
| 134 |
-
768
|
| 135 |
-
],
|
| 136 |
-
"choice_blocks.14.attn.Wqkv.weight": [
|
| 137 |
-
2304,
|
| 138 |
-
768
|
| 139 |
-
],
|
| 140 |
-
"choice_blocks.14.attn_norm.weight": [
|
| 141 |
-
768
|
| 142 |
-
],
|
| 143 |
-
"choice_blocks.14.mlp.Wi.weight": [
|
| 144 |
-
2304,
|
| 145 |
-
768
|
| 146 |
-
],
|
| 147 |
-
"choice_blocks.14.mlp.Wo.weight": [
|
| 148 |
-
768,
|
| 149 |
-
1152
|
| 150 |
-
],
|
| 151 |
-
"choice_blocks.14.mlp_norm.weight": [
|
| 152 |
-
768
|
| 153 |
-
],
|
| 154 |
-
"choice_blocks.15.attn.Wo.weight": [
|
| 155 |
-
768,
|
| 156 |
-
768
|
| 157 |
-
],
|
| 158 |
-
"choice_blocks.15.attn.Wqkv.weight": [
|
| 159 |
-
2304,
|
| 160 |
-
768
|
| 161 |
-
],
|
| 162 |
-
"choice_blocks.15.attn_norm.weight": [
|
| 163 |
-
768
|
| 164 |
-
],
|
| 165 |
-
"choice_blocks.15.mlp.Wi.weight": [
|
| 166 |
-
2304,
|
| 167 |
-
768
|
| 168 |
-
],
|
| 169 |
-
"choice_blocks.15.mlp.Wo.weight": [
|
| 170 |
-
768,
|
| 171 |
-
1152
|
| 172 |
-
],
|
| 173 |
-
"choice_blocks.15.mlp_norm.weight": [
|
| 174 |
-
768
|
| 175 |
-
],
|
| 176 |
-
"choice_blocks.16.attn.Wo.weight": [
|
| 177 |
-
768,
|
| 178 |
-
768
|
| 179 |
-
],
|
| 180 |
-
"choice_blocks.16.attn.Wqkv.weight": [
|
| 181 |
-
2304,
|
| 182 |
-
768
|
| 183 |
-
],
|
| 184 |
-
"choice_blocks.16.attn_norm.weight": [
|
| 185 |
-
768
|
| 186 |
-
],
|
| 187 |
-
"choice_blocks.16.mlp.Wi.weight": [
|
| 188 |
-
2304,
|
| 189 |
-
768
|
| 190 |
-
],
|
| 191 |
-
"choice_blocks.16.mlp.Wo.weight": [
|
| 192 |
-
768,
|
| 193 |
-
1152
|
| 194 |
-
],
|
| 195 |
-
"choice_blocks.16.mlp_norm.weight": [
|
| 196 |
-
768
|
| 197 |
-
],
|
| 198 |
-
"choice_blocks.17.attn.Wo.weight": [
|
| 199 |
-
768,
|
| 200 |
-
768
|
| 201 |
-
],
|
| 202 |
-
"choice_blocks.17.attn.Wqkv.weight": [
|
| 203 |
-
2304,
|
| 204 |
-
768
|
| 205 |
-
],
|
| 206 |
-
"choice_blocks.17.attn_norm.weight": [
|
| 207 |
-
768
|
| 208 |
-
],
|
| 209 |
-
"choice_blocks.17.mlp.Wi.weight": [
|
| 210 |
-
2304,
|
| 211 |
-
768
|
| 212 |
-
],
|
| 213 |
-
"choice_blocks.17.mlp.Wo.weight": [
|
| 214 |
-
768,
|
| 215 |
-
1152
|
| 216 |
-
],
|
| 217 |
-
"choice_blocks.17.mlp_norm.weight": [
|
| 218 |
-
768
|
| 219 |
-
],
|
| 220 |
-
"choice_blocks.18.attn.Wo.weight": [
|
| 221 |
-
768,
|
| 222 |
-
768
|
| 223 |
-
],
|
| 224 |
-
"choice_blocks.18.attn.Wqkv.weight": [
|
| 225 |
-
2304,
|
| 226 |
-
768
|
| 227 |
-
],
|
| 228 |
-
"choice_blocks.18.attn_norm.weight": [
|
| 229 |
-
768
|
| 230 |
-
],
|
| 231 |
-
"choice_blocks.18.mlp.Wi.weight": [
|
| 232 |
-
2304,
|
| 233 |
-
768
|
| 234 |
-
],
|
| 235 |
-
"choice_blocks.18.mlp.Wo.weight": [
|
| 236 |
-
768,
|
| 237 |
-
1152
|
| 238 |
-
],
|
| 239 |
-
"choice_blocks.18.mlp_norm.weight": [
|
| 240 |
-
768
|
| 241 |
-
],
|
| 242 |
-
"choice_blocks.19.attn.Wo.weight": [
|
| 243 |
-
768,
|
| 244 |
-
768
|
| 245 |
-
],
|
| 246 |
-
"choice_blocks.19.attn.Wqkv.weight": [
|
| 247 |
-
2304,
|
| 248 |
-
768
|
| 249 |
-
],
|
| 250 |
-
"choice_blocks.19.attn_norm.weight": [
|
| 251 |
-
768
|
| 252 |
-
],
|
| 253 |
-
"choice_blocks.19.mlp.Wi.weight": [
|
| 254 |
-
2304,
|
| 255 |
-
768
|
| 256 |
-
],
|
| 257 |
-
"choice_blocks.19.mlp.Wo.weight": [
|
| 258 |
-
768,
|
| 259 |
-
1152
|
| 260 |
-
],
|
| 261 |
-
"choice_blocks.19.mlp_norm.weight": [
|
| 262 |
-
768
|
| 263 |
-
],
|
| 264 |
-
"choice_blocks.2.attn.Wo.weight": [
|
| 265 |
-
768,
|
| 266 |
-
768
|
| 267 |
-
],
|
| 268 |
-
"choice_blocks.2.attn.Wqkv.weight": [
|
| 269 |
-
2304,
|
| 270 |
-
768
|
| 271 |
-
],
|
| 272 |
-
"choice_blocks.2.attn_norm.weight": [
|
| 273 |
-
768
|
| 274 |
-
],
|
| 275 |
-
"choice_blocks.2.mlp.Wi.weight": [
|
| 276 |
-
2304,
|
| 277 |
-
768
|
| 278 |
-
],
|
| 279 |
-
"choice_blocks.2.mlp.Wo.weight": [
|
| 280 |
-
768,
|
| 281 |
-
1152
|
| 282 |
-
],
|
| 283 |
-
"choice_blocks.2.mlp_norm.weight": [
|
| 284 |
-
768
|
| 285 |
-
],
|
| 286 |
-
"choice_blocks.20.attn.Wo.weight": [
|
| 287 |
-
768,
|
| 288 |
-
768
|
| 289 |
-
],
|
| 290 |
-
"choice_blocks.20.attn.Wqkv.weight": [
|
| 291 |
-
2304,
|
| 292 |
-
768
|
| 293 |
-
],
|
| 294 |
-
"choice_blocks.20.attn_norm.weight": [
|
| 295 |
-
768
|
| 296 |
-
],
|
| 297 |
-
"choice_blocks.20.mlp.Wi.weight": [
|
| 298 |
-
2304,
|
| 299 |
-
768
|
| 300 |
-
],
|
| 301 |
-
"choice_blocks.20.mlp.Wo.weight": [
|
| 302 |
-
768,
|
| 303 |
-
1152
|
| 304 |
-
],
|
| 305 |
-
"choice_blocks.20.mlp_norm.weight": [
|
| 306 |
-
768
|
| 307 |
-
],
|
| 308 |
-
"choice_blocks.21.attn.Wo.weight": [
|
| 309 |
-
768,
|
| 310 |
-
768
|
| 311 |
-
],
|
| 312 |
-
"choice_blocks.21.attn.Wqkv.weight": [
|
| 313 |
-
2304,
|
| 314 |
-
768
|
| 315 |
-
],
|
| 316 |
-
"choice_blocks.21.attn_norm.weight": [
|
| 317 |
-
768
|
| 318 |
-
],
|
| 319 |
-
"choice_blocks.21.mlp.Wi.weight": [
|
| 320 |
-
2304,
|
| 321 |
-
768
|
| 322 |
-
],
|
| 323 |
-
"choice_blocks.21.mlp.Wo.weight": [
|
| 324 |
-
768,
|
| 325 |
-
1152
|
| 326 |
-
],
|
| 327 |
-
"choice_blocks.21.mlp_norm.weight": [
|
| 328 |
-
768
|
| 329 |
-
],
|
| 330 |
-
"choice_blocks.3.attn.Wo.weight": [
|
| 331 |
-
768,
|
| 332 |
-
768
|
| 333 |
-
],
|
| 334 |
-
"choice_blocks.3.attn.Wqkv.weight": [
|
| 335 |
-
2304,
|
| 336 |
-
768
|
| 337 |
-
],
|
| 338 |
-
"choice_blocks.3.attn_norm.weight": [
|
| 339 |
-
768
|
| 340 |
-
],
|
| 341 |
-
"choice_blocks.3.mlp.Wi.weight": [
|
| 342 |
-
2304,
|
| 343 |
-
768
|
| 344 |
-
],
|
| 345 |
-
"choice_blocks.3.mlp.Wo.weight": [
|
| 346 |
-
768,
|
| 347 |
-
1152
|
| 348 |
-
],
|
| 349 |
-
"choice_blocks.3.mlp_norm.weight": [
|
| 350 |
-
768
|
| 351 |
-
],
|
| 352 |
-
"choice_blocks.4.attn.Wo.weight": [
|
| 353 |
-
768,
|
| 354 |
-
768
|
| 355 |
-
],
|
| 356 |
-
"choice_blocks.4.attn.Wqkv.weight": [
|
| 357 |
-
2304,
|
| 358 |
-
768
|
| 359 |
-
],
|
| 360 |
-
"choice_blocks.4.attn_norm.weight": [
|
| 361 |
-
768
|
| 362 |
-
],
|
| 363 |
-
"choice_blocks.4.mlp.Wi.weight": [
|
| 364 |
-
2304,
|
| 365 |
-
768
|
| 366 |
-
],
|
| 367 |
-
"choice_blocks.4.mlp.Wo.weight": [
|
| 368 |
-
768,
|
| 369 |
-
1152
|
| 370 |
-
],
|
| 371 |
-
"choice_blocks.4.mlp_norm.weight": [
|
| 372 |
-
768
|
| 373 |
-
],
|
| 374 |
-
"choice_blocks.5.attn.Wo.weight": [
|
| 375 |
-
768,
|
| 376 |
-
768
|
| 377 |
-
],
|
| 378 |
-
"choice_blocks.5.attn.Wqkv.weight": [
|
| 379 |
-
2304,
|
| 380 |
-
768
|
| 381 |
-
],
|
| 382 |
-
"choice_blocks.5.attn_norm.weight": [
|
| 383 |
-
768
|
| 384 |
-
],
|
| 385 |
-
"choice_blocks.5.mlp.Wi.weight": [
|
| 386 |
-
2304,
|
| 387 |
-
768
|
| 388 |
-
],
|
| 389 |
-
"choice_blocks.5.mlp.Wo.weight": [
|
| 390 |
-
768,
|
| 391 |
-
1152
|
| 392 |
-
],
|
| 393 |
-
"choice_blocks.5.mlp_norm.weight": [
|
| 394 |
-
768
|
| 395 |
-
],
|
| 396 |
-
"choice_blocks.6.attn.Wo.weight": [
|
| 397 |
-
768,
|
| 398 |
-
768
|
| 399 |
-
],
|
| 400 |
-
"choice_blocks.6.attn.Wqkv.weight": [
|
| 401 |
-
2304,
|
| 402 |
-
768
|
| 403 |
-
],
|
| 404 |
-
"choice_blocks.6.attn_norm.weight": [
|
| 405 |
-
768
|
| 406 |
-
],
|
| 407 |
-
"choice_blocks.6.mlp.Wi.weight": [
|
| 408 |
-
2304,
|
| 409 |
-
768
|
| 410 |
-
],
|
| 411 |
-
"choice_blocks.6.mlp.Wo.weight": [
|
| 412 |
-
768,
|
| 413 |
-
1152
|
| 414 |
-
],
|
| 415 |
-
"choice_blocks.6.mlp_norm.weight": [
|
| 416 |
-
768
|
| 417 |
-
],
|
| 418 |
-
"choice_blocks.7.attn.Wo.weight": [
|
| 419 |
-
768,
|
| 420 |
-
768
|
| 421 |
-
],
|
| 422 |
-
"choice_blocks.7.attn.Wqkv.weight": [
|
| 423 |
-
2304,
|
| 424 |
-
768
|
| 425 |
-
],
|
| 426 |
-
"choice_blocks.7.attn_norm.weight": [
|
| 427 |
-
768
|
| 428 |
-
],
|
| 429 |
-
"choice_blocks.7.mlp.Wi.weight": [
|
| 430 |
-
2304,
|
| 431 |
-
768
|
| 432 |
-
],
|
| 433 |
-
"choice_blocks.7.mlp.Wo.weight": [
|
| 434 |
-
768,
|
| 435 |
-
1152
|
| 436 |
-
],
|
| 437 |
-
"choice_blocks.7.mlp_norm.weight": [
|
| 438 |
-
768
|
| 439 |
-
],
|
| 440 |
-
"choice_blocks.8.attn.Wo.weight": [
|
| 441 |
-
768,
|
| 442 |
-
768
|
| 443 |
-
],
|
| 444 |
-
"choice_blocks.8.attn.Wqkv.weight": [
|
| 445 |
-
2304,
|
| 446 |
-
768
|
| 447 |
-
],
|
| 448 |
-
"choice_blocks.8.attn_norm.weight": [
|
| 449 |
-
768
|
| 450 |
-
],
|
| 451 |
-
"choice_blocks.8.mlp.Wi.weight": [
|
| 452 |
-
2304,
|
| 453 |
-
768
|
| 454 |
-
],
|
| 455 |
-
"choice_blocks.8.mlp.Wo.weight": [
|
| 456 |
-
768,
|
| 457 |
-
1152
|
| 458 |
-
],
|
| 459 |
-
"choice_blocks.8.mlp_norm.weight": [
|
| 460 |
-
768
|
| 461 |
-
],
|
| 462 |
-
"choice_blocks.9.attn.Wo.weight": [
|
| 463 |
-
768,
|
| 464 |
-
768
|
| 465 |
-
],
|
| 466 |
-
"choice_blocks.9.attn.Wqkv.weight": [
|
| 467 |
-
2304,
|
| 468 |
-
768
|
| 469 |
-
],
|
| 470 |
-
"choice_blocks.9.attn_norm.weight": [
|
| 471 |
-
768
|
| 472 |
-
],
|
| 473 |
-
"choice_blocks.9.mlp.Wi.weight": [
|
| 474 |
-
2304,
|
| 475 |
-
768
|
| 476 |
-
],
|
| 477 |
-
"choice_blocks.9.mlp.Wo.weight": [
|
| 478 |
-
768,
|
| 479 |
-
1152
|
| 480 |
-
],
|
| 481 |
-
"choice_blocks.9.mlp_norm.weight": [
|
| 482 |
-
768
|
| 483 |
-
],
|
| 484 |
-
"choice_final_norm.weight": [
|
| 485 |
-
768
|
| 486 |
-
]
|
| 487 |
-
},
|
| 488 |
-
"counts": {
|
| 489 |
-
"frozen_parameters": 446810114,
|
| 490 |
-
"frozen_tensors": 327,
|
| 491 |
-
"integrated_choice_suffix_tensors": 132,
|
| 492 |
-
"score_suffix_parameters": 110330880,
|
| 493 |
-
"score_suffix_tensors": 132,
|
| 494 |
-
"shared_encoder_tensors": 134,
|
| 495 |
-
"trainable_parameters": 125099521,
|
| 496 |
-
"trainable_tensors": 162,
|
| 497 |
-
"type_and_head_tensors": 91,
|
| 498 |
-
"unique_parameters": 571909635,
|
| 499 |
-
"unique_tensors": 489
|
| 500 |
-
},
|
| 501 |
-
"encoder_shapes": {
|
| 502 |
-
"embeddings.norm.weight": [
|
| 503 |
-
768
|
| 504 |
-
],
|
| 505 |
-
"embeddings.tok_embeddings.weight": [
|
| 506 |
-
256000,
|
| 507 |
-
768
|
| 508 |
-
],
|
| 509 |
-
"final_norm.weight": [
|
| 510 |
-
768
|
| 511 |
-
],
|
| 512 |
-
"layers.0.attn.Wo.weight": [
|
| 513 |
-
768,
|
| 514 |
-
768
|
| 515 |
-
],
|
| 516 |
-
"layers.0.attn.Wqkv.weight": [
|
| 517 |
-
2304,
|
| 518 |
-
768
|
| 519 |
-
],
|
| 520 |
-
"layers.0.mlp.Wi.weight": [
|
| 521 |
-
2304,
|
| 522 |
-
768
|
| 523 |
-
],
|
| 524 |
-
"layers.0.mlp.Wo.weight": [
|
| 525 |
-
768,
|
| 526 |
-
1152
|
| 527 |
-
],
|
| 528 |
-
"layers.0.mlp_norm.weight": [
|
| 529 |
-
768
|
| 530 |
-
],
|
| 531 |
-
"layers.1.attn.Wo.weight": [
|
| 532 |
-
768,
|
| 533 |
-
768
|
| 534 |
-
],
|
| 535 |
-
"layers.1.attn.Wqkv.weight": [
|
| 536 |
-
2304,
|
| 537 |
-
768
|
| 538 |
-
],
|
| 539 |
-
"layers.1.attn_norm.weight": [
|
| 540 |
-
768
|
| 541 |
-
],
|
| 542 |
-
"layers.1.mlp.Wi.weight": [
|
| 543 |
-
2304,
|
| 544 |
-
768
|
| 545 |
-
],
|
| 546 |
-
"layers.1.mlp.Wo.weight": [
|
| 547 |
-
768,
|
| 548 |
-
1152
|
| 549 |
-
],
|
| 550 |
-
"layers.1.mlp_norm.weight": [
|
| 551 |
-
768
|
| 552 |
-
],
|
| 553 |
-
"layers.10.attn.Wo.weight": [
|
| 554 |
-
768,
|
| 555 |
-
768
|
| 556 |
-
],
|
| 557 |
-
"layers.10.attn.Wqkv.weight": [
|
| 558 |
-
2304,
|
| 559 |
-
768
|
| 560 |
-
],
|
| 561 |
-
"layers.10.attn_norm.weight": [
|
| 562 |
-
768
|
| 563 |
-
],
|
| 564 |
-
"layers.10.mlp.Wi.weight": [
|
| 565 |
-
2304,
|
| 566 |
-
768
|
| 567 |
-
],
|
| 568 |
-
"layers.10.mlp.Wo.weight": [
|
| 569 |
-
768,
|
| 570 |
-
1152
|
| 571 |
-
],
|
| 572 |
-
"layers.10.mlp_norm.weight": [
|
| 573 |
-
768
|
| 574 |
-
],
|
| 575 |
-
"layers.11.attn.Wo.weight": [
|
| 576 |
-
768,
|
| 577 |
-
768
|
| 578 |
-
],
|
| 579 |
-
"layers.11.attn.Wqkv.weight": [
|
| 580 |
-
2304,
|
| 581 |
-
768
|
| 582 |
-
],
|
| 583 |
-
"layers.11.attn_norm.weight": [
|
| 584 |
-
768
|
| 585 |
-
],
|
| 586 |
-
"layers.11.mlp.Wi.weight": [
|
| 587 |
-
2304,
|
| 588 |
-
768
|
| 589 |
-
],
|
| 590 |
-
"layers.11.mlp.Wo.weight": [
|
| 591 |
-
768,
|
| 592 |
-
1152
|
| 593 |
-
],
|
| 594 |
-
"layers.11.mlp_norm.weight": [
|
| 595 |
-
768
|
| 596 |
-
],
|
| 597 |
-
"layers.12.attn.Wo.weight": [
|
| 598 |
-
768,
|
| 599 |
-
768
|
| 600 |
-
],
|
| 601 |
-
"layers.12.attn.Wqkv.weight": [
|
| 602 |
-
2304,
|
| 603 |
-
768
|
| 604 |
-
],
|
| 605 |
-
"layers.12.attn_norm.weight": [
|
| 606 |
-
768
|
| 607 |
-
],
|
| 608 |
-
"layers.12.mlp.Wi.weight": [
|
| 609 |
-
2304,
|
| 610 |
-
768
|
| 611 |
-
],
|
| 612 |
-
"layers.12.mlp.Wo.weight": [
|
| 613 |
-
768,
|
| 614 |
-
1152
|
| 615 |
-
],
|
| 616 |
-
"layers.12.mlp_norm.weight": [
|
| 617 |
-
768
|
| 618 |
-
],
|
| 619 |
-
"layers.13.attn.Wo.weight": [
|
| 620 |
-
768,
|
| 621 |
-
768
|
| 622 |
-
],
|
| 623 |
-
"layers.13.attn.Wqkv.weight": [
|
| 624 |
-
2304,
|
| 625 |
-
768
|
| 626 |
-
],
|
| 627 |
-
"layers.13.attn_norm.weight": [
|
| 628 |
-
768
|
| 629 |
-
],
|
| 630 |
-
"layers.13.mlp.Wi.weight": [
|
| 631 |
-
2304,
|
| 632 |
-
768
|
| 633 |
-
],
|
| 634 |
-
"layers.13.mlp.Wo.weight": [
|
| 635 |
-
768,
|
| 636 |
-
1152
|
| 637 |
-
],
|
| 638 |
-
"layers.13.mlp_norm.weight": [
|
| 639 |
-
768
|
| 640 |
-
],
|
| 641 |
-
"layers.14.attn.Wo.weight": [
|
| 642 |
-
768,
|
| 643 |
-
768
|
| 644 |
-
],
|
| 645 |
-
"layers.14.attn.Wqkv.weight": [
|
| 646 |
-
2304,
|
| 647 |
-
768
|
| 648 |
-
],
|
| 649 |
-
"layers.14.attn_norm.weight": [
|
| 650 |
-
768
|
| 651 |
-
],
|
| 652 |
-
"layers.14.mlp.Wi.weight": [
|
| 653 |
-
2304,
|
| 654 |
-
768
|
| 655 |
-
],
|
| 656 |
-
"layers.14.mlp.Wo.weight": [
|
| 657 |
-
768,
|
| 658 |
-
1152
|
| 659 |
-
],
|
| 660 |
-
"layers.14.mlp_norm.weight": [
|
| 661 |
-
768
|
| 662 |
-
],
|
| 663 |
-
"layers.15.attn.Wo.weight": [
|
| 664 |
-
768,
|
| 665 |
-
768
|
| 666 |
-
],
|
| 667 |
-
"layers.15.attn.Wqkv.weight": [
|
| 668 |
-
2304,
|
| 669 |
-
768
|
| 670 |
-
],
|
| 671 |
-
"layers.15.attn_norm.weight": [
|
| 672 |
-
768
|
| 673 |
-
],
|
| 674 |
-
"layers.15.mlp.Wi.weight": [
|
| 675 |
-
2304,
|
| 676 |
-
768
|
| 677 |
-
],
|
| 678 |
-
"layers.15.mlp.Wo.weight": [
|
| 679 |
-
768,
|
| 680 |
-
1152
|
| 681 |
-
],
|
| 682 |
-
"layers.15.mlp_norm.weight": [
|
| 683 |
-
768
|
| 684 |
-
],
|
| 685 |
-
"layers.16.attn.Wo.weight": [
|
| 686 |
-
768,
|
| 687 |
-
768
|
| 688 |
-
],
|
| 689 |
-
"layers.16.attn.Wqkv.weight": [
|
| 690 |
-
2304,
|
| 691 |
-
768
|
| 692 |
-
],
|
| 693 |
-
"layers.16.attn_norm.weight": [
|
| 694 |
-
768
|
| 695 |
-
],
|
| 696 |
-
"layers.16.mlp.Wi.weight": [
|
| 697 |
-
2304,
|
| 698 |
-
768
|
| 699 |
-
],
|
| 700 |
-
"layers.16.mlp.Wo.weight": [
|
| 701 |
-
768,
|
| 702 |
-
1152
|
| 703 |
-
],
|
| 704 |
-
"layers.16.mlp_norm.weight": [
|
| 705 |
-
768
|
| 706 |
-
],
|
| 707 |
-
"layers.17.attn.Wo.weight": [
|
| 708 |
-
768,
|
| 709 |
-
768
|
| 710 |
-
],
|
| 711 |
-
"layers.17.attn.Wqkv.weight": [
|
| 712 |
-
2304,
|
| 713 |
-
768
|
| 714 |
-
],
|
| 715 |
-
"layers.17.attn_norm.weight": [
|
| 716 |
-
768
|
| 717 |
-
],
|
| 718 |
-
"layers.17.mlp.Wi.weight": [
|
| 719 |
-
2304,
|
| 720 |
-
768
|
| 721 |
-
],
|
| 722 |
-
"layers.17.mlp.Wo.weight": [
|
| 723 |
-
768,
|
| 724 |
-
1152
|
| 725 |
-
],
|
| 726 |
-
"layers.17.mlp_norm.weight": [
|
| 727 |
-
768
|
| 728 |
-
],
|
| 729 |
-
"layers.18.attn.Wo.weight": [
|
| 730 |
-
768,
|
| 731 |
-
768
|
| 732 |
-
],
|
| 733 |
-
"layers.18.attn.Wqkv.weight": [
|
| 734 |
-
2304,
|
| 735 |
-
768
|
| 736 |
-
],
|
| 737 |
-
"layers.18.attn_norm.weight": [
|
| 738 |
-
768
|
| 739 |
-
],
|
| 740 |
-
"layers.18.mlp.Wi.weight": [
|
| 741 |
-
2304,
|
| 742 |
-
768
|
| 743 |
-
],
|
| 744 |
-
"layers.18.mlp.Wo.weight": [
|
| 745 |
-
768,
|
| 746 |
-
1152
|
| 747 |
-
],
|
| 748 |
-
"layers.18.mlp_norm.weight": [
|
| 749 |
-
768
|
| 750 |
-
],
|
| 751 |
-
"layers.19.attn.Wo.weight": [
|
| 752 |
-
768,
|
| 753 |
-
768
|
| 754 |
-
],
|
| 755 |
-
"layers.19.attn.Wqkv.weight": [
|
| 756 |
-
2304,
|
| 757 |
-
768
|
| 758 |
-
],
|
| 759 |
-
"layers.19.attn_norm.weight": [
|
| 760 |
-
768
|
| 761 |
-
],
|
| 762 |
-
"layers.19.mlp.Wi.weight": [
|
| 763 |
-
2304,
|
| 764 |
-
768
|
| 765 |
-
],
|
| 766 |
-
"layers.19.mlp.Wo.weight": [
|
| 767 |
-
768,
|
| 768 |
-
1152
|
| 769 |
-
],
|
| 770 |
-
"layers.19.mlp_norm.weight": [
|
| 771 |
-
768
|
| 772 |
-
],
|
| 773 |
-
"layers.2.attn.Wo.weight": [
|
| 774 |
-
768,
|
| 775 |
-
768
|
| 776 |
-
],
|
| 777 |
-
"layers.2.attn.Wqkv.weight": [
|
| 778 |
-
2304,
|
| 779 |
-
768
|
| 780 |
-
],
|
| 781 |
-
"layers.2.attn_norm.weight": [
|
| 782 |
-
768
|
| 783 |
-
],
|
| 784 |
-
"layers.2.mlp.Wi.weight": [
|
| 785 |
-
2304,
|
| 786 |
-
768
|
| 787 |
-
],
|
| 788 |
-
"layers.2.mlp.Wo.weight": [
|
| 789 |
-
768,
|
| 790 |
-
1152
|
| 791 |
-
],
|
| 792 |
-
"layers.2.mlp_norm.weight": [
|
| 793 |
-
768
|
| 794 |
-
],
|
| 795 |
-
"layers.20.attn.Wo.weight": [
|
| 796 |
-
768,
|
| 797 |
-
768
|
| 798 |
-
],
|
| 799 |
-
"layers.20.attn.Wqkv.weight": [
|
| 800 |
-
2304,
|
| 801 |
-
768
|
| 802 |
-
],
|
| 803 |
-
"layers.20.attn_norm.weight": [
|
| 804 |
-
768
|
| 805 |
-
],
|
| 806 |
-
"layers.20.mlp.Wi.weight": [
|
| 807 |
-
2304,
|
| 808 |
-
768
|
| 809 |
-
],
|
| 810 |
-
"layers.20.mlp.Wo.weight": [
|
| 811 |
-
768,
|
| 812 |
-
1152
|
| 813 |
-
],
|
| 814 |
-
"layers.20.mlp_norm.weight": [
|
| 815 |
-
768
|
| 816 |
-
],
|
| 817 |
-
"layers.21.attn.Wo.weight": [
|
| 818 |
-
768,
|
| 819 |
-
768
|
| 820 |
-
],
|
| 821 |
-
"layers.21.attn.Wqkv.weight": [
|
| 822 |
-
2304,
|
| 823 |
-
768
|
| 824 |
-
],
|
| 825 |
-
"layers.21.attn_norm.weight": [
|
| 826 |
-
768
|
| 827 |
-
],
|
| 828 |
-
"layers.21.mlp.Wi.weight": [
|
| 829 |
-
2304,
|
| 830 |
-
768
|
| 831 |
-
],
|
| 832 |
-
"layers.21.mlp.Wo.weight": [
|
| 833 |
-
768,
|
| 834 |
-
1152
|
| 835 |
-
],
|
| 836 |
-
"layers.21.mlp_norm.weight": [
|
| 837 |
-
768
|
| 838 |
-
],
|
| 839 |
-
"layers.3.attn.Wo.weight": [
|
| 840 |
-
768,
|
| 841 |
-
768
|
| 842 |
-
],
|
| 843 |
-
"layers.3.attn.Wqkv.weight": [
|
| 844 |
-
2304,
|
| 845 |
-
768
|
| 846 |
-
],
|
| 847 |
-
"layers.3.attn_norm.weight": [
|
| 848 |
-
768
|
| 849 |
-
],
|
| 850 |
-
"layers.3.mlp.Wi.weight": [
|
| 851 |
-
2304,
|
| 852 |
-
768
|
| 853 |
-
],
|
| 854 |
-
"layers.3.mlp.Wo.weight": [
|
| 855 |
-
768,
|
| 856 |
-
1152
|
| 857 |
-
],
|
| 858 |
-
"layers.3.mlp_norm.weight": [
|
| 859 |
-
768
|
| 860 |
-
],
|
| 861 |
-
"layers.4.attn.Wo.weight": [
|
| 862 |
-
768,
|
| 863 |
-
768
|
| 864 |
-
],
|
| 865 |
-
"layers.4.attn.Wqkv.weight": [
|
| 866 |
-
2304,
|
| 867 |
-
768
|
| 868 |
-
],
|
| 869 |
-
"layers.4.attn_norm.weight": [
|
| 870 |
-
768
|
| 871 |
-
],
|
| 872 |
-
"layers.4.mlp.Wi.weight": [
|
| 873 |
-
2304,
|
| 874 |
-
768
|
| 875 |
-
],
|
| 876 |
-
"layers.4.mlp.Wo.weight": [
|
| 877 |
-
768,
|
| 878 |
-
1152
|
| 879 |
-
],
|
| 880 |
-
"layers.4.mlp_norm.weight": [
|
| 881 |
-
768
|
| 882 |
-
],
|
| 883 |
-
"layers.5.attn.Wo.weight": [
|
| 884 |
-
768,
|
| 885 |
-
768
|
| 886 |
-
],
|
| 887 |
-
"layers.5.attn.Wqkv.weight": [
|
| 888 |
-
2304,
|
| 889 |
-
768
|
| 890 |
-
],
|
| 891 |
-
"layers.5.attn_norm.weight": [
|
| 892 |
-
768
|
| 893 |
-
],
|
| 894 |
-
"layers.5.mlp.Wi.weight": [
|
| 895 |
-
2304,
|
| 896 |
-
768
|
| 897 |
-
],
|
| 898 |
-
"layers.5.mlp.Wo.weight": [
|
| 899 |
-
768,
|
| 900 |
-
1152
|
| 901 |
-
],
|
| 902 |
-
"layers.5.mlp_norm.weight": [
|
| 903 |
-
768
|
| 904 |
-
],
|
| 905 |
-
"layers.6.attn.Wo.weight": [
|
| 906 |
-
768,
|
| 907 |
-
768
|
| 908 |
-
],
|
| 909 |
-
"layers.6.attn.Wqkv.weight": [
|
| 910 |
-
2304,
|
| 911 |
-
768
|
| 912 |
-
],
|
| 913 |
-
"layers.6.attn_norm.weight": [
|
| 914 |
-
768
|
| 915 |
-
],
|
| 916 |
-
"layers.6.mlp.Wi.weight": [
|
| 917 |
-
2304,
|
| 918 |
-
768
|
| 919 |
-
],
|
| 920 |
-
"layers.6.mlp.Wo.weight": [
|
| 921 |
-
768,
|
| 922 |
-
1152
|
| 923 |
-
],
|
| 924 |
-
"layers.6.mlp_norm.weight": [
|
| 925 |
-
768
|
| 926 |
-
],
|
| 927 |
-
"layers.7.attn.Wo.weight": [
|
| 928 |
-
768,
|
| 929 |
-
768
|
| 930 |
-
],
|
| 931 |
-
"layers.7.attn.Wqkv.weight": [
|
| 932 |
-
2304,
|
| 933 |
-
768
|
| 934 |
-
],
|
| 935 |
-
"layers.7.attn_norm.weight": [
|
| 936 |
-
768
|
| 937 |
-
],
|
| 938 |
-
"layers.7.mlp.Wi.weight": [
|
| 939 |
-
2304,
|
| 940 |
-
768
|
| 941 |
-
],
|
| 942 |
-
"layers.7.mlp.Wo.weight": [
|
| 943 |
-
768,
|
| 944 |
-
1152
|
| 945 |
-
],
|
| 946 |
-
"layers.7.mlp_norm.weight": [
|
| 947 |
-
768
|
| 948 |
-
],
|
| 949 |
-
"layers.8.attn.Wo.weight": [
|
| 950 |
-
768,
|
| 951 |
-
768
|
| 952 |
-
],
|
| 953 |
-
"layers.8.attn.Wqkv.weight": [
|
| 954 |
-
2304,
|
| 955 |
-
768
|
| 956 |
-
],
|
| 957 |
-
"layers.8.attn_norm.weight": [
|
| 958 |
-
768
|
| 959 |
-
],
|
| 960 |
-
"layers.8.mlp.Wi.weight": [
|
| 961 |
-
2304,
|
| 962 |
-
768
|
| 963 |
-
],
|
| 964 |
-
"layers.8.mlp.Wo.weight": [
|
| 965 |
-
768,
|
| 966 |
-
1152
|
| 967 |
-
],
|
| 968 |
-
"layers.8.mlp_norm.weight": [
|
| 969 |
-
768
|
| 970 |
-
],
|
| 971 |
-
"layers.9.attn.Wo.weight": [
|
| 972 |
-
768,
|
| 973 |
-
768
|
| 974 |
-
],
|
| 975 |
-
"layers.9.attn.Wqkv.weight": [
|
| 976 |
-
2304,
|
| 977 |
-
768
|
| 978 |
-
],
|
| 979 |
-
"layers.9.attn_norm.weight": [
|
| 980 |
-
768
|
| 981 |
-
],
|
| 982 |
-
"layers.9.mlp.Wi.weight": [
|
| 983 |
-
2304,
|
| 984 |
-
768
|
| 985 |
-
],
|
| 986 |
-
"layers.9.mlp.Wo.weight": [
|
| 987 |
-
768,
|
| 988 |
-
1152
|
| 989 |
-
],
|
| 990 |
-
"layers.9.mlp_norm.weight": [
|
| 991 |
-
768
|
| 992 |
-
]
|
| 993 |
-
},
|
| 994 |
-
"head_shapes": {
|
| 995 |
-
"heads.choice.0.linear1.bias": [
|
| 996 |
-
3072
|
| 997 |
-
],
|
| 998 |
-
"heads.choice.0.linear1.weight": [
|
| 999 |
-
3072,
|
| 1000 |
-
768
|
| 1001 |
-
],
|
| 1002 |
-
"heads.choice.0.linear2.bias": [
|
| 1003 |
-
768
|
| 1004 |
-
],
|
| 1005 |
-
"heads.choice.0.linear2.weight": [
|
| 1006 |
-
768,
|
| 1007 |
-
3072
|
| 1008 |
-
],
|
| 1009 |
-
"heads.choice.0.norm1.bias": [
|
| 1010 |
-
768
|
| 1011 |
-
],
|
| 1012 |
-
"heads.choice.0.norm1.weight": [
|
| 1013 |
-
768
|
| 1014 |
-
],
|
| 1015 |
-
"heads.choice.0.norm2.bias": [
|
| 1016 |
-
768
|
| 1017 |
-
],
|
| 1018 |
-
"heads.choice.0.norm2.weight": [
|
| 1019 |
-
768
|
| 1020 |
-
],
|
| 1021 |
-
"heads.choice.0.self_attn.in_proj_bias": [
|
| 1022 |
-
2304
|
| 1023 |
-
],
|
| 1024 |
-
"heads.choice.0.self_attn.in_proj_weight": [
|
| 1025 |
-
2304,
|
| 1026 |
-
768
|
| 1027 |
-
],
|
| 1028 |
-
"heads.choice.0.self_attn.out_proj.bias": [
|
| 1029 |
-
768
|
| 1030 |
-
],
|
| 1031 |
-
"heads.choice.0.self_attn.out_proj.weight": [
|
| 1032 |
-
768,
|
| 1033 |
-
768
|
| 1034 |
-
],
|
| 1035 |
-
"heads.choice.1.linear1.bias": [
|
| 1036 |
-
3072
|
| 1037 |
-
],
|
| 1038 |
-
"heads.choice.1.linear1.weight": [
|
| 1039 |
-
3072,
|
| 1040 |
-
768
|
| 1041 |
-
],
|
| 1042 |
-
"heads.choice.1.linear2.bias": [
|
| 1043 |
-
768
|
| 1044 |
-
],
|
| 1045 |
-
"heads.choice.1.linear2.weight": [
|
| 1046 |
-
768,
|
| 1047 |
-
3072
|
| 1048 |
-
],
|
| 1049 |
-
"heads.choice.1.norm1.bias": [
|
| 1050 |
-
768
|
| 1051 |
-
],
|
| 1052 |
-
"heads.choice.1.norm1.weight": [
|
| 1053 |
-
768
|
| 1054 |
-
],
|
| 1055 |
-
"heads.choice.1.norm2.bias": [
|
| 1056 |
-
768
|
| 1057 |
-
],
|
| 1058 |
-
"heads.choice.1.norm2.weight": [
|
| 1059 |
-
768
|
| 1060 |
-
],
|
| 1061 |
-
"heads.choice.1.self_attn.in_proj_bias": [
|
| 1062 |
-
2304
|
| 1063 |
-
],
|
| 1064 |
-
"heads.choice.1.self_attn.in_proj_weight": [
|
| 1065 |
-
2304,
|
| 1066 |
-
768
|
| 1067 |
-
],
|
| 1068 |
-
"heads.choice.1.self_attn.out_proj.bias": [
|
| 1069 |
-
768
|
| 1070 |
-
],
|
| 1071 |
-
"heads.choice.1.self_attn.out_proj.weight": [
|
| 1072 |
-
768,
|
| 1073 |
-
768
|
| 1074 |
-
],
|
| 1075 |
-
"heads.noul.0.linear1.bias": [
|
| 1076 |
-
3072
|
| 1077 |
-
],
|
| 1078 |
-
"heads.noul.0.linear1.weight": [
|
| 1079 |
-
3072,
|
| 1080 |
-
768
|
| 1081 |
-
],
|
| 1082 |
-
"heads.noul.0.linear2.bias": [
|
| 1083 |
-
768
|
| 1084 |
-
],
|
| 1085 |
-
"heads.noul.0.linear2.weight": [
|
| 1086 |
-
768,
|
| 1087 |
-
3072
|
| 1088 |
-
],
|
| 1089 |
-
"heads.noul.0.norm1.bias": [
|
| 1090 |
-
768
|
| 1091 |
-
],
|
| 1092 |
-
"heads.noul.0.norm1.weight": [
|
| 1093 |
-
768
|
| 1094 |
-
],
|
| 1095 |
-
"heads.noul.0.norm2.bias": [
|
| 1096 |
-
768
|
| 1097 |
-
],
|
| 1098 |
-
"heads.noul.0.norm2.weight": [
|
| 1099 |
-
768
|
| 1100 |
-
],
|
| 1101 |
-
"heads.noul.0.self_attn.in_proj_bias": [
|
| 1102 |
-
2304
|
| 1103 |
-
],
|
| 1104 |
-
"heads.noul.0.self_attn.in_proj_weight": [
|
| 1105 |
-
2304,
|
| 1106 |
-
768
|
| 1107 |
-
],
|
| 1108 |
-
"heads.noul.0.self_attn.out_proj.bias": [
|
| 1109 |
-
768
|
| 1110 |
-
],
|
| 1111 |
-
"heads.noul.0.self_attn.out_proj.weight": [
|
| 1112 |
-
768,
|
| 1113 |
-
768
|
| 1114 |
-
],
|
| 1115 |
-
"heads.noul.1.linear1.bias": [
|
| 1116 |
-
3072
|
| 1117 |
-
],
|
| 1118 |
-
"heads.noul.1.linear1.weight": [
|
| 1119 |
-
3072,
|
| 1120 |
-
768
|
| 1121 |
-
],
|
| 1122 |
-
"heads.noul.1.linear2.bias": [
|
| 1123 |
-
768
|
| 1124 |
-
],
|
| 1125 |
-
"heads.noul.1.linear2.weight": [
|
| 1126 |
-
768,
|
| 1127 |
-
3072
|
| 1128 |
-
],
|
| 1129 |
-
"heads.noul.1.norm1.bias": [
|
| 1130 |
-
768
|
| 1131 |
-
],
|
| 1132 |
-
"heads.noul.1.norm1.weight": [
|
| 1133 |
-
768
|
| 1134 |
-
],
|
| 1135 |
-
"heads.noul.1.norm2.bias": [
|
| 1136 |
-
768
|
| 1137 |
-
],
|
| 1138 |
-
"heads.noul.1.norm2.weight": [
|
| 1139 |
-
768
|
| 1140 |
-
],
|
| 1141 |
-
"heads.noul.1.self_attn.in_proj_bias": [
|
| 1142 |
-
2304
|
| 1143 |
-
],
|
| 1144 |
-
"heads.noul.1.self_attn.in_proj_weight": [
|
| 1145 |
-
2304,
|
| 1146 |
-
768
|
| 1147 |
-
],
|
| 1148 |
-
"heads.noul.1.self_attn.out_proj.bias": [
|
| 1149 |
-
768
|
| 1150 |
-
],
|
| 1151 |
-
"heads.noul.1.self_attn.out_proj.weight": [
|
| 1152 |
-
768,
|
| 1153 |
-
768
|
| 1154 |
-
],
|
| 1155 |
-
"heads.score.0.linear1.bias": [
|
| 1156 |
-
3072
|
| 1157 |
-
],
|
| 1158 |
-
"heads.score.0.linear1.weight": [
|
| 1159 |
-
3072,
|
| 1160 |
-
768
|
| 1161 |
-
],
|
| 1162 |
-
"heads.score.0.linear2.bias": [
|
| 1163 |
-
768
|
| 1164 |
-
],
|
| 1165 |
-
"heads.score.0.linear2.weight": [
|
| 1166 |
-
768,
|
| 1167 |
-
3072
|
| 1168 |
-
],
|
| 1169 |
-
"heads.score.0.norm1.bias": [
|
| 1170 |
-
768
|
| 1171 |
-
],
|
| 1172 |
-
"heads.score.0.norm1.weight": [
|
| 1173 |
-
768
|
| 1174 |
-
],
|
| 1175 |
-
"heads.score.0.norm2.bias": [
|
| 1176 |
-
768
|
| 1177 |
-
],
|
| 1178 |
-
"heads.score.0.norm2.weight": [
|
| 1179 |
-
768
|
| 1180 |
-
],
|
| 1181 |
-
"heads.score.0.self_attn.in_proj_bias": [
|
| 1182 |
-
2304
|
| 1183 |
-
],
|
| 1184 |
-
"heads.score.0.self_attn.in_proj_weight": [
|
| 1185 |
-
2304,
|
| 1186 |
-
768
|
| 1187 |
-
],
|
| 1188 |
-
"heads.score.0.self_attn.out_proj.bias": [
|
| 1189 |
-
768
|
| 1190 |
-
],
|
| 1191 |
-
"heads.score.0.self_attn.out_proj.weight": [
|
| 1192 |
-
768,
|
| 1193 |
-
768
|
| 1194 |
-
],
|
| 1195 |
-
"heads.score.1.linear1.bias": [
|
| 1196 |
-
3072
|
| 1197 |
-
],
|
| 1198 |
-
"heads.score.1.linear1.weight": [
|
| 1199 |
-
3072,
|
| 1200 |
-
768
|
| 1201 |
-
],
|
| 1202 |
-
"heads.score.1.linear2.bias": [
|
| 1203 |
-
768
|
| 1204 |
-
],
|
| 1205 |
-
"heads.score.1.linear2.weight": [
|
| 1206 |
-
768,
|
| 1207 |
-
3072
|
| 1208 |
-
],
|
| 1209 |
-
"heads.score.1.norm1.bias": [
|
| 1210 |
-
768
|
| 1211 |
-
],
|
| 1212 |
-
"heads.score.1.norm1.weight": [
|
| 1213 |
-
768
|
| 1214 |
-
],
|
| 1215 |
-
"heads.score.1.norm2.bias": [
|
| 1216 |
-
768
|
| 1217 |
-
],
|
| 1218 |
-
"heads.score.1.norm2.weight": [
|
| 1219 |
-
768
|
| 1220 |
-
],
|
| 1221 |
-
"heads.score.1.self_attn.in_proj_bias": [
|
| 1222 |
-
2304
|
| 1223 |
-
],
|
| 1224 |
-
"heads.score.1.self_attn.in_proj_weight": [
|
| 1225 |
-
2304,
|
| 1226 |
-
768
|
| 1227 |
-
],
|
| 1228 |
-
"heads.score.1.self_attn.out_proj.bias": [
|
| 1229 |
-
768
|
| 1230 |
-
],
|
| 1231 |
-
"heads.score.1.self_attn.out_proj.weight": [
|
| 1232 |
-
768,
|
| 1233 |
-
768
|
| 1234 |
-
],
|
| 1235 |
-
"scorers.choice.0.bias": [
|
| 1236 |
-
768
|
| 1237 |
-
],
|
| 1238 |
-
"scorers.choice.0.weight": [
|
| 1239 |
-
768
|
| 1240 |
-
],
|
| 1241 |
-
"scorers.choice.1.bias": [
|
| 1242 |
-
768
|
| 1243 |
-
],
|
| 1244 |
-
"scorers.choice.1.weight": [
|
| 1245 |
-
768,
|
| 1246 |
-
768
|
| 1247 |
-
],
|
| 1248 |
-
"scorers.choice.3.bias": [
|
| 1249 |
-
1
|
| 1250 |
-
],
|
| 1251 |
-
"scorers.choice.3.weight": [
|
| 1252 |
-
1,
|
| 1253 |
-
768
|
| 1254 |
-
],
|
| 1255 |
-
"scorers.noul.0.bias": [
|
| 1256 |
-
768
|
| 1257 |
-
],
|
| 1258 |
-
"scorers.noul.0.weight": [
|
| 1259 |
-
768
|
| 1260 |
-
],
|
| 1261 |
-
"scorers.noul.1.bias": [
|
| 1262 |
-
768
|
| 1263 |
-
],
|
| 1264 |
-
"scorers.noul.1.weight": [
|
| 1265 |
-
768,
|
| 1266 |
-
768
|
| 1267 |
-
],
|
| 1268 |
-
"scorers.noul.3.bias": [
|
| 1269 |
-
1
|
| 1270 |
-
],
|
| 1271 |
-
"scorers.noul.3.weight": [
|
| 1272 |
-
1,
|
| 1273 |
-
768
|
| 1274 |
-
],
|
| 1275 |
-
"scorers.score.0.bias": [
|
| 1276 |
-
768
|
| 1277 |
-
],
|
| 1278 |
-
"scorers.score.0.weight": [
|
| 1279 |
-
768
|
| 1280 |
-
],
|
| 1281 |
-
"scorers.score.1.bias": [
|
| 1282 |
-
768
|
| 1283 |
-
],
|
| 1284 |
-
"scorers.score.1.weight": [
|
| 1285 |
-
768,
|
| 1286 |
-
768
|
| 1287 |
-
],
|
| 1288 |
-
"scorers.score.3.bias": [
|
| 1289 |
-
1
|
| 1290 |
-
],
|
| 1291 |
-
"scorers.score.3.weight": [
|
| 1292 |
-
1,
|
| 1293 |
-
768
|
| 1294 |
-
],
|
| 1295 |
-
"type_embedding.weight": [
|
| 1296 |
-
3,
|
| 1297 |
-
768
|
| 1298 |
-
]
|
| 1299 |
-
},
|
| 1300 |
-
"score_suffix_shapes": {
|
| 1301 |
-
"score_blocks.0.attn.Wo.weight": [
|
| 1302 |
-
768,
|
| 1303 |
-
768
|
| 1304 |
-
],
|
| 1305 |
-
"score_blocks.0.attn.Wqkv.weight": [
|
| 1306 |
-
2304,
|
| 1307 |
-
768
|
| 1308 |
-
],
|
| 1309 |
-
"score_blocks.0.mlp.Wi.weight": [
|
| 1310 |
-
2304,
|
| 1311 |
-
768
|
| 1312 |
-
],
|
| 1313 |
-
"score_blocks.0.mlp.Wo.weight": [
|
| 1314 |
-
768,
|
| 1315 |
-
1152
|
| 1316 |
-
],
|
| 1317 |
-
"score_blocks.0.mlp_norm.weight": [
|
| 1318 |
-
768
|
| 1319 |
-
],
|
| 1320 |
-
"score_blocks.1.attn.Wo.weight": [
|
| 1321 |
-
768,
|
| 1322 |
-
768
|
| 1323 |
-
],
|
| 1324 |
-
"score_blocks.1.attn.Wqkv.weight": [
|
| 1325 |
-
2304,
|
| 1326 |
-
768
|
| 1327 |
-
],
|
| 1328 |
-
"score_blocks.1.attn_norm.weight": [
|
| 1329 |
-
768
|
| 1330 |
-
],
|
| 1331 |
-
"score_blocks.1.mlp.Wi.weight": [
|
| 1332 |
-
2304,
|
| 1333 |
-
768
|
| 1334 |
-
],
|
| 1335 |
-
"score_blocks.1.mlp.Wo.weight": [
|
| 1336 |
-
768,
|
| 1337 |
-
1152
|
| 1338 |
-
],
|
| 1339 |
-
"score_blocks.1.mlp_norm.weight": [
|
| 1340 |
-
768
|
| 1341 |
-
],
|
| 1342 |
-
"score_blocks.10.attn.Wo.weight": [
|
| 1343 |
-
768,
|
| 1344 |
-
768
|
| 1345 |
-
],
|
| 1346 |
-
"score_blocks.10.attn.Wqkv.weight": [
|
| 1347 |
-
2304,
|
| 1348 |
-
768
|
| 1349 |
-
],
|
| 1350 |
-
"score_blocks.10.attn_norm.weight": [
|
| 1351 |
-
768
|
| 1352 |
-
],
|
| 1353 |
-
"score_blocks.10.mlp.Wi.weight": [
|
| 1354 |
-
2304,
|
| 1355 |
-
768
|
| 1356 |
-
],
|
| 1357 |
-
"score_blocks.10.mlp.Wo.weight": [
|
| 1358 |
-
768,
|
| 1359 |
-
1152
|
| 1360 |
-
],
|
| 1361 |
-
"score_blocks.10.mlp_norm.weight": [
|
| 1362 |
-
768
|
| 1363 |
-
],
|
| 1364 |
-
"score_blocks.11.attn.Wo.weight": [
|
| 1365 |
-
768,
|
| 1366 |
-
768
|
| 1367 |
-
],
|
| 1368 |
-
"score_blocks.11.attn.Wqkv.weight": [
|
| 1369 |
-
2304,
|
| 1370 |
-
768
|
| 1371 |
-
],
|
| 1372 |
-
"score_blocks.11.attn_norm.weight": [
|
| 1373 |
-
768
|
| 1374 |
-
],
|
| 1375 |
-
"score_blocks.11.mlp.Wi.weight": [
|
| 1376 |
-
2304,
|
| 1377 |
-
768
|
| 1378 |
-
],
|
| 1379 |
-
"score_blocks.11.mlp.Wo.weight": [
|
| 1380 |
-
768,
|
| 1381 |
-
1152
|
| 1382 |
-
],
|
| 1383 |
-
"score_blocks.11.mlp_norm.weight": [
|
| 1384 |
-
768
|
| 1385 |
-
],
|
| 1386 |
-
"score_blocks.12.attn.Wo.weight": [
|
| 1387 |
-
768,
|
| 1388 |
-
768
|
| 1389 |
-
],
|
| 1390 |
-
"score_blocks.12.attn.Wqkv.weight": [
|
| 1391 |
-
2304,
|
| 1392 |
-
768
|
| 1393 |
-
],
|
| 1394 |
-
"score_blocks.12.attn_norm.weight": [
|
| 1395 |
-
768
|
| 1396 |
-
],
|
| 1397 |
-
"score_blocks.12.mlp.Wi.weight": [
|
| 1398 |
-
2304,
|
| 1399 |
-
768
|
| 1400 |
-
],
|
| 1401 |
-
"score_blocks.12.mlp.Wo.weight": [
|
| 1402 |
-
768,
|
| 1403 |
-
1152
|
| 1404 |
-
],
|
| 1405 |
-
"score_blocks.12.mlp_norm.weight": [
|
| 1406 |
-
768
|
| 1407 |
-
],
|
| 1408 |
-
"score_blocks.13.attn.Wo.weight": [
|
| 1409 |
-
768,
|
| 1410 |
-
768
|
| 1411 |
-
],
|
| 1412 |
-
"score_blocks.13.attn.Wqkv.weight": [
|
| 1413 |
-
2304,
|
| 1414 |
-
768
|
| 1415 |
-
],
|
| 1416 |
-
"score_blocks.13.attn_norm.weight": [
|
| 1417 |
-
768
|
| 1418 |
-
],
|
| 1419 |
-
"score_blocks.13.mlp.Wi.weight": [
|
| 1420 |
-
2304,
|
| 1421 |
-
768
|
| 1422 |
-
],
|
| 1423 |
-
"score_blocks.13.mlp.Wo.weight": [
|
| 1424 |
-
768,
|
| 1425 |
-
1152
|
| 1426 |
-
],
|
| 1427 |
-
"score_blocks.13.mlp_norm.weight": [
|
| 1428 |
-
768
|
| 1429 |
-
],
|
| 1430 |
-
"score_blocks.14.attn.Wo.weight": [
|
| 1431 |
-
768,
|
| 1432 |
-
768
|
| 1433 |
-
],
|
| 1434 |
-
"score_blocks.14.attn.Wqkv.weight": [
|
| 1435 |
-
2304,
|
| 1436 |
-
768
|
| 1437 |
-
],
|
| 1438 |
-
"score_blocks.14.attn_norm.weight": [
|
| 1439 |
-
768
|
| 1440 |
-
],
|
| 1441 |
-
"score_blocks.14.mlp.Wi.weight": [
|
| 1442 |
-
2304,
|
| 1443 |
-
768
|
| 1444 |
-
],
|
| 1445 |
-
"score_blocks.14.mlp.Wo.weight": [
|
| 1446 |
-
768,
|
| 1447 |
-
1152
|
| 1448 |
-
],
|
| 1449 |
-
"score_blocks.14.mlp_norm.weight": [
|
| 1450 |
-
768
|
| 1451 |
-
],
|
| 1452 |
-
"score_blocks.15.attn.Wo.weight": [
|
| 1453 |
-
768,
|
| 1454 |
-
768
|
| 1455 |
-
],
|
| 1456 |
-
"score_blocks.15.attn.Wqkv.weight": [
|
| 1457 |
-
2304,
|
| 1458 |
-
768
|
| 1459 |
-
],
|
| 1460 |
-
"score_blocks.15.attn_norm.weight": [
|
| 1461 |
-
768
|
| 1462 |
-
],
|
| 1463 |
-
"score_blocks.15.mlp.Wi.weight": [
|
| 1464 |
-
2304,
|
| 1465 |
-
768
|
| 1466 |
-
],
|
| 1467 |
-
"score_blocks.15.mlp.Wo.weight": [
|
| 1468 |
-
768,
|
| 1469 |
-
1152
|
| 1470 |
-
],
|
| 1471 |
-
"score_blocks.15.mlp_norm.weight": [
|
| 1472 |
-
768
|
| 1473 |
-
],
|
| 1474 |
-
"score_blocks.16.attn.Wo.weight": [
|
| 1475 |
-
768,
|
| 1476 |
-
768
|
| 1477 |
-
],
|
| 1478 |
-
"score_blocks.16.attn.Wqkv.weight": [
|
| 1479 |
-
2304,
|
| 1480 |
-
768
|
| 1481 |
-
],
|
| 1482 |
-
"score_blocks.16.attn_norm.weight": [
|
| 1483 |
-
768
|
| 1484 |
-
],
|
| 1485 |
-
"score_blocks.16.mlp.Wi.weight": [
|
| 1486 |
-
2304,
|
| 1487 |
-
768
|
| 1488 |
-
],
|
| 1489 |
-
"score_blocks.16.mlp.Wo.weight": [
|
| 1490 |
-
768,
|
| 1491 |
-
1152
|
| 1492 |
-
],
|
| 1493 |
-
"score_blocks.16.mlp_norm.weight": [
|
| 1494 |
-
768
|
| 1495 |
-
],
|
| 1496 |
-
"score_blocks.17.attn.Wo.weight": [
|
| 1497 |
-
768,
|
| 1498 |
-
768
|
| 1499 |
-
],
|
| 1500 |
-
"score_blocks.17.attn.Wqkv.weight": [
|
| 1501 |
-
2304,
|
| 1502 |
-
768
|
| 1503 |
-
],
|
| 1504 |
-
"score_blocks.17.attn_norm.weight": [
|
| 1505 |
-
768
|
| 1506 |
-
],
|
| 1507 |
-
"score_blocks.17.mlp.Wi.weight": [
|
| 1508 |
-
2304,
|
| 1509 |
-
768
|
| 1510 |
-
],
|
| 1511 |
-
"score_blocks.17.mlp.Wo.weight": [
|
| 1512 |
-
768,
|
| 1513 |
-
1152
|
| 1514 |
-
],
|
| 1515 |
-
"score_blocks.17.mlp_norm.weight": [
|
| 1516 |
-
768
|
| 1517 |
-
],
|
| 1518 |
-
"score_blocks.18.attn.Wo.weight": [
|
| 1519 |
-
768,
|
| 1520 |
-
768
|
| 1521 |
-
],
|
| 1522 |
-
"score_blocks.18.attn.Wqkv.weight": [
|
| 1523 |
-
2304,
|
| 1524 |
-
768
|
| 1525 |
-
],
|
| 1526 |
-
"score_blocks.18.attn_norm.weight": [
|
| 1527 |
-
768
|
| 1528 |
-
],
|
| 1529 |
-
"score_blocks.18.mlp.Wi.weight": [
|
| 1530 |
-
2304,
|
| 1531 |
-
768
|
| 1532 |
-
],
|
| 1533 |
-
"score_blocks.18.mlp.Wo.weight": [
|
| 1534 |
-
768,
|
| 1535 |
-
1152
|
| 1536 |
-
],
|
| 1537 |
-
"score_blocks.18.mlp_norm.weight": [
|
| 1538 |
-
768
|
| 1539 |
-
],
|
| 1540 |
-
"score_blocks.19.attn.Wo.weight": [
|
| 1541 |
-
768,
|
| 1542 |
-
768
|
| 1543 |
-
],
|
| 1544 |
-
"score_blocks.19.attn.Wqkv.weight": [
|
| 1545 |
-
2304,
|
| 1546 |
-
768
|
| 1547 |
-
],
|
| 1548 |
-
"score_blocks.19.attn_norm.weight": [
|
| 1549 |
-
768
|
| 1550 |
-
],
|
| 1551 |
-
"score_blocks.19.mlp.Wi.weight": [
|
| 1552 |
-
2304,
|
| 1553 |
-
768
|
| 1554 |
-
],
|
| 1555 |
-
"score_blocks.19.mlp.Wo.weight": [
|
| 1556 |
-
768,
|
| 1557 |
-
1152
|
| 1558 |
-
],
|
| 1559 |
-
"score_blocks.19.mlp_norm.weight": [
|
| 1560 |
-
768
|
| 1561 |
-
],
|
| 1562 |
-
"score_blocks.2.attn.Wo.weight": [
|
| 1563 |
-
768,
|
| 1564 |
-
768
|
| 1565 |
-
],
|
| 1566 |
-
"score_blocks.2.attn.Wqkv.weight": [
|
| 1567 |
-
2304,
|
| 1568 |
-
768
|
| 1569 |
-
],
|
| 1570 |
-
"score_blocks.2.attn_norm.weight": [
|
| 1571 |
-
768
|
| 1572 |
-
],
|
| 1573 |
-
"score_blocks.2.mlp.Wi.weight": [
|
| 1574 |
-
2304,
|
| 1575 |
-
768
|
| 1576 |
-
],
|
| 1577 |
-
"score_blocks.2.mlp.Wo.weight": [
|
| 1578 |
-
768,
|
| 1579 |
-
1152
|
| 1580 |
-
],
|
| 1581 |
-
"score_blocks.2.mlp_norm.weight": [
|
| 1582 |
-
768
|
| 1583 |
-
],
|
| 1584 |
-
"score_blocks.20.attn.Wo.weight": [
|
| 1585 |
-
768,
|
| 1586 |
-
768
|
| 1587 |
-
],
|
| 1588 |
-
"score_blocks.20.attn.Wqkv.weight": [
|
| 1589 |
-
2304,
|
| 1590 |
-
768
|
| 1591 |
-
],
|
| 1592 |
-
"score_blocks.20.attn_norm.weight": [
|
| 1593 |
-
768
|
| 1594 |
-
],
|
| 1595 |
-
"score_blocks.20.mlp.Wi.weight": [
|
| 1596 |
-
2304,
|
| 1597 |
-
768
|
| 1598 |
-
],
|
| 1599 |
-
"score_blocks.20.mlp.Wo.weight": [
|
| 1600 |
-
768,
|
| 1601 |
-
1152
|
| 1602 |
-
],
|
| 1603 |
-
"score_blocks.20.mlp_norm.weight": [
|
| 1604 |
-
768
|
| 1605 |
-
],
|
| 1606 |
-
"score_blocks.21.attn.Wo.weight": [
|
| 1607 |
-
768,
|
| 1608 |
-
768
|
| 1609 |
-
],
|
| 1610 |
-
"score_blocks.21.attn.Wqkv.weight": [
|
| 1611 |
-
2304,
|
| 1612 |
-
768
|
| 1613 |
-
],
|
| 1614 |
-
"score_blocks.21.attn_norm.weight": [
|
| 1615 |
-
768
|
| 1616 |
-
],
|
| 1617 |
-
"score_blocks.21.mlp.Wi.weight": [
|
| 1618 |
-
2304,
|
| 1619 |
-
768
|
| 1620 |
-
],
|
| 1621 |
-
"score_blocks.21.mlp.Wo.weight": [
|
| 1622 |
-
768,
|
| 1623 |
-
1152
|
| 1624 |
-
],
|
| 1625 |
-
"score_blocks.21.mlp_norm.weight": [
|
| 1626 |
-
768
|
| 1627 |
-
],
|
| 1628 |
-
"score_blocks.3.attn.Wo.weight": [
|
| 1629 |
-
768,
|
| 1630 |
-
768
|
| 1631 |
-
],
|
| 1632 |
-
"score_blocks.3.attn.Wqkv.weight": [
|
| 1633 |
-
2304,
|
| 1634 |
-
768
|
| 1635 |
-
],
|
| 1636 |
-
"score_blocks.3.attn_norm.weight": [
|
| 1637 |
-
768
|
| 1638 |
-
],
|
| 1639 |
-
"score_blocks.3.mlp.Wi.weight": [
|
| 1640 |
-
2304,
|
| 1641 |
-
768
|
| 1642 |
-
],
|
| 1643 |
-
"score_blocks.3.mlp.Wo.weight": [
|
| 1644 |
-
768,
|
| 1645 |
-
1152
|
| 1646 |
-
],
|
| 1647 |
-
"score_blocks.3.mlp_norm.weight": [
|
| 1648 |
-
768
|
| 1649 |
-
],
|
| 1650 |
-
"score_blocks.4.attn.Wo.weight": [
|
| 1651 |
-
768,
|
| 1652 |
-
768
|
| 1653 |
-
],
|
| 1654 |
-
"score_blocks.4.attn.Wqkv.weight": [
|
| 1655 |
-
2304,
|
| 1656 |
-
768
|
| 1657 |
-
],
|
| 1658 |
-
"score_blocks.4.attn_norm.weight": [
|
| 1659 |
-
768
|
| 1660 |
-
],
|
| 1661 |
-
"score_blocks.4.mlp.Wi.weight": [
|
| 1662 |
-
2304,
|
| 1663 |
-
768
|
| 1664 |
-
],
|
| 1665 |
-
"score_blocks.4.mlp.Wo.weight": [
|
| 1666 |
-
768,
|
| 1667 |
-
1152
|
| 1668 |
-
],
|
| 1669 |
-
"score_blocks.4.mlp_norm.weight": [
|
| 1670 |
-
768
|
| 1671 |
-
],
|
| 1672 |
-
"score_blocks.5.attn.Wo.weight": [
|
| 1673 |
-
768,
|
| 1674 |
-
768
|
| 1675 |
-
],
|
| 1676 |
-
"score_blocks.5.attn.Wqkv.weight": [
|
| 1677 |
-
2304,
|
| 1678 |
-
768
|
| 1679 |
-
],
|
| 1680 |
-
"score_blocks.5.attn_norm.weight": [
|
| 1681 |
-
768
|
| 1682 |
-
],
|
| 1683 |
-
"score_blocks.5.mlp.Wi.weight": [
|
| 1684 |
-
2304,
|
| 1685 |
-
768
|
| 1686 |
-
],
|
| 1687 |
-
"score_blocks.5.mlp.Wo.weight": [
|
| 1688 |
-
768,
|
| 1689 |
-
1152
|
| 1690 |
-
],
|
| 1691 |
-
"score_blocks.5.mlp_norm.weight": [
|
| 1692 |
-
768
|
| 1693 |
-
],
|
| 1694 |
-
"score_blocks.6.attn.Wo.weight": [
|
| 1695 |
-
768,
|
| 1696 |
-
768
|
| 1697 |
-
],
|
| 1698 |
-
"score_blocks.6.attn.Wqkv.weight": [
|
| 1699 |
-
2304,
|
| 1700 |
-
768
|
| 1701 |
-
],
|
| 1702 |
-
"score_blocks.6.attn_norm.weight": [
|
| 1703 |
-
768
|
| 1704 |
-
],
|
| 1705 |
-
"score_blocks.6.mlp.Wi.weight": [
|
| 1706 |
-
2304,
|
| 1707 |
-
768
|
| 1708 |
-
],
|
| 1709 |
-
"score_blocks.6.mlp.Wo.weight": [
|
| 1710 |
-
768,
|
| 1711 |
-
1152
|
| 1712 |
-
],
|
| 1713 |
-
"score_blocks.6.mlp_norm.weight": [
|
| 1714 |
-
768
|
| 1715 |
-
],
|
| 1716 |
-
"score_blocks.7.attn.Wo.weight": [
|
| 1717 |
-
768,
|
| 1718 |
-
768
|
| 1719 |
-
],
|
| 1720 |
-
"score_blocks.7.attn.Wqkv.weight": [
|
| 1721 |
-
2304,
|
| 1722 |
-
768
|
| 1723 |
-
],
|
| 1724 |
-
"score_blocks.7.attn_norm.weight": [
|
| 1725 |
-
768
|
| 1726 |
-
],
|
| 1727 |
-
"score_blocks.7.mlp.Wi.weight": [
|
| 1728 |
-
2304,
|
| 1729 |
-
768
|
| 1730 |
-
],
|
| 1731 |
-
"score_blocks.7.mlp.Wo.weight": [
|
| 1732 |
-
768,
|
| 1733 |
-
1152
|
| 1734 |
-
],
|
| 1735 |
-
"score_blocks.7.mlp_norm.weight": [
|
| 1736 |
-
768
|
| 1737 |
-
],
|
| 1738 |
-
"score_blocks.8.attn.Wo.weight": [
|
| 1739 |
-
768,
|
| 1740 |
-
768
|
| 1741 |
-
],
|
| 1742 |
-
"score_blocks.8.attn.Wqkv.weight": [
|
| 1743 |
-
2304,
|
| 1744 |
-
768
|
| 1745 |
-
],
|
| 1746 |
-
"score_blocks.8.attn_norm.weight": [
|
| 1747 |
-
768
|
| 1748 |
-
],
|
| 1749 |
-
"score_blocks.8.mlp.Wi.weight": [
|
| 1750 |
-
2304,
|
| 1751 |
-
768
|
| 1752 |
-
],
|
| 1753 |
-
"score_blocks.8.mlp.Wo.weight": [
|
| 1754 |
-
768,
|
| 1755 |
-
1152
|
| 1756 |
-
],
|
| 1757 |
-
"score_blocks.8.mlp_norm.weight": [
|
| 1758 |
-
768
|
| 1759 |
-
],
|
| 1760 |
-
"score_blocks.9.attn.Wo.weight": [
|
| 1761 |
-
768,
|
| 1762 |
-
768
|
| 1763 |
-
],
|
| 1764 |
-
"score_blocks.9.attn.Wqkv.weight": [
|
| 1765 |
-
2304,
|
| 1766 |
-
768
|
| 1767 |
-
],
|
| 1768 |
-
"score_blocks.9.attn_norm.weight": [
|
| 1769 |
-
768
|
| 1770 |
-
],
|
| 1771 |
-
"score_blocks.9.mlp.Wi.weight": [
|
| 1772 |
-
2304,
|
| 1773 |
-
768
|
| 1774 |
-
],
|
| 1775 |
-
"score_blocks.9.mlp.Wo.weight": [
|
| 1776 |
-
768,
|
| 1777 |
-
1152
|
| 1778 |
-
],
|
| 1779 |
-
"score_blocks.9.mlp_norm.weight": [
|
| 1780 |
-
768
|
| 1781 |
-
],
|
| 1782 |
-
"score_final_norm.weight": [
|
| 1783 |
-
768
|
| 1784 |
-
]
|
| 1785 |
-
}
|
| 1786 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
native/MANIFEST.json
DELETED
|
@@ -1,125 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"files": {
|
| 3 |
-
"INVENTORY.json": {
|
| 4 |
-
"bytes": 31007,
|
| 5 |
-
"sha256": "06aed00ac4571582b992a995a793875804a0546bdae1461c4161b30fba2143a9"
|
| 6 |
-
},
|
| 7 |
-
"STATE_LAYOUT.json": {
|
| 8 |
-
"bytes": 10971,
|
| 9 |
-
"sha256": "a6b24716e2b21240d327ba8763a73f1b54862bc86d5e1f92bb9e670be968e494"
|
| 10 |
-
},
|
| 11 |
-
"__init__.py": {
|
| 12 |
-
"bytes": 78,
|
| 13 |
-
"sha256": "1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010"
|
| 14 |
-
},
|
| 15 |
-
"artifacts.py": {
|
| 16 |
-
"bytes": 7958,
|
| 17 |
-
"sha256": "c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3"
|
| 18 |
-
},
|
| 19 |
-
"choice_encoder.safetensors": {
|
| 20 |
-
"bytes": 441337216,
|
| 21 |
-
"sha256": "9516cc841c485c98b27b4f63d2ea8e604fe8064121173064b113da8a6bf57ef6"
|
| 22 |
-
},
|
| 23 |
-
"contract.py": {
|
| 24 |
-
"bytes": 10886,
|
| 25 |
-
"sha256": "51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726"
|
| 26 |
-
},
|
| 27 |
-
"decision_config.json": {
|
| 28 |
-
"bytes": 14308,
|
| 29 |
-
"sha256": "1fefb4ad7dede00c4633bb2ce5fc44fb04237207b9e9ce90fcf8ea8b01a1328d"
|
| 30 |
-
},
|
| 31 |
-
"decision_heads.safetensors": {
|
| 32 |
-
"bytes": 177241884,
|
| 33 |
-
"sha256": "bce3ee658a978a19c48c605b921cff994f892e7fc17abd6f8645f07428fbc36f"
|
| 34 |
-
},
|
| 35 |
-
"encoder/config.json": {
|
| 36 |
-
"bytes": 2769,
|
| 37 |
-
"sha256": "7aff915e9f159305e0bef3eb0206416f99b8560260b8969b35f1dcd54aaad1a5"
|
| 38 |
-
},
|
| 39 |
-
"encoder/model.safetensors": {
|
| 40 |
-
"bytes": 1227771752,
|
| 41 |
-
"sha256": "daaafd81c4ed767d203e24226d78e792800068a7dac344aa043844b82a6306c7"
|
| 42 |
-
},
|
| 43 |
-
"infer.py": {
|
| 44 |
-
"bytes": 1095,
|
| 45 |
-
"sha256": "9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315"
|
| 46 |
-
},
|
| 47 |
-
"model.py": {
|
| 48 |
-
"bytes": 11098,
|
| 49 |
-
"sha256": "8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b"
|
| 50 |
-
},
|
| 51 |
-
"packing.py": {
|
| 52 |
-
"bytes": 6748,
|
| 53 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 54 |
-
},
|
| 55 |
-
"policy/__init__.py": {
|
| 56 |
-
"bytes": 131,
|
| 57 |
-
"sha256": "0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114"
|
| 58 |
-
},
|
| 59 |
-
"policy/artifacts.py": {
|
| 60 |
-
"bytes": 7877,
|
| 61 |
-
"sha256": "5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2"
|
| 62 |
-
},
|
| 63 |
-
"policy/contract.py": {
|
| 64 |
-
"bytes": 4334,
|
| 65 |
-
"sha256": "0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67"
|
| 66 |
-
},
|
| 67 |
-
"policy/infer.py": {
|
| 68 |
-
"bytes": 1545,
|
| 69 |
-
"sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
|
| 70 |
-
},
|
| 71 |
-
"policy/model.py": {
|
| 72 |
-
"bytes": 7678,
|
| 73 |
-
"sha256": "eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca"
|
| 74 |
-
},
|
| 75 |
-
"policy/packing.py": {
|
| 76 |
-
"bytes": 6748,
|
| 77 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 78 |
-
},
|
| 79 |
-
"policy/reference/__init__.py": {
|
| 80 |
-
"bytes": 192,
|
| 81 |
-
"sha256": "c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28"
|
| 82 |
-
},
|
| 83 |
-
"policy/reference/artifacts.py": {
|
| 84 |
-
"bytes": 9342,
|
| 85 |
-
"sha256": "3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159"
|
| 86 |
-
},
|
| 87 |
-
"policy/reference/infer.py": {
|
| 88 |
-
"bytes": 1545,
|
| 89 |
-
"sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
|
| 90 |
-
},
|
| 91 |
-
"policy/reference/model.py": {
|
| 92 |
-
"bytes": 5060,
|
| 93 |
-
"sha256": "733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26"
|
| 94 |
-
},
|
| 95 |
-
"policy/reference/modernbert_sdpa_layout.py": {
|
| 96 |
-
"bytes": 3365,
|
| 97 |
-
"sha256": "0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8"
|
| 98 |
-
},
|
| 99 |
-
"policy/reference/packing.py": {
|
| 100 |
-
"bytes": 6748,
|
| 101 |
-
"sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
|
| 102 |
-
},
|
| 103 |
-
"score_encoder.safetensors": {
|
| 104 |
-
"bytes": 441337080,
|
| 105 |
-
"sha256": "4f45795977846ef4c31b34e4f35bf95dabf5951d71358bab32c9a94814a84a53"
|
| 106 |
-
},
|
| 107 |
-
"tokenizer/special_tokens_map.json": {
|
| 108 |
-
"bytes": 1051,
|
| 109 |
-
"sha256": "6e3204c5a4004719c185007a745d1940471a6fb02521cfa0a00306dbb6bc5903"
|
| 110 |
-
},
|
| 111 |
-
"tokenizer/tokenizer.json": {
|
| 112 |
-
"bytes": 34363188,
|
| 113 |
-
"sha256": "609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f"
|
| 114 |
-
},
|
| 115 |
-
"tokenizer/tokenizer_config.json": {
|
| 116 |
-
"bytes": 46470,
|
| 117 |
-
"sha256": "74a259bb1a3811a7e3028adcd07a65765d866d5e66e0e883f0994ccfa67e8455"
|
| 118 |
-
},
|
| 119 |
-
"training_policy.py": {
|
| 120 |
-
"bytes": 4761,
|
| 121 |
-
"sha256": "6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5"
|
| 122 |
-
}
|
| 123 |
-
},
|
| 124 |
-
"schema": "decision.files.v1"
|
| 125 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|