Xunzhuo commited on
Commit
db8a7ed
·
1 Parent(s): e4a96e9

Publish clean Decision model repository

Browse files

Signed-off-by: Xunzhuo <Xunzhuo@users.noreply.huggingface.co>

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +0 -6
  2. ARCHITECTURE.md +1 -1
  3. BATCH_CAPACITY.json +0 -1015
  4. DISTRIBUTION_TERMS.md +1 -1
  5. FINETUNING.md +0 -32
  6. LICENSING_STATUS.md +3 -3
  7. PACKAGE_MANIFEST.json +0 -493
  8. PERFORMANCE.md +0 -66
  9. README.md +13 -7
  10. RUNTIME_UPDATE.json +0 -32
  11. SYSTEM_ONE.md +0 -134
  12. SYSTEM_ONE_VALIDATION.json +0 -66
  13. USAGE.md +0 -121
  14. assets/architecture.pdf +0 -3
  15. assets/attention-geglu.pdf +0 -3
  16. assets/attention-geglu.png +0 -3
  17. assets/candidate-readout.pdf +0 -3
  18. assets/candidate-readout.svg +0 -121
  19. assets/residual-layers.pdf +0 -3
  20. assets/residual-layers.png +0 -3
  21. config.json +21 -0
  22. decision_finetune/__init__.py +0 -1
  23. decision_finetune/__main__.py +0 -80
  24. decision_finetune/data.py +0 -139
  25. decision_finetune/metrics.py +0 -61
  26. decision_finetune/run.py +0 -237
  27. decision_finetune/state.py +0 -110
  28. decision_finetune/system_one.py +0 -34
  29. decision_inference/__init__.py +0 -5
  30. decision_inference/_auto.py +0 -34
  31. decision_inference/_grouped.py +0 -16
  32. decision_inference/_request.py +0 -58
  33. decision_inference/_system_one.py +0 -217
  34. decision_inference/profile.py +0 -51
  35. decision_runtime/__init__.py +0 -11
  36. decision_runtime/_compat.py +0 -20
  37. decision_runtime/native.py +0 -152
  38. decision_runtime/training.py +0 -310
  39. EVALUATION.md → evaluation/EVALUATION.md +0 -0
  40. METHODS.md → evaluation/METHODS.md +1 -1
  41. MIXED_QUESTION_SCALING.json → evaluation/MIXED_QUESTION_SCALING.json +0 -0
  42. evaluation/PERFORMANCE.md +17 -0
  43. TECHNICAL_VALIDATION.json → evaluation/TECHNICAL_VALIDATION.json +0 -0
  44. VALIDATION.md → evaluation/VALIDATION.md +2 -2
  45. examples/decisions.jsonl +0 -3
  46. examples/finetune.sh +0 -21
  47. examples/system-one.json +0 -29
  48. infer.py +0 -46
  49. native/INVENTORY.json +0 -1786
  50. native/MANIFEST.json +0 -125
.gitattributes CHANGED
@@ -1,13 +1,7 @@
1
  *.safetensors filter=lfs diff=lfs merge=lfs -text
2
  native/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
3
  assets/architecture-atlas.pdf filter=lfs diff=lfs merge=lfs -text
4
- assets/architecture.pdf filter=lfs diff=lfs merge=lfs -text
5
  assets/architecture.png filter=lfs diff=lfs merge=lfs -text
6
- assets/attention-geglu.pdf filter=lfs diff=lfs merge=lfs -text
7
- assets/attention-geglu.png filter=lfs diff=lfs merge=lfs -text
8
- assets/candidate-readout.pdf filter=lfs diff=lfs merge=lfs -text
9
  assets/candidate-readout.png filter=lfs diff=lfs merge=lfs -text
10
  assets/decision-lex-header.png filter=lfs diff=lfs merge=lfs -text
11
- assets/residual-layers.pdf filter=lfs diff=lfs merge=lfs -text
12
- assets/residual-layers.png filter=lfs diff=lfs merge=lfs -text
13
  assets/mixed-question-scaling.png filter=lfs diff=lfs merge=lfs -text
 
1
  *.safetensors filter=lfs diff=lfs merge=lfs -text
2
  native/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
3
  assets/architecture-atlas.pdf filter=lfs diff=lfs merge=lfs -text
 
4
  assets/architecture.png filter=lfs diff=lfs merge=lfs -text
 
 
 
5
  assets/candidate-readout.png filter=lfs diff=lfs merge=lfs -text
6
  assets/decision-lex-header.png filter=lfs diff=lfs merge=lfs -text
 
 
7
  assets/mixed-question-scaling.png filter=lfs diff=lfs merge=lfs -text
ARCHITECTURE.md CHANGED
@@ -10,6 +10,6 @@ A question and all of its candidate descriptions are encoded jointly with the co
10
 
11
  Choice returns a candidate and probabilities; Noul returns the probability of yes; Score returns the ordered distribution and its expected value. Candidate IDs preserve associations outside the token stream. Multiple questions use batches, with repeated context encoding.
12
 
13
- The public API accepts complete packed inputs of at most 1,024 tokens per question. No field is silently truncated.
14
 
15
  [Main SVG](assets/architecture.svg) · [Residual layers](assets/residual-layers.svg) · [Attention and GEGLU](assets/attention-geglu.svg) · [PDF atlas](assets/architecture-atlas.pdf)
 
10
 
11
  Choice returns a candidate and probabilities; Noul returns the probability of yes; Score returns the ordered distribution and its expected value. Candidate IDs preserve associations outside the token stream. Multiple questions use batches, with repeated context encoding.
12
 
13
+ The evaluated Decision runtime accepted complete packed inputs of at most 1,024 tokens per question without truncation. A separately distributed runtime must enforce its own supported input limit.
14
 
15
  [Main SVG](assets/architecture.svg) · [Residual layers](assets/residual-layers.svg) · [Attention and GEGLU](assets/attention-geglu.svg) · [PDF atlas](assets/architecture-atlas.pdf)
BATCH_CAPACITY.json DELETED
@@ -1,1015 +0,0 @@
1
- {
2
- "complete_input_tokens": 1024,
3
- "default_entry": "predict_1k unchanged, integer1..8/default8",
4
- "final_guard_separately_timed": false,
5
- "limitations": [
6
- "225 technical row-occurrences/model include fixed repetition; not independent quality support.",
7
- "The optional padding-protected entry was not separately timed; timing benefits refer to the earlier paired B8/B32 experiment.",
8
- "No claim of Studio speedup, universal batch invariance or reduced peak memory.",
9
- "Published weights/runtime and Studio were not changed; this result does not automatically publish."
10
- ],
11
- "model": "Decision-1.0-Lex",
12
- "native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
13
- "optional_entry": "predict_auto_1k",
14
- "package_validation": {
15
- "buffers_unchanged": 66,
16
- "fallback_exact": true,
17
- "forwards": 46,
18
- "late_invalid_and_empty_calls": 0,
19
- "numerics": {
20
- "by_type": {
21
- "Choice": {
22
- "flips": 0,
23
- "rows": 116
24
- },
25
- "Noul": {
26
- "flips": 0,
27
- "rows": 52
28
- },
29
- "Score": {
30
- "flips": 0,
31
- "rows": 57
32
- }
33
- },
34
- "checks": {
35
- "logits": true,
36
- "no_hard_flips": true,
37
- "probabilities": true,
38
- "score_normalized": true
39
- },
40
- "full_rank_changes": 0,
41
- "hard_flips": 0,
42
- "max_logit_error": 1.3217329978942871e-05,
43
- "max_probability_error": 1.4901161193847656e-06,
44
- "rows": 225,
45
- "strict_fp32_contract_pass": true
46
- },
47
- "parameters_unchanged": 489,
48
- "row_occurrences": 225,
49
- "source_closure_sha256": "d2b92dbf73abb3f80905e6c7e3445b48aa69df9b5a912d6a1de09ae7af56835c",
50
- "source_complete_sha256": "1f93cd36e3ccd80950679712f05e193bf0cbe9d7fc31c2df30d0dcae0a7b6e10",
51
- "source_plan_sha256": "dd43fd06ff147d4a2abd0c6c11df3dae9b7c73552365ae81f3cbc913f00d3586",
52
- "source_result_sha256": "c60a00a2aa571f01dff8a71bfc658b23419d0b7141c642eb9ea43a1766f3de5b"
53
- },
54
- "performance": {
55
- "bootstrap": {
56
- "estimator": "median(cap32)/median(B8)",
57
- "interval": "percentile95",
58
- "order_strata": "five AB and five BA blocks resampled separately",
59
- "replicates": 2000,
60
- "seed": 20260922,
61
- "unit": "whole paired five-request-per-mode block"
62
- },
63
- "forwards": 1605,
64
- "numerics": {
65
- "by_type": {
66
- "Choice": {
67
- "flips": 0,
68
- "rows": 247
69
- },
70
- "Noul": {
71
- "flips": 0,
72
- "rows": 53
73
- },
74
- "Score": {
75
- "flips": 0,
76
- "rows": 57
77
- }
78
- },
79
- "checks": {
80
- "logits": true,
81
- "no_hard_flips": true,
82
- "probabilities": true,
83
- "score_normalized": true
84
- },
85
- "full_rank_changes": 0,
86
- "hard_flips": 0,
87
- "max_logit_error": 1.5974044799804688e-05,
88
- "max_probability_error": 1.9073486328125e-06,
89
- "rows": 357,
90
- "strict_fp32_contract_pass": true
91
- },
92
- "original_checks": {
93
- "all_p95_no_more_5pct_regression": true,
94
- "choice-l256-q16_p50_ratio_CI_below1": true,
95
- "choice-l256-q16_p50_reduction_15pct": false,
96
- "choice-l256-q32_p50_ratio_CI_below1": true,
97
- "choice-l256-q32_p50_reduction_15pct": true,
98
- "multicontext-choice-32_p50_ratio_CI_below1": true,
99
- "multicontext-choice-32_p50_reduction_15pct": true,
100
- "q1_q8_p50_no_more_5pct_regression": true
101
- },
102
- "original_utility_pass": false,
103
- "points": {
104
- "choice-l1024-q1": {
105
- "blocks": 10,
106
- "clocks": {
107
- "device_forward_ms": {
108
- "p50": {
109
- "b8_ms": 10.556464672088623,
110
- "cap32_ms": 10.566385269165039,
111
- "difference_ms": 0.009920597076416016,
112
- "difference_ms_percentile95": [
113
- -0.012168836593627929,
114
- 0.029139995574951172
115
- ],
116
- "ratio": 1.00093976509983,
117
- "ratio_percentile95": [
118
- 0.9988491046005276,
119
- 1.0027634771344436
120
- ]
121
- },
122
- "p95": {
123
- "b8_ms": 10.661027860641479,
124
- "cap32_ms": 10.661649703979492,
125
- "difference_ms": 0.0006218433380134059,
126
- "difference_ms_percentile95": [
127
- -0.03483018875122035,
128
- 0.0834893226623521
129
- ],
130
- "ratio": 1.0000583286476823,
131
- "ratio_percentile95": [
132
- 0.9967408769534367,
133
- 1.007839916995741
134
- ]
135
- }
136
- },
137
- "predict_wall_ms": {
138
- "p50": {
139
- "b8_ms": 18.094859493430704,
140
- "cap32_ms": 18.080359499435872,
141
- "difference_ms": -0.014499993994832039,
142
- "difference_ms_percentile95": [
143
- -0.040014972910284996,
144
- 0.004804728087037776
145
- ],
146
- "ratio": 0.9991986677763319,
147
- "ratio_percentile95": [
148
- 0.99779205443173,
149
- 1.00026553617198
150
- ]
151
- },
152
- "p95": {
153
- "b8_ms": 18.238159001339227,
154
- "cap32_ms": 18.20715542708058,
155
- "difference_ms": -0.03100357425864786,
156
- "difference_ms_percentile95": [
157
- -0.14951042539905757,
158
- 0.26860943471547216
159
- ],
160
- "ratio": 0.998300071062196,
161
- "ratio_percentile95": [
162
- 0.9918458693139535,
163
- 1.0147638224804787
164
- ]
165
- }
166
- },
167
- "synchronized_forward_wall_ms": {
168
- "p50": {
169
- "b8_ms": 10.582302988041192,
170
- "cap32_ms": 10.592097998596728,
171
- "difference_ms": 0.009795010555535555,
172
- "difference_ms_percentile95": [
173
- -0.011939986143261194,
174
- 0.02943403160315936
175
- ],
176
- "ratio": 1.0009256029208957,
177
- "ratio_percentile95": [
178
- 0.9988732997077118,
179
- 1.0027851328214283
180
- ]
181
- },
182
- "p95": {
183
- "b8_ms": 10.686860964051448,
184
- "cap32_ms": 10.687752513331361,
185
- "difference_ms": 0.0008915492799133062,
186
- "difference_ms_percentile95": [
187
- -0.034992871223948896,
188
- 0.0834054546430707
189
- ],
190
- "ratio": 1.0000834248038701,
191
- "ratio_percentile95": [
192
- 0.9967335955665599,
193
- 1.0078132144378036
194
- ]
195
- }
196
- }
197
- },
198
- "memory": {
199
- "b8": {
200
- "peak_allocated_bytes": 2481425408,
201
- "peak_incremental_allocated_bytes": 53504512,
202
- "peak_reserved_bytes_shared_cache": 2518679552
203
- },
204
- "cap32": {
205
- "peak_allocated_bytes": 2481425408,
206
- "peak_incremental_allocated_bytes": 53504512,
207
- "peak_reserved_bytes_shared_cache": 2518679552
208
- }
209
- },
210
- "order_strata": {
211
- "AB": 5,
212
- "BA": 5
213
- },
214
- "replicates": 2000,
215
- "timed_requests_per_mode": 50
216
- },
217
- "choice-l1024-q32": {
218
- "blocks": 10,
219
- "clocks": {
220
- "device_forward_ms": {
221
- "p50": {
222
- "b8_ms": 207.20293426513672,
223
- "cap32_ms": 185.14539337158203,
224
- "difference_ms": -22.057540893554688,
225
- "difference_ms_percentile95": [
226
- -22.2838134765625,
227
- -21.868221282958984
228
- ],
229
- "ratio": 0.8935461943539377,
230
- "ratio_percentile95": [
231
- 0.8923934919811013,
232
- 0.8943251695490113
233
- ]
234
- },
235
- "p95": {
236
- "b8_ms": 207.52946338653564,
237
- "cap32_ms": 185.77925491333008,
238
- "difference_ms": -21.75020847320556,
239
- "difference_ms_percentile95": [
240
- -22.083112335205072,
241
- -20.28153457641602
242
- ],
243
- "ratio": 0.8951945997533153,
244
- "ratio_percentile95": [
245
- 0.8936076948402836,
246
- 0.9022578623509261
247
- ]
248
- }
249
- },
250
- "predict_wall_ms": {
251
- "p50": {
252
- "b8_ms": 242.37409501802176,
253
- "cap32_ms": 217.62683399720117,
254
- "difference_ms": -24.747261020820588,
255
- "difference_ms_percentile95": [
256
- -25.16183662446565,
257
- -24.575807009387063
258
- ],
259
- "ratio": 0.8978964273431098,
260
- "ratio_percentile95": [
261
- 0.8961426985350305,
262
- 0.8985948091719478
263
- ]
264
- },
265
- "p95": {
266
- "b8_ms": 243.72325239528436,
267
- "cap32_ms": 218.37091989000328,
268
- "difference_ms": -25.352332505281083,
269
- "difference_ms_percentile95": [
270
- -26.005168532719836,
271
- -23.27408672135789
272
- ],
273
- "ratio": 0.8959790161335808,
274
- "ratio_percentile95": [
275
- 0.893447528320472,
276
- 0.9042453228782044
277
- ]
278
- }
279
- },
280
- "synchronized_forward_wall_ms": {
281
- "p50": {
282
- "b8_ms": 207.32908698846586,
283
- "cap32_ms": 185.17211652942933,
284
- "difference_ms": -22.15697045903653,
285
- "difference_ms_percentile95": [
286
- -22.36067791818641,
287
- -21.93638199241832
288
- ],
289
- "ratio": 0.8931313942444112,
290
- "ratio_percentile95": [
291
- 0.8921173649815071,
292
- 0.8940271149855218
293
- ]
294
- },
295
- "p95": {
296
- "b8_ms": 207.64006946410518,
297
- "cap32_ms": 185.8059855090687,
298
- "difference_ms": -21.834083955036476,
299
- "difference_ms_percentile95": [
300
- -22.15991159901023,
301
- -20.358516700798646
302
- ],
303
- "ratio": 0.8948464811662427,
304
- "ratio_percentile95": [
305
- 0.8932978401930624,
306
- 0.9019359263090917
307
- ]
308
- }
309
- }
310
- },
311
- "memory": {
312
- "b8": {
313
- "peak_allocated_bytes": 2847486976,
314
- "peak_incremental_allocated_bytes": 419566080,
315
- "peak_reserved_bytes_shared_cache": 5786042368
316
- },
317
- "cap32": {
318
- "peak_allocated_bytes": 4106146304,
319
- "peak_incremental_allocated_bytes": 1678225408,
320
- "peak_reserved_bytes_shared_cache": 5786042368
321
- }
322
- },
323
- "order_strata": {
324
- "AB": 5,
325
- "BA": 5
326
- },
327
- "replicates": 2000,
328
- "timed_requests_per_mode": 50
329
- },
330
- "choice-l1024-q8": {
331
- "blocks": 10,
332
- "clocks": {
333
- "device_forward_ms": {
334
- "p50": {
335
- "b8_ms": 52.41438102722168,
336
- "cap32_ms": 52.384565353393555,
337
- "difference_ms": -0.029815673828125,
338
- "difference_ms_percentile95": [
339
- -0.058559417724609375,
340
- 0.017009687423706003
341
- ],
342
- "ratio": 0.9994311547089979,
343
- "ratio_percentile95": [
344
- 0.9988832372979973,
345
- 1.0003247540003477
346
- ]
347
- },
348
- "p95": {
349
- "b8_ms": 52.496505928039554,
350
- "cap32_ms": 52.48605175018311,
351
- "difference_ms": -0.010454177856445312,
352
- "difference_ms_percentile95": [
353
- -0.0377529144287152,
354
- 0.2926578712463391
355
- ],
356
- "ratio": 0.9998008595491903,
357
- "ratio_percentile95": [
358
- 0.999281028941322,
359
- 1.0055735259439396
360
- ]
361
- }
362
- },
363
- "predict_wall_ms": {
364
- "p50": {
365
- "b8_ms": 65.74378252844326,
366
- "cap32_ms": 65.72207299177535,
367
- "difference_ms": -0.0217095366679132,
368
- "difference_ms_percentile95": [
369
- -0.06555735380970873,
370
- 0.025685253785923123
371
- ],
372
- "ratio": 0.9996697857069827,
373
- "ratio_percentile95": [
374
- 0.9990025610937415,
375
- 1.0003905551675811
376
- ]
377
- },
378
- "p95": {
379
- "b8_ms": 65.88881341158412,
380
- "cap32_ms": 66.52213780616876,
381
- "difference_ms": 0.6333243945846334,
382
- "difference_ms_percentile95": [
383
- 0.0010406914952909609,
384
- 1.8839879310689867
385
- ],
386
- "ratio": 1.0096120169994334,
387
- "ratio_percentile95": [
388
- 1.0000158055055854,
389
- 1.0285920667545798
390
- ]
391
- }
392
- },
393
- "synchronized_forward_wall_ms": {
394
- "p50": {
395
- "b8_ms": 52.44034499628469,
396
- "cap32_ms": 52.414220495847985,
397
- "difference_ms": -0.02612450043670833,
398
- "difference_ms_percentile95": [
399
- -0.0563944922760129,
400
- 0.019360042642802
401
- ],
402
- "ratio": 0.9995018243980173,
403
- "ratio_percentile95": [
404
- 0.9989250543350029,
405
- 1.0003694742542426
406
- ]
407
- },
408
- "p95": {
409
- "b8_ms": 52.52235445950646,
410
- "cap32_ms": 52.51286399143282,
411
- "difference_ms": -0.009490468073636293,
412
- "difference_ms_percentile95": [
413
- -0.03923154145013541,
414
- 0.29656406215508296
415
- ],
416
- "ratio": 0.999819306118865,
417
- "ratio_percentile95": [
418
- 0.9992532410156931,
419
- 1.0056450304951374
420
- ]
421
- }
422
- }
423
- },
424
- "memory": {
425
- "b8": {
426
- "peak_allocated_bytes": 2847485952,
427
- "peak_incremental_allocated_bytes": 419565056,
428
- "peak_reserved_bytes_shared_cache": 3168796672
429
- },
430
- "cap32": {
431
- "peak_allocated_bytes": 2847485952,
432
- "peak_incremental_allocated_bytes": 419565056,
433
- "peak_reserved_bytes_shared_cache": 3168796672
434
- }
435
- },
436
- "order_strata": {
437
- "AB": 5,
438
- "BA": 5
439
- },
440
- "replicates": 2000,
441
- "timed_requests_per_mode": 50
442
- },
443
- "choice-l256-q1": {
444
- "blocks": 10,
445
- "clocks": {
446
- "device_forward_ms": {
447
- "p50": {
448
- "b8_ms": 6.985419034957886,
449
- "cap32_ms": 6.974940061569214,
450
- "difference_ms": -0.010478973388671875,
451
- "difference_ms_percentile95": [
452
- -0.022879636287689208,
453
- 0.010601043701171875
454
- ],
455
- "ratio": 0.998499879057186,
456
- "ratio_percentile95": [
457
- 0.9967288947675313,
458
- 1.0015206567387591
459
- ]
460
- },
461
- "p95": {
462
- "b8_ms": 7.041934943199157,
463
- "cap32_ms": 7.064635586738587,
464
- "difference_ms": 0.022700643539429244,
465
- "difference_ms_percentile95": [
466
- -0.024966001510620117,
467
- 0.06541033685207343
468
- ],
469
- "ratio": 1.003223637213711,
470
- "ratio_percentile95": [
471
- 0.996459745978026,
472
- 1.0093022538220333
473
- ]
474
- }
475
- },
476
- "predict_wall_ms": {
477
- "p50": {
478
- "b8_ms": 13.920774013968185,
479
- "cap32_ms": 13.921903999289498,
480
- "difference_ms": 0.0011299853213131428,
481
- "difference_ms_percentile95": [
482
- -0.034331005736021325,
483
- 0.02732866196311077
484
- ],
485
- "ratio": 1.0000811725928587,
486
- "ratio_percentile95": [
487
- 0.9975367080592278,
488
- 1.0019657993273317
489
- ]
490
- },
491
- "p95": {
492
- "b8_ms": 14.007961520110257,
493
- "cap32_ms": 14.045515967882238,
494
- "difference_ms": 0.03755444777198136,
495
- "difference_ms_percentile95": [
496
- -0.008158999844454229,
497
- 0.12227892875671387
498
- ],
499
- "ratio": 1.0026809359604585,
500
- "ratio_percentile95": [
501
- 0.9994190895790983,
502
- 1.008731703172336
503
- ]
504
- }
505
- },
506
- "synchronized_forward_wall_ms": {
507
- "p50": {
508
- "b8_ms": 7.020359509624541,
509
- "cap32_ms": 7.008063985267654,
510
- "difference_ms": -0.012295524356886744,
511
- "difference_ms_percentile95": [
512
- -0.02304499503225088,
513
- 0.007555005140602589
514
- ],
515
- "ratio": 0.9982485904973911,
516
- "ratio_percentile95": [
517
- 0.9967212883959398,
518
- 1.0010784238669923
519
- ]
520
- },
521
- "p95": {
522
- "b8_ms": 7.073595060501248,
523
- "cap32_ms": 7.098784536356106,
524
- "difference_ms": 0.025189475854858756,
525
- "difference_ms_percentile95": [
526
- -0.021360546816140413,
527
- 0.06566304175066759
528
- ],
529
- "ratio": 1.0035610570918196,
530
- "ratio_percentile95": [
531
- 0.9969902183560572,
532
- 1.0092935466259783
533
- ]
534
- }
535
- }
536
- },
537
- "memory": {
538
- "b8": {
539
- "peak_allocated_bytes": 2438808064,
540
- "peak_incremental_allocated_bytes": 10887168,
541
- "peak_reserved_bytes_shared_cache": 2472542208
542
- },
543
- "cap32": {
544
- "peak_allocated_bytes": 2438808064,
545
- "peak_incremental_allocated_bytes": 10887168,
546
- "peak_reserved_bytes_shared_cache": 2472542208
547
- }
548
- },
549
- "order_strata": {
550
- "AB": 5,
551
- "BA": 5
552
- },
553
- "replicates": 2000,
554
- "timed_requests_per_mode": 50
555
- },
556
- "choice-l256-q16": {
557
- "blocks": 10,
558
- "clocks": {
559
- "device_forward_ms": {
560
- "p50": {
561
- "b8_ms": 26.544885635375977,
562
- "cap32_ms": 21.927971839904785,
563
- "difference_ms": -4.616913795471191,
564
- "difference_ms_percentile95": [
565
- -4.648672580718994,
566
- -4.596253395080566
567
- ],
568
- "ratio": 0.8260714376814531,
569
- "ratio_percentile95": [
570
- 0.8248601346258421,
571
- 0.8267891393024169
572
- ]
573
- },
574
- "p95": {
575
- "b8_ms": 26.64890079498291,
576
- "cap32_ms": 21.992683506011964,
577
- "difference_ms": -4.6562172889709466,
578
- "difference_ms_percentile95": [
579
- -5.2368529319763155,
580
- -4.580370044708253
581
- ],
582
- "ratio": 0.8252754466387764,
583
- "ratio_percentile95": [
584
- 0.8076432393515695,
585
- 0.8278200215346474
586
- ]
587
- }
588
- },
589
- "predict_wall_ms": {
590
- "p50": {
591
- "b8_ms": 38.862298999447376,
592
- "cap32_ms": 33.38937502121553,
593
- "difference_ms": -5.472923978231847,
594
- "difference_ms_percentile95": [
595
- -5.532156530534849,
596
- -5.428209843375953
597
- ],
598
- "ratio": 0.8591713789678354,
599
- "ratio_percentile95": [
600
- 0.8578086140267409,
601
- 0.8603004537961789
602
- ]
603
- },
604
- "p95": {
605
- "b8_ms": 39.418455920531414,
606
- "cap32_ms": 33.553791453596205,
607
- "difference_ms": -5.86466446693521,
608
- "difference_ms_percentile95": [
609
- -118.63445058697778,
610
- -5.444934926345013
611
- ],
612
- "ratio": 0.8512203400671371,
613
- "ratio_percentile95": [
614
- 0.22047959173207704,
615
- 0.8605261779182554
616
- ]
617
- }
618
- },
619
- "synchronized_forward_wall_ms": {
620
- "p50": {
621
- "b8_ms": 26.595044997520745,
622
- "cap32_ms": 21.95358253084123,
623
- "difference_ms": -4.6414624666795135,
624
- "difference_ms_percentile95": [
625
- -4.673314448882593,
626
- -4.62049143970944
627
- ],
628
- "ratio": 0.8254764198702351,
629
- "ratio_percentile95": [
630
- 0.8242482464394296,
631
- 0.8262020510690964
632
- ]
633
- },
634
- "p95": {
635
- "b8_ms": 26.699701527832076,
636
- "cap32_ms": 22.01862498477567,
637
- "difference_ms": -4.681076543056406,
638
- "difference_ms_percentile95": [
639
- -5.2700117637868935,
640
- -4.603384018992074
641
- ],
642
- "ratio": 0.8246768212679532,
643
- "ratio_percentile95": [
644
- 0.8068452917646015,
645
- 0.8272821360169423
646
- ]
647
- }
648
- }
649
- },
650
- "memory": {
651
- "b8": {
652
- "peak_allocated_bytes": 2516791808,
653
- "peak_incremental_allocated_bytes": 88870912,
654
- "peak_reserved_bytes_shared_cache": 2791309312
655
- },
656
- "cap32": {
657
- "peak_allocated_bytes": 2604008448,
658
- "peak_incremental_allocated_bytes": 176087552,
659
- "peak_reserved_bytes_shared_cache": 2791309312
660
- }
661
- },
662
- "order_strata": {
663
- "AB": 5,
664
- "BA": 5
665
- },
666
- "replicates": 2000,
667
- "timed_requests_per_mode": 50
668
- },
669
- "choice-l256-q32": {
670
- "blocks": 10,
671
- "clocks": {
672
- "device_forward_ms": {
673
- "p50": {
674
- "b8_ms": 52.45026874542236,
675
- "cap32_ms": 38.73085975646973,
676
- "difference_ms": -13.719408988952637,
677
- "difference_ms_percentile95": [
678
- -13.757874083518981,
679
- -13.673646450042725
680
- ],
681
- "ratio": 0.7384301488417062,
682
- "ratio_percentile95": [
683
- 0.737865882587114,
684
- 0.739298436956481
685
- ]
686
- },
687
- "p95": {
688
- "b8_ms": 52.81323375701904,
689
- "cap32_ms": 43.84300689697263,
690
- "difference_ms": -8.97022686004641,
691
- "difference_ms_percentile95": [
692
- -25.975542593002217,
693
- -5.001253128051758
694
- ],
695
- "ratio": 0.8301519103845021,
696
- "ratio_percentile95": [
697
- 0.7010933474185724,
698
- 0.9055233060373526
699
- ]
700
- }
701
- },
702
- "predict_wall_ms": {
703
- "p50": {
704
- "b8_ms": 71.03448649286292,
705
- "cap32_ms": 54.82690001372248,
706
- "difference_ms": -16.207586479140446,
707
- "difference_ms_percentile95": [
708
- -16.303276085091056,
709
- -16.163301013875753
710
- ],
711
- "ratio": 0.771834959618257,
712
- "ratio_percentile95": [
713
- 0.7706202855890361,
714
- 0.772370733763932
715
- ]
716
- },
717
- "p95": {
718
- "b8_ms": 71.4557929430157,
719
- "cap32_ms": 60.150451309164026,
720
- "difference_ms": -11.305341633851668,
721
- "difference_ms_percentile95": [
722
- -28.958630622946544,
723
- -8.23459104867652
724
- ],
725
- "ratio": 0.8417855128573353,
726
- "ratio_percentile95": [
727
- 0.728043189474086,
728
- 0.8878310636443568
729
- ]
730
- }
731
- },
732
- "synchronized_forward_wall_ms": {
733
- "p50": {
734
- "b8_ms": 52.54925854387693,
735
- "cap32_ms": 38.762504496844485,
736
- "difference_ms": -13.786754047032446,
737
- "difference_ms_percentile95": [
738
- -13.831284875050187,
739
- -13.745294068939984
740
- ],
741
- "ratio": 0.7376413211326103,
742
- "ratio_percentile95": [
743
- 0.7369902659433524,
744
- 0.738419261821391
745
- ]
746
- },
747
- "p95": {
748
- "b8_ms": 52.91530743124895,
749
- "cap32_ms": 43.87024541210846,
750
- "difference_ms": -9.045062019140488,
751
- "difference_ms_percentile95": [
752
- -26.08439096948122,
753
- -5.122330039739609
754
- ],
755
- "ratio": 0.8290653034399842,
756
- "ratio_percentile95": [
757
- 0.7003076207316908,
758
- 0.9035067818750958
759
- ]
760
- }
761
- }
762
- },
763
- "memory": {
764
- "b8": {
765
- "peak_allocated_bytes": 2516447744,
766
- "peak_incremental_allocated_bytes": 88526848,
767
- "peak_reserved_bytes_shared_cache": 3160408064
768
- },
769
- "cap32": {
770
- "peak_allocated_bytes": 2776091136,
771
- "peak_incremental_allocated_bytes": 348170240,
772
- "peak_reserved_bytes_shared_cache": 3160408064
773
- }
774
- },
775
- "order_strata": {
776
- "AB": 5,
777
- "BA": 5
778
- },
779
- "replicates": 2000,
780
- "timed_requests_per_mode": 50
781
- },
782
- "choice-l256-q8": {
783
- "blocks": 10,
784
- "clocks": {
785
- "device_forward_ms": {
786
- "p50": {
787
- "b8_ms": 13.475781917572021,
788
- "cap32_ms": 13.454821586608887,
789
- "difference_ms": -0.020960330963134766,
790
- "difference_ms_percentile95": [
791
- -0.03996642827987671,
792
- 0.0022802352905273438
793
- ],
794
- "ratio": 0.9984445925964561,
795
- "ratio_percentile95": [
796
- 0.9970365481188822,
797
- 1.0001694271325683
798
- ]
799
- },
800
- "p95": {
801
- "b8_ms": 13.56228952407837,
802
- "cap32_ms": 13.560133504867554,
803
- "difference_ms": -0.0021560192108154297,
804
- "difference_ms_percentile95": [
805
- -0.0447998046875,
806
- 0.044429731369017844
807
- ],
808
- "ratio": 0.9998410283745243,
809
- "ratio_percentile95": [
810
- 0.9966977480632591,
811
- 1.0032788231911733
812
- ]
813
- }
814
- },
815
- "predict_wall_ms": {
816
- "p50": {
817
- "b8_ms": 22.647707461146638,
818
- "cap32_ms": 22.604158468311653,
819
- "difference_ms": -0.043548992834985256,
820
- "difference_ms_percentile95": [
821
- -0.07484605885110795,
822
- -0.013918994227424264
823
- ],
824
- "ratio": 0.9980771125329265,
825
- "ratio_percentile95": [
826
- 0.9966953876066275,
827
- 0.999385067302049
828
- ]
829
- },
830
- "p95": {
831
- "b8_ms": 22.747458069352433,
832
- "cap32_ms": 22.772549570072442,
833
- "difference_ms": 0.025091500720009208,
834
- "difference_ms_percentile95": [
835
- -0.08666022906254511,
836
- 0.27172756381332874
837
- ],
838
- "ratio": 1.0011030463554877,
839
- "ratio_percentile95": [
840
- 0.9962034198121077,
841
- 1.011953229139044
842
- ]
843
- }
844
- },
845
- "synchronized_forward_wall_ms": {
846
- "p50": {
847
- "b8_ms": 13.501687004463747,
848
- "cap32_ms": 13.480431021889672,
849
- "difference_ms": -0.02125598257407546,
850
- "difference_ms_percentile95": [
851
- -0.03631441213656217,
852
- 0.00041483872337264694
853
- ],
854
- "ratio": 0.9984256795045651,
855
- "ratio_percentile95": [
856
- 0.9973128364072793,
857
- 1.000030753064779
858
- ]
859
- },
860
- "p95": {
861
- "b8_ms": 13.590919968555681,
862
- "cap32_ms": 13.590195949655026,
863
- "difference_ms": -0.00072401890065521,
864
- "difference_ms_percentile95": [
865
- -0.04874391604971606,
866
- 0.04526954144239426
867
- ],
868
- "ratio": 0.9999467277489434,
869
- "ratio_percentile95": [
870
- 0.9964148455762563,
871
- 1.0033364560799924
872
- ]
873
- }
874
- }
875
- },
876
- "memory": {
877
- "b8": {
878
- "peak_allocated_bytes": 2516446720,
879
- "peak_incremental_allocated_bytes": 88525824,
880
- "peak_reserved_bytes_shared_cache": 2594177024
881
- },
882
- "cap32": {
883
- "peak_allocated_bytes": 2516446720,
884
- "peak_incremental_allocated_bytes": 88525824,
885
- "peak_reserved_bytes_shared_cache": 2594177024
886
- }
887
- },
888
- "order_strata": {
889
- "AB": 5,
890
- "BA": 5
891
- },
892
- "replicates": 2000,
893
- "timed_requests_per_mode": 50
894
- },
895
- "multicontext-choice-32": {
896
- "blocks": 10,
897
- "clocks": {
898
- "device_forward_ms": {
899
- "p50": {
900
- "b8_ms": 28.839591026306152,
901
- "cap32_ms": 15.876487731933594,
902
- "difference_ms": -12.963103294372559,
903
- "difference_ms_percentile95": [
904
- -13.00510287284851,
905
- -12.939247131347656
906
- ],
907
- "ratio": 0.5505101552047597,
908
- "ratio_percentile95": [
909
- 0.5495901472067748,
910
- 0.5512636790172465
911
- ]
912
- },
913
- "p95": {
914
- "b8_ms": 28.963391041755678,
915
- "cap32_ms": 15.983575963974,
916
- "difference_ms": -12.979815077781678,
917
- "difference_ms_percentile95": [
918
- -13.066112565994263,
919
- -12.911408408880234
920
- ],
921
- "ratio": 0.5518544406948394,
922
- "ratio_percentile95": [
923
- 0.5498390139393741,
924
- 0.5537668232778035
925
- ]
926
- }
927
- },
928
- "predict_wall_ms": {
929
- "p50": {
930
- "b8_ms": 44.08348348806612,
931
- "cap32_ms": 28.755783481756225,
932
- "difference_ms": -15.327700006309897,
933
- "difference_ms_percentile95": [
934
- -15.348734974395484,
935
- -15.245991060510278
936
- ],
937
- "ratio": 0.6523028854909056,
938
- "ratio_percentile95": [
939
- 0.6518971153285599,
940
- 0.6537317276429896
941
- ]
942
- },
943
- "p95": {
944
- "b8_ms": 44.253829002263956,
945
- "cap32_ms": 28.868743509519845,
946
- "difference_ms": -15.38508549274411,
947
- "difference_ms_percentile95": [
948
- -15.46525440440746,
949
- -15.249868051614612
950
- ],
951
- "ratio": 0.6523445351597252,
952
- "ratio_percentile95": [
953
- 0.6507656214790302,
954
- 0.655196504823658
955
- ]
956
- }
957
- },
958
- "synchronized_forward_wall_ms": {
959
- "p50": {
960
- "b8_ms": 28.93799147568643,
961
- "cap32_ms": 15.90245749684982,
962
- "difference_ms": -13.035533978836611,
963
- "difference_ms_percentile95": [
964
- -13.076403469312936,
965
- -13.010836963076144
966
- ],
967
- "ratio": 0.5495356341579889,
968
- "ratio_percentile95": [
969
- 0.5486493944100644,
970
- 0.5503878064714475
971
- ]
972
- },
973
- "p95": {
974
- "b8_ms": 29.059543382027186,
975
- "cap32_ms": 16.009619031683542,
976
- "difference_ms": -13.049924350343645,
977
- "difference_ms_percentile95": [
978
- -13.136312982533127,
979
- -12.980406277420116
980
- ],
981
- "ratio": 0.5509246591116503,
982
- "ratio_percentile95": [
983
- 0.5489527496966714,
984
- 0.5528477239894252
985
- ]
986
- }
987
- }
988
- },
989
- "memory": {
990
- "b8": {
991
- "peak_allocated_bytes": 2465764864,
992
- "peak_incremental_allocated_bytes": 37843968,
993
- "peak_reserved_bytes_shared_cache": 2713714688
994
- },
995
- "cap32": {
996
- "peak_allocated_bytes": 2578972672,
997
- "peak_incremental_allocated_bytes": 151051776,
998
- "peak_reserved_bytes_shared_cache": 2713714688
999
- }
1000
- },
1001
- "order_strata": {
1002
- "AB": 5,
1003
- "BA": 5
1004
- },
1005
- "replicates": 2000,
1006
- "timed_requests_per_mode": 50
1007
- }
1008
- },
1009
- "source_plan_sha256": "6df07d17e4481acd200b886cbb85112ba6b773320f6c542cc5d7ddcab8480a35",
1010
- "source_public_aggregate_sha256": "0251e0ba6aeabafafbb759fefdeeec42b6e498f5fa51136adf3dfd78b5a57d88"
1011
- },
1012
- "precision": "fp32",
1013
- "studio_performance_claim": false,
1014
- "weights_unchanged": true
1015
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DISTRIBUTION_TERMS.md CHANGED
@@ -2,7 +2,7 @@
2
 
3
  This repository is public and ungated. No Hugging Face approval or access form is required.
4
 
5
- Decision's new code and model-weight contributions are provided under the included [Apache License 2.0](LICENSE). Retained third-party material remains subject to its original license and notices; the project grant does not replace those conditions.
6
 
7
  The inherited tokenizer is distributed with the [Gemma Terms of Use](LICENSES/gemma/GEMMA_TERMS.html), including the incorporated [Gemma Prohibited Use Policy](LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html). By accessing, using or distributing the covered tokenizer material, you agree to comply with those terms and restrictions. Those documents are incorporated into this distribution agreement for that material. Redistributors must pass on the applicable agreement, restrictions and required notices, and identify their modifications as required by the upstream terms.
8
 
 
2
 
3
  This repository is public and ungated. No Hugging Face approval or access form is required.
4
 
5
+ Decision's model-weight and documentation contributions are provided under the included [Apache License 2.0](LICENSE). Retained third-party material remains subject to its original license and notices; the project grant does not replace those conditions.
6
 
7
  The inherited tokenizer is distributed with the [Gemma Terms of Use](LICENSES/gemma/GEMMA_TERMS.html), including the incorporated [Gemma Prohibited Use Policy](LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html). By accessing, using or distributing the covered tokenizer material, you agree to comply with those terms and restrictions. Those documents are incorporated into this distribution agreement for that material. Redistributors must pass on the applicable agreement, restrictions and required notices, and identify their modifications as required by the upstream terms.
8
 
FINETUNING.md DELETED
@@ -1,32 +0,0 @@
1
- # Fine-tune on your data
2
-
3
- Already using System One requests? [Convert the same state/questions/criteria into training rows](SYSTEM_ONE.md#fine-tune-with-the-same-inputs) with separate hard or soft targets and shared source-component IDs.
4
-
5
- The included CLI trains the existing Choice, Noul and Score paths without adding new parameters. It supports hard or soft labels, deterministic scheduling, full-input admission, DEV checkpoint selection and optimizer/RNG resume. Token embedding, embedding normalization and type embedding remain frozen. The other 486 parameter tensors are trainable when their type is present.
6
-
7
- Provide your own TRAIN and DEV JSONL. The CLI rejects shared IDs, shared normalized full inputs and same-source components where supplied. It does not prove semantic independence; use an appropriate development split and do not train on a release/test set.
8
-
9
- | Type | Target |
10
- |---|---|
11
- | Choice | `{"choice_id":"candidate-id"}` or `{"probabilities":[0.2,0.8]}` |
12
- | Noul | `{"probability":0.8}`; hard labels are 0 or 1 |
13
- | Score | `{"probabilities":[0.1,0.2,0.7]}` in increasing-value level order; hard labels are one-hot |
14
-
15
- Every row retains `id`, full `state_text` and a complete typed `question`. Probability vectors sum to one. A scalar Score is not silently converted into a distribution.
16
-
17
- ```bash
18
- TRAIN=/data/train.jsonl DEV=/data/dev.jsonl OUTPUT=/runs/my-decision \
19
- ROCR_VISIBLE_DEVICES=0 bash examples/finetune.sh
20
- ```
21
-
22
- `examples/finetune.sh` is the editable example configuration; it uses actual supported flags, not a nonexistent `--config` option. Its four-epoch recipe uses logical batch 64 / microbatch 8, BF16 autocast with FP32 parameters/loss, AdamW, encoder LR 2.5e-5 / head LR 1e-4, clip 1, CE plus 0.1 ordinal RPS for Score, and a cosine floor of 1e-6 after 10% warmup. Logical tails use their actual row count. These are example defaults, not a guarantee of improvement.
23
-
24
- The default selector minimizes equal-type macro soft NLL at step 0 and epoch ends. The separate `--selection hard-accuracy` policy requires an explicit `hard_target_id` on every DEV row, uses row hard accuracy then soft NLL then the earlier step, and never derives hard gold from soft targets. Noul hard selection uses `p_yes >= .5`. Record the selector and cadence used in any model comparison; these defaults are starting points rather than a guarantee of task quality.
25
-
26
- Successful runs export the selected native bundle and report its manifest in `COMPLETE.json`. Selecting step 0 means no adopted adaptation. Immutable checkpoints include model, AdamW and RNG state; allow substantial disk space. For an interrupted run, use exactly the same original arguments and output plus `--resume <checkpoint> --resume-sha256 <hash>`. Read `LATEST.json` for the last completely saved boundary. `--stop-after-step N` is an optional clean pause; it does not restart or shorten the LR horizon.
27
-
28
- The included source underwent a bounded AMD continuous-versus-resume check and independent native reload. That interface evidence is separate from dataset quality or long training stability. Validate the quality of your selected model on an appropriately isolated evaluation set; no automatic release or adoption is performed.
29
-
30
- ## Lex checkpoint provenance
31
-
32
- The editable example above is a downstream fine-tuning recipe, not the exact Lex release recipe. Lex was refit from Kai on all 6,000 original typed-decisions TRAIN decisions using eight epochs, fixed final step 750 and blended original hard/soft cross-entropy. It used no final-refit DEV selector, no warmup and no RPS term. See [METHODS.md](METHODS.md) and [TRAINING_PROVENANCE.json](TRAINING_PROVENANCE.json). The public CLI does not provide a named `blend50` switch; explicitly form normalized blended targets if using that objective. Reusing Lex as an initializer is a new experiment.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
LICENSING_STATUS.md CHANGED
@@ -1,14 +1,14 @@
1
  # Decision-1.0-Lex licensing
2
 
3
- **Decision's model-weight contributions, original code and documentation are licensed under Apache 2.0.** The repository is public and ungated: no account approval or acceptance form is required to download it.
4
 
5
  | Material | License and attribution |
6
  |---|---|
7
  | Decision contributions | [Apache License 2.0](LICENSE) |
8
  | Upstream Vela/mmBERT contributions | Retained [MIT terms](LICENSES/Upstream-MIT.txt) and author/source attribution in [NOTICE](NOTICE) |
9
- | Adapted ModernBERT/Transformers code | [Apache 2.0](LICENSES/Transformers-Apache-2.0.txt), with original copyright and modification notices |
10
  | Inherited Gemma-origin tokenizer material | Retained [tokenizer terms](LICENSES/gemma/TOKENIZER_TERMS.md), agreement, use policy and required Notice |
11
 
12
  The Apache-2.0 project license does not replace third-party terms. In particular, the tokenizer inherited through Vela/mmBERT retains its upstream conditions; this does not describe Decision's independently trained encoder weights as Gemma weights. [Distribution terms](DISTRIBUTION_TERMS.md) preserve that component's requirements without a Hugging Face access gate.
13
 
14
- The license update changes documentation and access settings only. The native weights, tokenizer bytes and runtime are unchanged. Upstream sources and pinned revisions remain recorded in [NOTICE_SOURCES.json](NOTICE_SOURCES.json). Training/evaluation data, Laya weights and Laya SDK source are not redistributed; separately installed dependencies retain their own licenses.
 
1
  # Decision-1.0-Lex licensing
2
 
3
+ **Decision's model-weight contributions and documentation are licensed under Apache 2.0.** The repository is public and ungated: no account approval or acceptance form is required to download it.
4
 
5
  | Material | License and attribution |
6
  |---|---|
7
  | Decision contributions | [Apache License 2.0](LICENSE) |
8
  | Upstream Vela/mmBERT contributions | Retained [MIT terms](LICENSES/Upstream-MIT.txt) and author/source attribution in [NOTICE](NOTICE) |
9
+ | Historical adapted ModernBERT/Transformers runtime code (no longer bundled) | Retained [Apache 2.0 terms](LICENSES/Transformers-Apache-2.0.txt) and modification notices |
10
  | Inherited Gemma-origin tokenizer material | Retained [tokenizer terms](LICENSES/gemma/TOKENIZER_TERMS.md), agreement, use policy and required Notice |
11
 
12
  The Apache-2.0 project license does not replace third-party terms. In particular, the tokenizer inherited through Vela/mmBERT retains its upstream conditions; this does not describe Decision's independently trained encoder weights as Gemma weights. [Distribution terms](DISTRIBUTION_TERMS.md) preserve that component's requirements without a Hugging Face access gate.
13
 
14
+ This model-only package retains the original weight and tokenizer objects but no longer bundles executable runtime code. Upstream sources and pinned revisions remain recorded in [NOTICE_SOURCES.json](NOTICE_SOURCES.json). Training/evaluation data, Laya weights and Laya SDK source are not redistributed; separately installed dependencies retain their own licenses.
PACKAGE_MANIFEST.json DELETED
@@ -1,493 +0,0 @@
1
- {
2
- "bytes_excluding_this_manifest": 2327567235,
3
- "display_name": "Decision-1.0-Lex-0.6B",
4
- "file_count_excluding_this_manifest": 107,
5
- "files": {
6
- ".gitattributes": {
7
- "bytes": 753,
8
- "sha256": "d401c47d3da9ae93830e327b487cde895438378a1f90620a70b5c7b3e5d3ed92"
9
- },
10
- "ARCHITECTURE.md": {
11
- "bytes": 1330,
12
- "sha256": "82ceb4a1b1d6372ec15cce9b2902dc52477383f5bc52ba4206aad93efce097ff"
13
- },
14
- "BATCH_CAPACITY.json": {
15
- "bytes": 33186,
16
- "sha256": "ee74265a3232807f94115476847bf63194e4eb68d9d8532c25525ecd1f9c9e85"
17
- },
18
- "DISTRIBUTION_TERMS.md": {
19
- "bytes": 1251,
20
- "sha256": "54a5895871af33845ee939f2e8f37b7f821cee8e89ff25ec1fd78e9cf83e9c63"
21
- },
22
- "EVALUATION.md": {
23
- "bytes": 2813,
24
- "sha256": "00862d0a49c69c406b7dbfa890f95f09a34680568309d1cc0571733abe7b933c"
25
- },
26
- "FINETUNING.md": {
27
- "bytes": 3842,
28
- "sha256": "2874eedec3b19462552a35e0dc99af29c456b3a2f8096cc61344b92e4cc6db41"
29
- },
30
- "LICENSE": {
31
- "bytes": 11358,
32
- "sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30"
33
- },
34
- "LICENSES/Transformers-Apache-2.0.txt": {
35
- "bytes": 11418,
36
- "sha256": "77fd4710def9ec3c0f6225800e0235f15a425abd4a8b03559127fcd782612049"
37
- },
38
- "LICENSES/Upstream-MIT.txt": {
39
- "bytes": 1191,
40
- "sha256": "3ec44d2f046b27986e28e3b4705110a33b139e6392ef3868ff594a0453311038"
41
- },
42
- "LICENSES/gemma/GEMMA_PROHIBITED_USE_POLICY.html": {
43
- "bytes": 4636,
44
- "sha256": "7e50ae0c7a6386ab51d36aa868e1d20a4fb28cbcab38778a5dbdac6a445f5a96"
45
- },
46
- "LICENSES/gemma/GEMMA_TERMS.html": {
47
- "bytes": 13510,
48
- "sha256": "01ccbe2f6504a6a1364db0e5d6d91ed37309435229c322a29b20ee5daebebc1a"
49
- },
50
- "LICENSES/gemma/Notice": {
51
- "bytes": 1088,
52
- "sha256": "a7b8b1b625d4a140b4dd55ee4479db7d64b5cda047f2cbc1be9d21516b470921"
53
- },
54
- "LICENSES/gemma/TOKENIZER_TERMS.md": {
55
- "bytes": 1345,
56
- "sha256": "1ca30bf8cc8d1054f78ad3481ada9cf39a137cec9e383f18c3a9f0f56b8d4241"
57
- },
58
- "LICENSING_STATUS.md": {
59
- "bytes": 1530,
60
- "sha256": "f82c305be9e8502c872efae0903b965ac1128109848dd2167e3fc59601cd9da0"
61
- },
62
- "METHODS.md": {
63
- "bytes": 2543,
64
- "sha256": "a7dd6dec55dfe463cc0211c3dffa114b88bb95db90816b70c18b62b5d7be7ad1"
65
- },
66
- "MIXED_QUESTION_SCALING.json": {
67
- "bytes": 32763,
68
- "sha256": "ed4a9e7a9aac0385f8a4516694a6d9ea843aacefde6c7189b9980bac86c9d773"
69
- },
70
- "NOTICE": {
71
- "bytes": 4371,
72
- "sha256": "e3f71eb2a262084fa572ccbf9b65110a4feb9d8c625dc26e792beba282cdf032"
73
- },
74
- "NOTICE_SOURCES.json": {
75
- "bytes": 9279,
76
- "sha256": "2e5936d3ea196e00968cc37b7662a6afbe8cfd6fd11d04b9fecba554b1e1e37b"
77
- },
78
- "PERFORMANCE.md": {
79
- "bytes": 6351,
80
- "sha256": "0f5c84a90b2c82be6146b1b7032a9f836e5858d59e29bd6c94f016be36d8b3b6"
81
- },
82
- "README.md": {
83
- "bytes": 6230,
84
- "sha256": "6ff9eb7a6ac0c6b90b017f497d75bd3366c9c598ce3c5f813dfa0f192d49b1ec"
85
- },
86
- "RUNTIME_UPDATE.json": {
87
- "bytes": 1087,
88
- "sha256": "0a77c012fe1bd23c5f1a06c106c1a29ae5ee156248db218fbbc6d631d1c5aff4"
89
- },
90
- "SYSTEM_ONE.md": {
91
- "bytes": 6177,
92
- "sha256": "c59c5d1fbab6c18033b1339701aadddc34653f9fb93f79aee93534bb34d66648"
93
- },
94
- "SYSTEM_ONE_VALIDATION.json": {
95
- "bytes": 2583,
96
- "sha256": "24c28a2b8c448507f26e116a7f6a6d4e40fb8eaae0384e6bee421a105e415e0a"
97
- },
98
- "TECHNICAL_VALIDATION.json": {
99
- "bytes": 1547,
100
- "sha256": "fcfd2ce54f623325f268a702fc0c06044447e651dad665d6430e92d48f8e7353"
101
- },
102
- "TRAINING_ATTRIBUTION.md": {
103
- "bytes": 5816,
104
- "sha256": "305e5435ccb5abb6a6ad6deb30b762b508261a4179a0f9756605838d2c477961"
105
- },
106
- "TRAINING_PROVENANCE.json": {
107
- "bytes": 2310,
108
- "sha256": "5c95cf851a9480e6037bc190d0deee87da35576489362262ad4151bafbb96990"
109
- },
110
- "USAGE.md": {
111
- "bytes": 8308,
112
- "sha256": "f7b6fe124f34667480f324f125d92d72e35045ea043c8b45da7184b9c7a28dcb"
113
- },
114
- "VALIDATION.md": {
115
- "bytes": 2120,
116
- "sha256": "85e1f45ddf0d5e2650490eefdbd50f81ccccd7f414d78a8075206da49ef0043d"
117
- },
118
- "assets/architecture-atlas.pdf": {
119
- "bytes": 635270,
120
- "sha256": "ff64805a39d16be56d401166ef4d310da86305058bb2c532751754c8c34d49ee"
121
- },
122
- "assets/architecture.pdf": {
123
- "bytes": 173099,
124
- "sha256": "6aa6895f5708cda7d935991cf9c1179d9e54e22a32459968a052acf05f62a14a"
125
- },
126
- "assets/architecture.png": {
127
- "bytes": 432419,
128
- "sha256": "daeed79f92a66a2655eb0594f59af4e90424a84ba9519eeaf2ef5bb8b7949fa2"
129
- },
130
- "assets/architecture.svg": {
131
- "bytes": 23430,
132
- "sha256": "7c3ae44a25e7d9035e6a9635608f7e491a22d9be6991243b85d85f30874e563c"
133
- },
134
- "assets/attention-geglu.pdf": {
135
- "bytes": 172797,
136
- "sha256": "43b3485510971d29e66bc17b30b4bffe934ad66078598cc3e2efd63ae9b9bc7c"
137
- },
138
- "assets/attention-geglu.png": {
139
- "bytes": 265587,
140
- "sha256": "de7d88e3cf036d3b8cba38df47d5b3182faecb9233e675104cee7c75dd7fd8c5"
141
- },
142
- "assets/attention-geglu.svg": {
143
- "bytes": 10256,
144
- "sha256": "0ae02a36ebe97d0b6eed7b23e140c1fdc85e710c00015c518cae67ed5ec06cea"
145
- },
146
- "assets/candidate-readout.pdf": {
147
- "bytes": 156814,
148
- "sha256": "56b9e06a343e9044c38cfb8d1ba5ba121bdc7d763db5ce45b927aacdcb00c995"
149
- },
150
- "assets/candidate-readout.png": {
151
- "bytes": 309736,
152
- "sha256": "ad4be86a2aa0136cd7f528c317340c38988b932aab653f25a5a57a9a3cbad834"
153
- },
154
- "assets/candidate-readout.svg": {
155
- "bytes": 9825,
156
- "sha256": "1d3709818d5b08014ba63042e5287f84f1c97a735a85ec49108ec6e9e77151ee"
157
- },
158
- "assets/decision-lex-header.png": {
159
- "bytes": 2188963,
160
- "sha256": "9980c6b901aff89b69e9cc6de6de40408b95c7378f18e931401198f42d32e40b"
161
- },
162
- "assets/mixed-question-scaling.pdf": {
163
- "bytes": 27984,
164
- "sha256": "e1261c998cdcdf6b72f57b4df4cb56cbf47fa521210c4778a0577e2f3a47c1cf"
165
- },
166
- "assets/mixed-question-scaling.png": {
167
- "bytes": 121576,
168
- "sha256": "4efadb4888c43122d461ca432081f2b5bd67c350f9605940b96f156829d65a54"
169
- },
170
- "assets/mixed-question-scaling.svg": {
171
- "bytes": 14623,
172
- "sha256": "0fcea2c3e7fd1c649a1dd7acf206c0d41e4f6fcab5e51abd838d6bb18a2560ff"
173
- },
174
- "assets/residual-layers.pdf": {
175
- "bytes": 131144,
176
- "sha256": "2c5b50bb840ae115a1264b66d535a1048172e891a33d02d507c193525cc45630"
177
- },
178
- "assets/residual-layers.png": {
179
- "bytes": 309369,
180
- "sha256": "f3ee6441fe1016e2435256644675b08bff99cec72d521e49a78f6cbe5b2dc3bf"
181
- },
182
- "assets/residual-layers.svg": {
183
- "bytes": 10990,
184
- "sha256": "cdf5cfb263b50eef0171eafead3062ee75eb805961517f313490fdca03b390c3"
185
- },
186
- "decision_finetune/__init__.py": {
187
- "bytes": 81,
188
- "sha256": "51ab50ae9a598c7b09970161be3551f9dff0851a209ec692f2e296b1bd44566d"
189
- },
190
- "decision_finetune/__main__.py": {
191
- "bytes": 5410,
192
- "sha256": "d5cabebd24c823d5dc03a4b528dae8b8b70a63dd0692eb6516a59bd4bd64d6a7"
193
- },
194
- "decision_finetune/data.py": {
195
- "bytes": 6679,
196
- "sha256": "9fae31cb39512952a1dd78fb0b6735434da6f6b94606341fc0200ab9cdaf447e"
197
- },
198
- "decision_finetune/metrics.py": {
199
- "bytes": 4046,
200
- "sha256": "c3de487e9294d5278064b7e1768915bf68f593c9d7fb067addc52c4c9ac3b07e"
201
- },
202
- "decision_finetune/run.py": {
203
- "bytes": 16750,
204
- "sha256": "045ff99dd95c69e2ccdf4b1f8af84797c2929a27b16a9c1a3208bba10b41361b"
205
- },
206
- "decision_finetune/state.py": {
207
- "bytes": 5751,
208
- "sha256": "87dd485127fea7b80ad19300c0d6d95a92b7c2b6bd5b62c99bdc9201b681c9fe"
209
- },
210
- "decision_finetune/system_one.py": {
211
- "bytes": 1784,
212
- "sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
213
- },
214
- "decision_inference/__init__.py": {
215
- "bytes": 282,
216
- "sha256": "8199829090f1a3ca5fe59aec1a6dce2beb57026c301d2c87d8c5b1c960981e66"
217
- },
218
- "decision_inference/_auto.py": {
219
- "bytes": 1325,
220
- "sha256": "3602e3a06386daab5494305413a08adb581f23007ffb5ca6e5df62fa23d11bd6"
221
- },
222
- "decision_inference/_grouped.py": {
223
- "bytes": 1016,
224
- "sha256": "80551e1bebd10634abcbbfce1aecf6cafab93b7e178399c428dd97022c6f9d95"
225
- },
226
- "decision_inference/_request.py": {
227
- "bytes": 3026,
228
- "sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
229
- },
230
- "decision_inference/_system_one.py": {
231
- "bytes": 10728,
232
- "sha256": "0dc61f7558eed31c9ac186827b2050c0d391c69ea3684c02e8cd9c71400b1690"
233
- },
234
- "decision_inference/profile.py": {
235
- "bytes": 2071,
236
- "sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
237
- },
238
- "decision_runtime/__init__.py": {
239
- "bytes": 723,
240
- "sha256": "b0f9db94cfbcc73cab2f09e0269b8bd2eb87c0cc533b34ea5c9a8bfbb1ee48ca"
241
- },
242
- "decision_runtime/_compat.py": {
243
- "bytes": 1847,
244
- "sha256": "a0dd42a4e3b20eabbaf0d0d62ffc290ef4d9d20bc2fb2ba39d980771a8d0744c"
245
- },
246
- "decision_runtime/native.py": {
247
- "bytes": 6903,
248
- "sha256": "21c674a0d8a1406156504390974505ba5f981e7ecc9f0c3b29985876ef78511c"
249
- },
250
- "decision_runtime/training.py": {
251
- "bytes": 15998,
252
- "sha256": "166ea54629d5c90e0e0972b82c0b96e008c799f384400886ffd64312ab972177"
253
- },
254
- "examples/decisions.jsonl": {
255
- "bytes": 867,
256
- "sha256": "d4b89583ccc9aa35e74dd28c46e01884103ea7a0fa018de8c34d4f88a94287f1"
257
- },
258
- "examples/finetune.sh": {
259
- "bytes": 1242,
260
- "sha256": "9b72d89237808e0019521ab4e42635007c39d20ea200d98afa9fdca6e49ae400"
261
- },
262
- "examples/system-one.json": {
263
- "bytes": 692,
264
- "sha256": "0b8320ecbeded8c3229dccc52f5045f00e8d1968183bc9d1371d24e949925dcf"
265
- },
266
- "infer.py": {
267
- "bytes": 2373,
268
- "sha256": "6313bfd5e295d5dc3d9b6ded5f85cda75cf1da8ed4b2dbae32f619a76021e410"
269
- },
270
- "native/INVENTORY.json": {
271
- "bytes": 31007,
272
- "sha256": "06aed00ac4571582b992a995a793875804a0546bdae1461c4161b30fba2143a9"
273
- },
274
- "native/MANIFEST.json": {
275
- "bytes": 4326,
276
- "sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6"
277
- },
278
- "native/STATE_LAYOUT.json": {
279
- "bytes": 10971,
280
- "sha256": "a6b24716e2b21240d327ba8763a73f1b54862bc86d5e1f92bb9e670be968e494"
281
- },
282
- "native/__init__.py": {
283
- "bytes": 78,
284
- "sha256": "1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010"
285
- },
286
- "native/artifacts.py": {
287
- "bytes": 7958,
288
- "sha256": "c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3"
289
- },
290
- "native/choice_encoder.safetensors": {
291
- "bytes": 441337216,
292
- "sha256": "9516cc841c485c98b27b4f63d2ea8e604fe8064121173064b113da8a6bf57ef6"
293
- },
294
- "native/contract.py": {
295
- "bytes": 10886,
296
- "sha256": "51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726"
297
- },
298
- "native/decision_config.json": {
299
- "bytes": 14308,
300
- "sha256": "1fefb4ad7dede00c4633bb2ce5fc44fb04237207b9e9ce90fcf8ea8b01a1328d"
301
- },
302
- "native/decision_heads.safetensors": {
303
- "bytes": 177241884,
304
- "sha256": "bce3ee658a978a19c48c605b921cff994f892e7fc17abd6f8645f07428fbc36f"
305
- },
306
- "native/encoder/config.json": {
307
- "bytes": 2769,
308
- "sha256": "7aff915e9f159305e0bef3eb0206416f99b8560260b8969b35f1dcd54aaad1a5"
309
- },
310
- "native/encoder/model.safetensors": {
311
- "bytes": 1227771752,
312
- "sha256": "daaafd81c4ed767d203e24226d78e792800068a7dac344aa043844b82a6306c7"
313
- },
314
- "native/infer.py": {
315
- "bytes": 1095,
316
- "sha256": "9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315"
317
- },
318
- "native/model.py": {
319
- "bytes": 11098,
320
- "sha256": "8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b"
321
- },
322
- "native/packing.py": {
323
- "bytes": 6748,
324
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
325
- },
326
- "native/policy/__init__.py": {
327
- "bytes": 131,
328
- "sha256": "0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114"
329
- },
330
- "native/policy/artifacts.py": {
331
- "bytes": 7877,
332
- "sha256": "5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2"
333
- },
334
- "native/policy/contract.py": {
335
- "bytes": 4334,
336
- "sha256": "0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67"
337
- },
338
- "native/policy/infer.py": {
339
- "bytes": 1545,
340
- "sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
341
- },
342
- "native/policy/model.py": {
343
- "bytes": 7678,
344
- "sha256": "eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca"
345
- },
346
- "native/policy/packing.py": {
347
- "bytes": 6748,
348
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
349
- },
350
- "native/policy/reference/__init__.py": {
351
- "bytes": 192,
352
- "sha256": "c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28"
353
- },
354
- "native/policy/reference/artifacts.py": {
355
- "bytes": 9342,
356
- "sha256": "3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159"
357
- },
358
- "native/policy/reference/infer.py": {
359
- "bytes": 1545,
360
- "sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
361
- },
362
- "native/policy/reference/model.py": {
363
- "bytes": 5060,
364
- "sha256": "733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26"
365
- },
366
- "native/policy/reference/modernbert_sdpa_layout.py": {
367
- "bytes": 3365,
368
- "sha256": "0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8"
369
- },
370
- "native/policy/reference/packing.py": {
371
- "bytes": 6748,
372
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
373
- },
374
- "native/score_encoder.safetensors": {
375
- "bytes": 441337080,
376
- "sha256": "4f45795977846ef4c31b34e4f35bf95dabf5951d71358bab32c9a94814a84a53"
377
- },
378
- "native/tokenizer/special_tokens_map.json": {
379
- "bytes": 1051,
380
- "sha256": "6e3204c5a4004719c185007a745d1940471a6fb02521cfa0a00306dbb6bc5903"
381
- },
382
- "native/tokenizer/tokenizer.json": {
383
- "bytes": 34363188,
384
- "sha256": "609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f"
385
- },
386
- "native/tokenizer/tokenizer_config.json": {
387
- "bytes": 46470,
388
- "sha256": "74a259bb1a3811a7e3028adcd07a65765d866d5e66e0e883f0994ccfa67e8455"
389
- },
390
- "native/training_policy.py": {
391
- "bytes": 4761,
392
- "sha256": "6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5"
393
- },
394
- "requirements.txt": {
395
- "bytes": 296,
396
- "sha256": "001dd433614bcbacee15858a38f432b97f44b01ad404714f442738f4a8ce2428"
397
- },
398
- "source-metadata/ENCODER_README.md": {
399
- "bytes": 2046,
400
- "sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
401
- },
402
- "source-metadata/ENCODER_SOURCE.json": {
403
- "bytes": 425,
404
- "sha256": "9bb1a753d9979117520152860a599064d92a8eaf7f23a0e0d458c66007c8fdf9"
405
- },
406
- "source-metadata/MMBERT_LICENSE_SOURCE_README.md": {
407
- "bytes": 29650,
408
- "sha256": "35724415037ee3159aa733c16050620cc19a76219112a6ceeb3949572e0b37e9"
409
- },
410
- "source-metadata/MMBERT_TOKENIZER_LINEAGE.yaml": {
411
- "bytes": 2768,
412
- "sha256": "0ffa0f1c8388868a42cdbb8ec4b709db6eab4b3e33d8495f4e5f0c32bd15afd3"
413
- },
414
- "source-metadata/MODEL_LICENSE_METADATA.json": {
415
- "bytes": 1538,
416
- "sha256": "696395f97428b851f136d50d850978af8c2a3b278c82620f63246a9312b74a30"
417
- },
418
- "source-metadata/TRANSFORMERS_LICENSE_SOURCE.json": {
419
- "bytes": 1178,
420
- "sha256": "b60e54814b6c70791318475ebd4737c29fe5816f3141df392144dd31ac316d77"
421
- },
422
- "source-metadata/TRANSFORMERS_MODERNBERT_SOURCE_HEADER.txt": {
423
- "bytes": 1497,
424
- "sha256": "4577f1803f9dbe8b43f5da709f8750c798def2c33f7eb055d62baf286ed472c8"
425
- },
426
- "source-metadata/VELA_LICENSE_SOURCE_README.md": {
427
- "bytes": 2046,
428
- "sha256": "817df5015d443e6a9e6bbc870d8bd4f0f5c0d48e29b8f50f665319609838f7ef"
429
- },
430
- "systemone.py": {
431
- "bytes": 2368,
432
- "sha256": "8ce311c5651f1b0c1bc99d5a4ccaa6c7878b5c509256e4668a2191ad8c35e2e3"
433
- }
434
- },
435
- "hub_repository": "llm-semantic-router/Decision-1.0-Lex-0.6B",
436
- "language_scope": "English typed-decisions specialist",
437
- "manifest_excludes_itself": true,
438
- "name": "Decision-1.0-Lex",
439
- "native_file_count": 31,
440
- "native_manifest": "native/MANIFEST.json",
441
- "optional_batch_capacity": {
442
- "default_batch_limit": 8,
443
- "entry": "predict_auto_1k",
444
- "maximum_batch_size": 32,
445
- "padding_guard": true,
446
- "validation": "BATCH_CAPACITY.json",
447
- "weights_unchanged": true
448
- },
449
- "parameter_tensors": 489,
450
- "parameters": 571909635,
451
- "public_non_native_bytes_excluding_this_manifest": 5308024,
452
- "publication": {
453
- "download_access": "public_ungated",
454
- "new_contribution_license": "Apache-2.0",
455
- "retained_third_party_terms": true
456
- },
457
- "runtime_update": {
458
- "request_local_input_reuse": true,
459
- "validation": "RUNTIME_UPDATE.json",
460
- "weights_unchanged": true
461
- },
462
- "schema": "decision.public-distribution.v1",
463
- "status": "READY_FOR_ROOT_PUBLICATION",
464
- "subject_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
465
- "system_one_api": {
466
- "current_scheduling_validation": "MIXED_QUESTION_SCALING.json",
467
- "default_physical_batch_limit": 8,
468
- "default_scheduling": "stable typed groups",
469
- "entry": "decision_inference.SystemOne",
470
- "finetune_conversion": "decision_finetune.system_one.system_one_training_rows",
471
- "max_batch_decisions": 512,
472
- "max_questions": 128,
473
- "max_request_bytes": 2097152,
474
- "max_requests": 128,
475
- "optional_auto_capacity": 32,
476
- "request": "state/model/questions",
477
- "response": "model/answers/usage",
478
- "validation": "SYSTEM_ONE_VALIDATION.json",
479
- "weights_unchanged": true
480
- },
481
- "technical_validation": "TECHNICAL_VALIDATION.json",
482
- "training_provenance": "TRAINING_PROVENANCE.json",
483
- "typed_scheduling_update": {
484
- "complete_input_limit_unchanged": 1024,
485
- "default_schedule": "Stable group by decision type; physicalB8; restore original caller order",
486
- "entry": "decision_inference.SystemOne",
487
- "native_weights_unchanged": true,
488
- "new_chat_api": false,
489
- "public_api_unchanged": true,
490
- "subject_native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
491
- "validation": "MIXED_QUESTION_SCALING.json"
492
- }
493
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
PERFORMANCE.md DELETED
@@ -1,66 +0,0 @@
1
- # Lex: AMD SystemOne latency
2
-
3
- ## Mixed questions, faster decisions
4
-
5
- **128 mixed questions: 154.41 ms median, 56.6% lower than the previous default runtime.**
6
-
7
- ![Lex: latency as mixed question count increases](assets/mixed-question-scaling.png)
8
-
9
- | Questions | Previous p50 / p95 (ms) | Typed scheduling p50 / p95 (ms) |
10
- |---:|---:|---:|
11
- | 1 | 13.22 / 13.32 | 13.20 / 13.40 |
12
- | 8 | 28.07 / 28.32 | 28.02 / 28.31 |
13
- | 32 | 93.76 / 94.06 | 53.43 / 53.77 |
14
- | 64 | 181.78 / 184.53 | 87.39 / 88.09 |
15
- | 128 | 355.97 / 360.25 | 154.41 / 155.93 |
16
-
17
- The request repeats three fixed questions—Choice, Noul and Score—over one unchanged context. Only question count and bookkeeping IDs change. The default SystemOne path groups admitted rows by type so each encoder path processes fuller batches, then restores the original answer order. Every question still has its own contextual computation. Weights, precision, complete-input limit and request schema are unchanged.
18
-
19
- Measured using the exact release package on AMD ROCm, FP32, physical batch size eight: 10 warmup pairs and 30 alternating AB/BA measurement pairs per point, with GPU synchronization. Measurements include local request conversion, tokenization, model execution and answer assembly; they exclude transport and Studio. Models ran serially after our training and data jobs completed. Results describe this fixed workload, not a service latency guarantee.
20
-
21
- Both models passed seven source panels: 4,160 admitted decisions without an argmax change and 57 identical whole-request refusals before any forward. Exact-package checks additionally cover 512 decisions, six current Studio examples, caller IDs/order, optional auto batching and input limits. Small floating-point probability differences remain possible.
22
-
23
- [All samples, workload, checks and earlier concurrent measurements](MIXED_QUESTION_SCALING.json) · [SVG](assets/mixed-question-scaling.svg) · [PDF](assets/mixed-question-scaling.pdf)
24
-
25
- ## Earlier measurements
26
-
27
- ## Lex: AMD batch capacity
28
-
29
- ### Optional larger batches
30
-
31
- ```python
32
- from decision_inference import predict_1k, predict_auto_1k
33
-
34
- ## native and records use the same objects as the Python usage examples.
35
- outputs = predict_1k(native, records) # unchanged default: up to 8
36
- outputs = predict_auto_1k(native, records) # opt in: up to 32 with the padding guard
37
- ```
38
-
39
- The optional entry admits every complete input before the first forward. It uses consecutive batches up to 32 only when the request has one decision type and total padded tokens do not increase relative to B8. Mixed types or increased padding fall back to B8; requests of eight or fewer take the unchanged default path. Input order, candidates, full 1,024-token limit and FP32 weights remain unchanged. It does not share contextual activations across questions. Larger physical batches can change floating-point rounding and peak memory.
40
-
41
- #### Paired B8/B32 study
42
-
43
- These are synchronized resident **Python API measurements**, not Studio or network latency. This earlier study compared default B8 with a homogeneous cap32 policy before the final padding guard was added. All eight measured points satisfy that guard, but **the final `predict_auto_1k` entry was validated separately and was not timed**. Existing default-B8 measurements above, if present, remain their original separate run.
44
-
45
- | Workload | B8 p50 / p95 (ms) | Cap32 p50 / p95 (ms) | p50 ratio · paired 95% CI |
46
- |---|---:|---:|---:|
47
- | 242 tokens × 1 question | 13.92 / 14.01 | 13.92 / 14.05 | 1.0001 · [0.9975, 1.0020] |
48
- | 242 tokens × 8 questions | 22.65 / 22.75 | 22.60 / 22.77 | 0.9981 · [0.9967, 0.9994] |
49
- | 242 tokens × 16 questions | 38.86 / 39.42 | 33.39 / 33.55 | 0.8592 · [0.8578, 0.8603] |
50
- | 242 tokens × 32 questions | 71.03 / 71.46 | 54.83 / 60.15 | 0.7718 · [0.7706, 0.7724] |
51
- | 1024 tokens × 1 question | 18.09 / 18.24 | 18.08 / 18.21 | 0.9992 · [0.9978, 1.0003] |
52
- | 1024 tokens × 8 questions | 65.74 / 65.89 | 65.72 / 66.52 | 0.9997 · [0.9990, 1.0004] |
53
- | 1024 tokens × 32 questions | 242.37 / 243.72 | 217.63 / 218.37 | 0.8979 · [0.8961, 0.8986] |
54
- | Multi-context · 32 decisions | 44.08 / 44.25 | 28.76 / 28.87 | 0.6523 · [0.6519, 0.6537] |
55
-
56
- AMD ROCm gfx942, FP32 SDPA, one resident model and one request at a time. Each point/mode had 10 warmups and 50 timed requests; the same validation, transfer and auditing hooks were present in both modes. Ten paired blocks (five AB, five BA; five observations per mode/block) were retained. The reported p50 ratio is median(cap32)/median(B8); 2,000 whole-block bootstrap resamples within order strata used fixed seed 20260922. p95 has only 50 observations/mode and correspondingly limited precision. Model loading is excluded.
57
-
58
- Choice count curves repeat the same synthetic four-candidate question with opaque ID changes. The multi-context fixture repeats eight existing contexts with two related questions twice (32 decisions). These are throughput shapes, not quality examples. The fixed 15% utility gate remained **FAIL** for both models: Q16 improved approximately 14%, despite a confidence interval below 1. Q32 and multi-context improvements do not retroactively change that outcome. The option is an explicit engineering choice, not a universal speed guarantee.
59
-
60
- At 1024×32, peak allocated memory increased from approximately 2.652 to 3.824 GiB. Shared allocator reserved-memory observations are not independent per-mode estimates. Equal total padding does not imply equal peak memory.
61
-
62
- #### Public-entry validation
63
-
64
- The actual imported package passed 46 forwards per model (92 total across Kai and Lex) over 225 technical row-occurrences/model, including all three types, dynamic candidate counts, enabled32, tail33, unequal-length fallback and mixed64. Late invalid and empty requests added zero forwards. The original strict bounds were 1e-4 logits, 2e-5 probability and normalized Score error, with zero hard-decision flips. Both models passed; fallback outputs were exact. All 489 parameters and 66 buffers retained identical contents and versions. No training, quality evaluation or timing was added in this integration check. Fixed repetition is not independent quality support.
65
-
66
- [Exact paired measurements, numerical results and retained failure gates](BATCH_CAPACITY.json). No weights or default entry behavior changed, and no Studio speedup is claimed.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
README.md CHANGED
@@ -46,9 +46,15 @@ Supply the context, question and candidate descriptions at runtime. **Choice** r
46
 
47
  Lex is evaluated as an English specialist on these four workflows. For broader multilingual tasks, explore the general [Kai model](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B).
48
 
49
- **Optional larger batches:** `predict_auto_1k` uses up to 32 same-type questions when padding does not increase. Earlier B8/B32 comparisons showed approximately 23% lower median latency for 32 short questions and 35% for the multi-context fixture; the final guard was validated separately, not timed. [Usage and measurements](PERFORMANCE.md#optional-larger-batches).
50
 
51
- **128 mixed questions in 154 ms — 57% lower latency.** Automatic typed scheduling accelerates the default SystemOne path with the same weights. Paired local AMD measurements on a fixed workload. [Latency and scaling](PERFORMANCE.md).
 
 
 
 
 
 
52
 
53
  ## Use
54
 
@@ -92,22 +98,22 @@ curl -X POST https://your-decision-endpoint.example/v1/systemone \
92
  }'
93
  ```
94
 
95
- [Official Python SDK](https://docs.typesafe.ai/sdk/python/usage) · [HTTP API](https://docs.typesafe.ai/api) · [Usage and deployment details](USAGE.md)
96
 
97
  ## Make it yours
98
 
99
- **One state. Many decisions.** Use the [System One API](SYSTEM_ONE.md) to submit up to 128 typed questions, or batch the same questions across independent contexts. Results return under your original question IDs.
100
 
101
- Run the [official SDK and curl examples](USAGE.md), or adapt Lex to your own labels and rubrics with the included [fine-tuning CLI](FINETUNING.md). Both hard and soft training labels are supported, with checkpoint resume.
102
 
103
  The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
104
 
105
  ## Architecture
106
 
107
- ![Lex architecture](https://gist.githubusercontent.com/Xunzhuo/4020f574e3d38e5e5eae00063bdf9dce/raw/34876f9fd939829a3026353d18a5c58c7bb233fd/lex-architecture.png)
108
 
109
  Three 22-layer bidirectional encoder paths share multilingual input embeddings, with separate interaction layers and candidate readouts for Choice, Noul and Score. Lex retains Kai's architecture and specializes its weights through supervised fine-tuning.
110
 
111
- [Architecture details](ARCHITECTURE.md) · [Training and methods](METHODS.md)
112
 
113
  Built on [Kai](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B) and [Vela Encoder](https://huggingface.co/llm-semantic-router/Vela-1.0-Encoder-307M). Probabilities are not calibrated confidence; candidate order and task wording can affect outputs. [Attribution and retained third-party terms](NOTICE) · [License scope](LICENSING_STATUS.md).
 
46
 
47
  Lex is evaluated as an English specialist on these four workflows. For broader multilingual tasks, explore the general [Kai model](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B).
48
 
49
+ **128 mixed questions in 154 ms — 57% lower latency.** Automatic typed scheduling accelerates the measured SystemOne runtime with the same weights. Paired local AMD measurements on a fixed workload. [Latency and scaling](evaluation/PERFORMANCE.md).
50
 
51
+ ## Download for local inference
52
+
53
+ ```bash
54
+ hf download llm-semantic-router/Decision-1.0-Lex-0.6B --local-dir Decision-1.0-Lex-0.6B
55
+ ```
56
+
57
+ This repository contains model files and provenance only. Local inference requires a compatible vLLM Semantic Router Decision runtime on AMD ROCm; the runtime is distributed separately. It must support `vllm-sr-decision` format version 1 and the file map in [`config.json`](config.json). `transformers.AutoModel.from_pretrained` does not load the complete decision model.
58
 
59
  ## Use
60
 
 
98
  }'
99
  ```
100
 
101
+ [Official Python SDK](https://docs.typesafe.ai/sdk/python/usage) · [HTTP API](https://docs.typesafe.ai/api)
102
 
103
  ## Make it yours
104
 
105
+ **One state. Many decisions.** A compatible System One endpoint can submit typed questions and return results under their original question IDs. Request limits and batching depend on that deployment.
106
 
107
+ Use the [SDK and curl examples](#use) with a compatible endpoint. Lex's specialization recipe and source data are documented in [Methods](evaluation/METHODS.md) and [Training provenance](TRAINING_PROVENANCE.json).
108
 
109
  The complete 1,024-token budget includes context, instructions, all candidates and special tokens. Overlength requests return an error. Native Choice and Score support 2–255 candidates or ordered levels; the System One and Studio interfaces use 2–10 Score levels.
110
 
111
  ## Architecture
112
 
113
+ ![Lex architecture](assets/architecture.png)
114
 
115
  Three 22-layer bidirectional encoder paths share multilingual input embeddings, with separate interaction layers and candidate readouts for Choice, Noul and Score. Lex retains Kai's architecture and specializes its weights through supervised fine-tuning.
116
 
117
+ [Architecture details](ARCHITECTURE.md) · [Training and methods](evaluation/METHODS.md)
118
 
119
  Built on [Kai](https://huggingface.co/llm-semantic-router/Decision-1.0-Kai-0.6B) and [Vela Encoder](https://huggingface.co/llm-semantic-router/Vela-1.0-Encoder-307M). Probabilities are not calibrated confidence; candidate order and task wording can affect outputs. [Attribution and retained third-party terms](NOTICE) · [License scope](LICENSING_STATUS.md).
RUNTIME_UPDATE.json DELETED
@@ -1,32 +0,0 @@
1
- {
2
- "schema": "decision.runtime-update.v1",
3
- "change": "Reuse complete admitted input encodings within one synchronous request.",
4
- "weights_unchanged": true,
5
- "state_activations_cached": false,
6
- "cross_request_cache": false,
7
- "complete_input_tokens": 1024,
8
- "maximum_batch_size": 8,
9
- "native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
10
- "verification": {
11
- "AMD_FP32": true,
12
- "per_model_calls": 24,
13
- "all_output_fields_equal": true,
14
- "all489_parameter_content_versions_unchanged": true,
15
- "all_types": true,
16
- "duplicate_external_ids": true,
17
- "partial_final_batch": true,
18
- "complete1024": true,
19
- "invalid_later_row_rejected_before_forward": true,
20
- "empty_request": true
21
- },
22
- "new_inference_sources": {
23
- "decision_inference/profile.py": {
24
- "bytes": 2071,
25
- "sha256": "7e732fb3a9be93922a2e7e94920d9c75be7a3d07fbc3d88b1c2056bd374008f7"
26
- },
27
- "decision_inference/_request.py": {
28
- "bytes": 3026,
29
- "sha256": "85b8349cdf08a3550027606b3108778955f84d66c52fa11dd908ea50311e69e5"
30
- }
31
- }
32
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
SYSTEM_ONE.md DELETED
@@ -1,134 +0,0 @@
1
- # One state. Many decisions.
2
-
3
- Decision uses the [System One API format](https://docs.typesafe.ai/api): supply
4
- `state`, `model` and a map of typed `questions`; receive an `answers` map under
5
- the same question IDs. Noul returns the probability of yes, Choice selects from
6
- your named options, and Score evaluates an ordered rubric.
7
-
8
- ```python
9
- from decision_inference import SystemOne
10
-
11
- # native is your loaded Decision checkpoint; see USAGE.md for loading.
12
- client = SystemOne(native, batching="auto")
13
- result = client.system_one(
14
- state={"message": "Please refund the duplicate charge. I need this fixed today."},
15
- questions={
16
- "refund_requested": {
17
- "type": "noul",
18
- "instructions": "Does the customer explicitly request a refund?",
19
- },
20
- "team": {
21
- "type": "choice",
22
- "instructions": "Which team should handle this request?",
23
- "criteria": {"Billing": "Charges and refunds", "Support": "Technical problems"},
24
- },
25
- "urgency": {
26
- "type": "score",
27
- "instructions": "How urgent is the request?",
28
- "criteria": ["No deadline", "Needed soon", "Needed today"],
29
- },
30
- },
31
- )
32
- print(result["answers"])
33
- ```
34
-
35
- `client.evaluate(request)` accepts the complete JSON request object, including
36
- `model`. A client is bound to its loaded checkpoint; a mismatched model is an
37
- error. Public Kai/Lex names are inferred from their verified native manifest.
38
- For your own fine-tune, use `SystemOne(native, model="my-decision-model")`.
39
-
40
- ## Batch the questions or the contexts
41
-
42
- One request accepts **up to 128 questions**. For one set of questions across many
43
- independent contexts, use `client.batch`:
44
-
45
- ```python
46
- questions = {
47
- "refund_requested": {
48
- "type": "noul",
49
- "instructions": "Does the customer explicitly request a refund?",
50
- }
51
- }
52
- results = client.batch([
53
- {"model": client.model, "state": message, "questions": questions}
54
- for message in messages
55
- ])
56
- # results[i] corresponds to messages[i]; question IDs can repeat across requests.
57
- ```
58
-
59
- `batch` is a Decision Python extension around ordinary System One requests. It
60
- accepts up to 128 requests and 512 total decisions, with a combined 2 MiB input
61
- limit. Responses preserve request order and question order. The entire batch
62
- must pass validation and complete-input token admission before any forward.
63
- An overlength or malformed input fails the call without partial answers.
64
-
65
- The default uses physical batches of up to eight. `batching="auto"` uses the
66
- released padding-aware scheduler: up to 32 consecutive same-type decisions when
67
- that adds no padding; otherwise it keeps B8. FP32 rounding can vary with physical
68
- batch shape. Each state/question pair still has its own encoder computation.
69
- One API call does not imply one forward or a shared state activation cache.
70
-
71
- ## Typed fields
72
-
73
- | Type | `criteria` | Answer |
74
- |---|---|---|
75
- | `noul` | Optional `true` / `false` descriptions | `type`, `noul` |
76
- | `choice` | 2–255 named options; descriptions may be null | `type`, `choice`, `probabilities`, `confidence` |
77
- | `score` | 2–10 ordered level descriptions | `type`, `score`, `legend`, `probabilities`, `confidence` |
78
-
79
- State, instructions and descriptions accept strings, JSON objects or arrays.
80
- Structured values become deterministic compact JSON with sorted object keys;
81
- array order, Choice option order and Score level order are preserved. Question
82
- IDs are bookkeeping only. Choice names are part of the semantic input, including
83
- when their description is null. Score levels are indexed from zero; `score`
84
- preserves the native probability-weighted FP32 expectation. Structured Score
85
- descriptions appear as JSON strings in `legend`.
86
-
87
- Every complete state/question/candidate sequence must fit **1,024 tokens**.
88
- Nothing is truncated. The token count includes the repeated state for each
89
- question; `usage.input_tokens` sums these complete sequences and
90
- `usage.output_tokens` is zero because the model returns scores without generating
91
- text. These are local computation counts, not TypeSafe billing counts.
92
-
93
- Decision's `confidence` is the largest candidate probability, matching its native
94
- runtime. It is not calibrated correctness. TypeSafe does not specify its own
95
- formula in the [confidence documentation](https://docs.typesafe.ai/confidence),
96
- so thresholds should not be transferred between models without validation.
97
- The shared request/answer schema does not claim identical weights, confidence
98
- values, hosted service limits or SDK behavior.
99
-
100
- ## Fine-tune with the same inputs
101
-
102
- Use the same conversion for training so structured inputs, option names and
103
- criteria have identical semantics at training and inference time:
104
-
105
- ```python
106
- import json
107
- from pathlib import Path
108
- from decision_finetune.system_one import system_one_training_rows
109
-
110
- request = json.loads(Path("examples/system-one.json").read_text())
111
- rows = system_one_training_rows(
112
- request, # the same model/state/questions object used for inference
113
- targets={
114
- "refund_requested": {"probability": 1.0},
115
- "team": {"choice_id": "Billing"},
116
- "urgency": {"probabilities": [0.0, 0.0, 1.0]},
117
- },
118
- request_id="ticket-1001",
119
- source_id="support-tickets",
120
- component_id="customer-42",
121
- )
122
- ```
123
-
124
- Save the rows as JSONL for the existing [fine-tuning CLI](FINETUNING.md). Targets
125
- stay separate from state, instructions and criteria. Related examples share a
126
- source/component ID and stay in one split; the CLI checks component and exact
127
- input overlap. Optional `hard_target_ids` supplies separate evaluation labels
128
- under the same question IDs. Noul supports soft yes probabilities; Choice and
129
- Score support complete soft distributions in the original candidate order.
130
-
131
-
132
- ### Default typed scheduling
133
-
134
- The default `SystemOne` path groups complete, admitted questions by decision type in physical batches of eight, then restores the original request and question order. This works for many questions over one state and questions across multiple contexts. No API changes or application-side sorting are needed. The optional `batching="auto"` policy is unchanged. [Measured mixed-question scaling](PERFORMANCE.md).
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
SYSTEM_ONE_VALIDATION.json DELETED
@@ -1,66 +0,0 @@
1
- {
2
- "all_reference_outputs_exact": true,
3
- "api_source": {
4
- "bytes": 10690,
5
- "sha256": "d2ef9aa1d0badcf168065a782f8a96316045455a676c08bdb5dd384a651e2194"
6
- },
7
- "complete_input_tokens": 1024,
8
- "coverage": [
9
- "Noul, Choice and Score",
10
- "128 questions",
11
- "128 contexts and 512 decisions",
12
- "question rename and reorder",
13
- "irrelevant question insertion",
14
- "single and multi-context agreement",
15
- "default B8 and opt-in padding-aware B32",
16
- "late invalid input before any forward",
17
- "same training and inference input encoding"
18
- ],
19
- "cross_batch_tolerance": 2e-05,
20
- "device": "AMD GPU / ROCm",
21
- "limits": "Technical compatibility validation on synthetic unlabeled cases; not evidence of accuracy, calibration or latency gains.",
22
- "max_decisions_per_batch": 512,
23
- "max_questions_per_request": 128,
24
- "max_requests_per_batch": 128,
25
- "models": {
26
- "Kai": {
27
- "all_489_parameters_and_66_buffers_unchanged": true,
28
- "cross_case_checks": 713,
29
- "forward_calls": 242,
30
- "forward_rows": 2756,
31
- "hard_flips": 0,
32
- "invalid_input_checks": 7,
33
- "invalid_input_forward_calls": 0,
34
- "max_probability_error": 2.0563602447509766e-06,
35
- "max_score_error": 2.980232238769531e-07,
36
- "native_manifest_sha256": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
37
- "result_sha256": "8cb3bc435e36b79d96b60058e9b4c57b9996dcb8a55a1cd3faf6c7324e7f917c",
38
- "same_physical_reference_outputs_exact": true
39
- },
40
- "Lex": {
41
- "all_489_parameters_and_66_buffers_unchanged": true,
42
- "cross_case_checks": 713,
43
- "forward_calls": 242,
44
- "forward_rows": 2756,
45
- "hard_flips": 0,
46
- "invalid_input_checks": 7,
47
- "invalid_input_forward_calls": 0,
48
- "max_probability_error": 1.1324882507324219e-06,
49
- "max_score_error": 7.152557373046875e-07,
50
- "native_manifest_sha256": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
51
- "result_sha256": "e94c5f1109f15005415a48623602d5255b38bd33451b55c758a0db0c16f68191",
52
- "same_physical_reference_outputs_exact": true
53
- }
54
- },
55
- "optimizer_updates": 0,
56
- "performance_benchmark": false,
57
- "quality_evaluation": false,
58
- "reference": "Independent System One mapping using the original Studio single-question conversion and native predictor; identical physical batches.",
59
- "status": "PASS_AMD_SYSTEM_ONE_API",
60
- "total_forward_calls": 484,
61
- "training_conversion_source": {
62
- "bytes": 1784,
63
- "sha256": "9b8e7cf14dafedb9bfc9f174db64be9f8c311cca3c48eba0c90f55a3bbd62c19"
64
- },
65
- "weights_unchanged": true
66
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
USAGE.md DELETED
@@ -1,121 +0,0 @@
1
- # Use Lex
2
-
3
- Use the official TypeSafe SDK or curl to send a shared context and named typed questions to a SystemOne-compatible endpoint. The example asks a routing question and an urgency question about the same delivery request.
4
-
5
- ## Endpoint setup
6
-
7
- Replace `https://your-decision-endpoint.example` with an endpoint configured to expose `Decision-1.0-Lex-0.6B`. Set the `DECISION_API_KEY` environment variable to the key issued by that endpoint's operator. The sized model name below is a deployment alias that the operator must configure.
8
-
9
- The Hugging Face repository distributes weights and local inference code; it does not provision an HTTP service or issue API keys. TypeSafe does not host these Decision weights. Installing the official SDK supplies a client for a compatible service, not a model deployment.
10
-
11
- These are request examples, not recorded model predictions. No example probabilities or performance results are implied.
12
-
13
- ## Official Python SDK
14
-
15
- ```bash
16
- pip install typesafe-sdk
17
- ```
18
-
19
- ```python
20
- import os
21
- from typesafe_sdk import TypeSafeClient, Choice, Noul
22
-
23
- client = TypeSafeClient(
24
- api_key=os.environ["DECISION_API_KEY"],
25
- base_url="https://your-decision-endpoint.example",
26
- model="Decision-1.0-Lex-0.6B",
27
- )
28
- questions = {
29
- "route": Choice(instructions="Which team should handle this request?",
30
- criteria={"delivery": "Damaged or missing parcels", "billing": "Payments and invoices"}),
31
- "urgent": Noul(instructions="Does the customer request action today?"),
32
- }
33
- response = client.system_one(state="The parcel arrived damaged. Please send a replacement today.", questions=questions)
34
- print(response.choices["route"].choice, response.nouls["urgent"].noul)
35
- ```
36
-
37
- `response.choices["route"].choice` is the selected candidate ID; `response.nouls["urgent"].noul` is the probability that the condition holds. The application decides how to act on the answers.
38
-
39
- ## Equivalent curl request
40
-
41
- The model, state, question IDs, instructions and candidate order are identical to the SDK example.
42
-
43
- ```bash
44
- curl -X POST https://your-decision-endpoint.example/v1/systemone \
45
- -H "Authorization: Bearer $DECISION_API_KEY" \
46
- -H "Content-Type: application/json" \
47
- --data '{
48
- "model": "Decision-1.0-Lex-0.6B",
49
- "state": "The parcel arrived damaged. Please send a replacement today.",
50
- "questions": {
51
- "route": {"type": "choice", "instructions": "Which team should handle this request?", "criteria": {"delivery": "Damaged or missing parcels", "billing": "Payments and invoices"}},
52
- "urgent": {"type": "noul", "instructions": "Does the customer request action today?"}
53
- }
54
- }'
55
- ```
56
-
57
- See the [official Python usage guide](https://docs.typesafe.ai/sdk/python/usage) and [HTTP request/response reference](https://docs.typesafe.ai/api).
58
-
59
- ## Several contexts, the same questions
60
-
61
- Keep the `questions` map and submit another `state` to `client.system_one`. Each request evaluates both questions against its own context. A larger question map expresses more decisions about that context; service concurrency, request limits and scheduling depend on the deployment.
62
-
63
- The bundled native implementations also support multi-context batching. This is a local inference capability; it does not imply that a deployment exposes a batch HTTP route. All complete-input limits still apply to each rendered state/question/candidate sequence.
64
-
65
- ## Deployment names and native compatibility
66
-
67
- The HTTP examples use the configured alias `Decision-1.0-Lex-0.6B`. The bundled local model retains the stable internal identifier `Decision-1.0-Lex`. A deployment maps its public alias to that native model; adding a size suffix to the Hub repository does not change the native identifier or its accepted aliases.
68
-
69
- Existing local SystemOne CLI examples continue to use the unsized internal identifier. The native code and weights are unchanged.
70
-
71
- ## Local native installation
72
-
73
- Download the complete public release with the Hugging Face CLI. No access approval or login is required:
74
-
75
- ```bash
76
- hf download llm-semantic-router/Decision-1.0-Lex-0.6B --local-dir Decision-1.0-Lex-0.6B
77
- cd Decision-1.0-Lex-0.6B
78
- ```
79
-
80
- Use `--local-dir` to materialize ordinary files for the native loader. Use the compatible ROCm/Python environment described below.
81
-
82
- This release bundles the verified native under `native/`, the public Python API. Run the following commands from the complete distribution root in a compatible environment. `PACKAGE_MANIFEST.json` records the exact payload; do not add files inside `native/` because its loader checks the complete file roster.
83
-
84
- Use an existing compatible AMD ROCm environment. The verified source pins Transformers 4.57.6; tested companion versions were Python 3.12.13, tokenizers 0.22.2 and safetensors 0.8.0. Actual validation used a ROCm PyTorch 2.12 development build, not a promised generic wheel installation. Choose a matching supported ROCm/PyTorch installation for your host; this release does not supply an installer or a CPU/NVIDIA inference path. Do not upgrade an active environment in place.
85
-
86
- ## System One: parallel typed questions
87
-
88
- For local execution, the existing native CLI remains available from the downloaded repository:
89
-
90
- ```bash
91
- ROCR_VISIBLE_DEVICES=0 python systemone.py \
92
- --input examples/system-one.json --output answers.json --batching auto
93
- ```
94
-
95
- The bundled wrapper accepts up to 128 questions per request and combines independent contexts with up to 512 total decisions. Each complete state/question/candidate sequence must fit 1,024 tokens; validation happens before the first forward. Default physical batching is B8; `auto` opts into the released homogeneous, padding-aware B32 scheduler. These local limits are not a promise about another endpoint's limits. [Native fields and fine-tuning conversion](SYSTEM_ONE.md).
96
-
97
- ## Native records
98
-
99
- The original native JSONL interface remains available for applications that
100
- need logits, explicit candidate IDs or arbitrary ordered Score values.
101
-
102
- ```bash
103
- export PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}"
104
- export PYTHONDONTWRITEBYTECODE=1
105
- export HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 TOKENIZERS_PARALLELISM=false
106
- ROCR_VISIBLE_DEVICES=0 python infer.py \
107
- --input examples/decisions.jsonl --output predictions.jsonl --batch-size 8
108
- ```
109
-
110
- The three examples illustrate the Choice, Noul and Score interfaces. They are synthetic interface examples, not evaluated model predictions. Each input line has exactly `id`, `state_text`, and `question`; do not attach labels. Choice uses `options`, Score uses ordered `levels`, and Noul uses the fixed no/yes pair with optional `false_criterion`/`true_criterion`. IDs identify outputs and do not become model tokens.
111
-
112
- The Python API returns original record/candidate identities, logits and probabilities in the supplied order. Choice includes the chosen candidate ID. Noul includes the yes probability; it does not emit a boolean decision. An application can choose first-argmax (an exact no/yes tie selects no) or `p_yes >= 0.5` (tie selects yes), but should declare that choice. Score includes `expected_value` over the supplied values and `score`, the expected ordinal index from 0 to K−1. Neither is automatically a calibrated business utility.
113
-
114
- All state/question/candidate text and markers count toward the complete 1024 limit. `predict_1k` checks every input before the first forward and rejects overflow without truncation. Batch sizes 1–8 are supported by this wrapper. Use it rather than directly invoking the native research-capacity entrypoint.
115
-
116
- A compatible fine-tuned export can be used with `--native <directory> --manifest-sha256 <trusted hash>`. Only the pinned runtime revision and three-path architecture are accepted. Downloaded HF symlinks must be materialized into real files before loading. The manifest hash is an integrity check, not a reason to execute arbitrary untrusted Python.
117
-
118
-
119
- ### Default typed scheduling
120
-
121
- The default `SystemOne` path groups complete, admitted questions by decision type in physical batches of eight, then restores the original request and question order. This works for many questions over one state and questions across multiple contexts. No API changes or application-side sorting are needed. The optional `batching="auto"` policy is unchanged. [Measured mixed-question scaling](PERFORMANCE.md).
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
assets/architecture.pdf DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:6aa6895f5708cda7d935991cf9c1179d9e54e22a32459968a052acf05f62a14a
3
- size 173099
 
 
 
 
assets/attention-geglu.pdf DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:43b3485510971d29e66bc17b30b4bffe934ad66078598cc3e2efd63ae9b9bc7c
3
- size 172797
 
 
 
 
assets/attention-geglu.png DELETED

Git LFS Details

  • SHA256: de7d88e3cf036d3b8cba38df47d5b3182faecb9233e675104cee7c75dd7fd8c5
  • Pointer size: 131 Bytes
  • Size of remote file: 266 kB
assets/candidate-readout.pdf DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:56b9e06a343e9044c38cfb8d1ba5ba121bdc7d763db5ce45b927aacdcb00c995
3
- size 156814
 
 
 
 
assets/candidate-readout.svg DELETED
assets/residual-layers.pdf DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:2c5b50bb840ae115a1264b66d535a1048172e891a33d02d507c193525cc45630
3
- size 131144
 
 
 
 
assets/residual-layers.png DELETED

Git LFS Details

  • SHA256: f3ee6441fe1016e2435256644675b08bff99cec72d521e49a78f6cbe5b2dc3bf
  • Pointer size: 131 Bytes
  • Size of remote file: 309 kB
config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "decision_format": "vllm-sr-decision",
3
+ "format_version": 1,
4
+ "model_name": "Decision-1.0-Lex-0.6B",
5
+ "runtime_family": "vela-encoder",
6
+ "model_config": "native/decision_config.json",
7
+ "backbone": {
8
+ "config": "native/encoder/config.json",
9
+ "weights": ["native/encoder/model.safetensors"]
10
+ },
11
+ "tokenizer": {
12
+ "json": "native/tokenizer/tokenizer.json",
13
+ "config": "native/tokenizer/tokenizer_config.json",
14
+ "special_tokens_map": "native/tokenizer/special_tokens_map.json"
15
+ },
16
+ "decision_weights": {
17
+ "choice_encoder": "native/choice_encoder.safetensors",
18
+ "score_encoder": "native/score_encoder.safetensors",
19
+ "decision_heads": "native/decision_heads.safetensors"
20
+ }
21
+ }
decision_finetune/__init__.py DELETED
@@ -1 +0,0 @@
1
- """Portable JSONL fine-tuning CLI; end-to-end AMD validation remains pending."""
 
 
decision_finetune/__main__.py DELETED
@@ -1,80 +0,0 @@
1
- """Run with PYTHONPATH=<release directory> python -m decision_finetune."""
2
- import argparse
3
- import json
4
- import math
5
- import sys
6
- from . import data
7
-
8
-
9
- def parser():
10
- p = argparse.ArgumentParser(description="Single-AMD full1K native Choice/Noul/Score fine-tuning")
11
- commands = p.add_subparsers(dest="command", required=True)
12
- v = commands.add_parser("validate-jsonl", help="Structural/split checks only; does NOT establish token support")
13
- v.add_argument("--train", required=True); v.add_argument("--dev", required=True)
14
- v.add_argument("--selection", choices=("macro-nll", "hard-accuracy"), default="macro-nll")
15
- t = commands.add_parser("train", help="Loads a real AMD model; complete admission precedes every forward")
16
- for name in ("native", "manifest-sha256", "train", "dev", "output"):
17
- t.add_argument("--" + name, required=True)
18
- t.add_argument("--epochs", type=int, default=4)
19
- t.add_argument("--logical-batch-size", type=int, default=64)
20
- t.add_argument("--micro-batch-size", type=int, default=8)
21
- t.add_argument("--seed", type=int, default=20260921)
22
- t.add_argument("--encoder-lr", type=float, default=2.5e-5)
23
- t.add_argument("--head-lr", type=float, default=1e-4)
24
- t.add_argument("--lr-min", type=float, default=1e-6)
25
- t.add_argument("--weight-decay", type=float, default=.01)
26
- t.add_argument("--clip-norm", type=float, default=1.)
27
- t.add_argument("--score-rps-weight", type=float, default=.1)
28
- t.add_argument("--warmup-steps", type=int)
29
- t.add_argument("--warmup-ratio", type=float, default=.1)
30
- t.add_argument("--max-steps", type=int)
31
- t.add_argument("--eval-steps", help="Explicit comma-separated update steps including0/final; default epoch ends")
32
- t.add_argument("--selection", choices=("macro-nll", "hard-accuracy"), default="macro-nll")
33
- t.add_argument("--selection-tolerance", type=float, default=1e-8)
34
- t.add_argument("--save-every", type=int, default=0, help="Additional update-boundary checkpoints;0 means evaluation points only")
35
- t.add_argument("--cpu-threads", type=int, default=2)
36
- t.add_argument("--max-reserved-gib", type=float, default=64.)
37
- t.add_argument("--allow-nondeterministic-kernels", action="store_true", help="Explicitly opt out of PyTorch deterministic algorithm enforcement")
38
- t.add_argument("--resume", help="Existing immutable update-boundary checkpoint directory")
39
- t.add_argument("--resume-sha256", help="Trusted manifest SHA for --resume")
40
- t.add_argument("--stop-after-step", type=int, help="Clean scheduling pause after this original update; does not change the LR horizon or training config")
41
- return p
42
-
43
-
44
- def configuration(args):
45
- fields = ("epochs", "logical_batch_size", "micro_batch_size", "seed", "encoder_lr", "head_lr", "lr_min", "weight_decay",
46
- "clip_norm", "score_rps_weight", "warmup_steps", "warmup_ratio", "max_steps", "selection", "selection_tolerance",
47
- "save_every", "cpu_threads", "max_reserved_gib")
48
- c = {k: getattr(args, k) for k in fields}
49
- data.need(all(c[k] > 0 for k in ("epochs", "logical_batch_size", "micro_batch_size", "cpu_threads")), "Positive counts required")
50
- data.need(c["micro_batch_size"] <= c["logical_batch_size"] and c["micro_batch_size"] <= 8, "Micro batch must fit logical batch and ≤8")
51
- data.need(0 <= c["seed"] < 2**32 and c["save_every"] >= 0, "Invalid seed/save cadence")
52
- data.need(all(math.isfinite(c[k]) and c[k] > 0 for k in ("encoder_lr", "head_lr", "clip_norm", "max_reserved_gib")), "Positive finite optimizer/memory values")
53
- data.need(c["max_reserved_gib"] <= 64 and all(math.isfinite(c[k]) and c[k] >= 0 for k in ("weight_decay", "score_rps_weight", "selection_tolerance"))
54
- and 0 <= c["warmup_ratio"] < 1, "Invalid budget/objective/warmup")
55
- data.need(math.isfinite(c["lr_min"]) and 0 <= c["lr_min"] <= min(c["encoder_lr"], c["head_lr"]), "Invalid LR floor")
56
- data.need(bool(args.resume) == bool(args.resume_sha256), "Resume directory AND manifest SHA required")
57
- c.update(eval_steps=None if args.eval_steps is None else [int(v) for v in args.eval_steps.split(",")],
58
- deterministic_algorithms=not args.allow_nondeterministic_kernels,
59
- train_precision="bf16_autocast_fp32_parameters_loss", eval_precision="fp32", eval_batch_size=8,
60
- lr_convention="linear warmup to base, then cosine to lr_min at final; with no warmup step1=base; one-step schedule uses base",
61
- input_limit=1024, truncation="error", selection_weighting=("equal_present_types" if args.selection == "macro-nll" else "row"))
62
- return c
63
-
64
-
65
- def main():
66
- sys.dont_write_bytecode = True
67
- args = parser().parse_args()
68
- if args.command == "validate-jsonl":
69
- train, dev = data.load_splits(args.train, args.dev, args.selection)
70
- print(json.dumps({"status": "PASS_JSONL_AND_EXACT_SPLIT_CHECKS", "train_rows": len(train), "dev_rows": len(dev),
71
- "train_sha256": data.file_sha(args.train), "dev_sha256": data.file_sha(args.dev),
72
- "token_admission_performed": False, "model_calls": 0}))
73
- return
74
- from .run import execute
75
- result = execute(args, configuration(args))
76
- print(json.dumps({k: result[k] for k in ("status", "native", "native_manifest_sha256", "selected0_no_adaptation", "checkpoint") if k in result}))
77
-
78
-
79
- if __name__ == "__main__":
80
- main()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_finetune/data.py DELETED
@@ -1,139 +0,0 @@
1
- """JSONL, split and deterministic schedule contracts. No model imports."""
2
- import copy
3
- import hashlib
4
- import json
5
- import math
6
- from pathlib import Path
7
- import unicodedata
8
-
9
- KINDS = ("choice", "noul", "score")
10
- DEFAULT_NO = "No. The statement or question is not satisfied."
11
- DEFAULT_YES = "Yes. The statement or question is satisfied."
12
-
13
-
14
- def need(ok, message):
15
- if not ok:
16
- raise ValueError(message)
17
-
18
-
19
- def digest(value):
20
- return hashlib.sha256(json.dumps(value, ensure_ascii=False, sort_keys=True,
21
- separators=(",", ":"), allow_nan=False).encode()).hexdigest()
22
-
23
-
24
- def file_sha(path):
25
- h = hashlib.sha256()
26
- with Path(path).open("rb") as stream:
27
- for chunk in iter(lambda: stream.read(1 << 20), b""):
28
- h.update(chunk)
29
- return h.hexdigest()
30
-
31
-
32
- def text(value):
33
- need(isinstance(value, str) and value.strip(), "Nonempty text required")
34
- return " ".join(unicodedata.normalize("NFKC", value).casefold().split())
35
-
36
-
37
- def semantics(row):
38
- need(isinstance(row, dict) and isinstance(row.get("question"), dict), "Record/question objects required")
39
- text(row.get("id")); text(row.get("state_text"))
40
- q = row["question"]; text(q.get("text"))
41
- kind = str(q.get("type", "")).lower()
42
- need(kind in KINDS, "Question type must be Choice, Noul or Score")
43
- if kind == "noul":
44
- options = [{"id": "no", "text": q.get("false_criterion", DEFAULT_NO)},
45
- {"id": "yes", "text": q.get("true_criterion", DEFAULT_YES)}]
46
- else:
47
- options = q.get("options" if kind == "choice" else "levels")
48
- need(isinstance(options, list) and 2 <= len(options) <= 255, "Require 2..255 candidates")
49
- for option in options:
50
- need(isinstance(option, dict), "Candidate object required")
51
- text(option.get("id")); text(option.get("text"))
52
- ids = [o["id"] for o in options]
53
- need(len(set(ids)) == len(ids), "Candidate IDs must be unique")
54
- values = [float(o["value"]) for o in options] if kind == "score" else list(range(len(ids)))
55
- need(all(math.isfinite(v) for v in values) and all(a < b for a, b in zip(values, values[1:])), "Increasing finite Score values")
56
- gold = row.get("target")
57
- need(isinstance(gold, dict), "Explicit target required")
58
- if kind == "noul":
59
- need(set(gold) == {"probability"}, "Noul target is probability of yes")
60
- yes = float(gold["probability"]); target = [1 - yes, yes]
61
- elif set(gold) == {"probabilities"}:
62
- target = [float(p) for p in gold["probabilities"]]
63
- else:
64
- need(kind == "choice" and set(gold) == {"choice_id"} and gold["choice_id"] in ids,
65
- "Choice needs choice_id/probabilities; Score needs probabilities (one-hot is hard)")
66
- target = [float(cid == gold["choice_id"]) for cid in ids]
67
- need(len(target) == len(ids) and all(math.isfinite(p) and 0 <= p <= 1 for p in target)
68
- and abs(sum(target) - 1) <= 1e-6, "Normalized finite hard/soft target required")
69
- hard = row.get("hard_target_id")
70
- need(hard is None or hard in ids, "hard_target_id must be an original candidate ID")
71
- return kind, options, values, target
72
-
73
-
74
- def input_signature(row):
75
- kind, options, values, _ = semantics(row)
76
- descriptions = [text(o["text"]) for o in options]
77
- if kind == "choice":
78
- descriptions.sort() # Reordering and opaque IDs cannot hide exact split overlap.
79
- return digest({"kind": kind, "state": text(row["state_text"]),
80
- "question": text(row["question"]["text"]), "descriptions": descriptions,
81
- "values": values if kind == "score" else None})
82
-
83
-
84
- def model_row(row):
85
- # External hard evaluation labels are never passed to the collator/model.
86
- return copy.deepcopy({k: v for k, v in row.items() if k != "hard_target_id"})
87
-
88
-
89
- def read_jsonl(path):
90
- rows = []
91
- with Path(path).open(encoding="utf-8") as stream:
92
- for number, line in enumerate(stream, 1):
93
- need(bool(line.strip()), f"Blank JSONL line {number}")
94
- row = json.loads(line); semantics(row); rows.append(row)
95
- need(bool(rows) and len({r["id"] for r in rows}) == len(rows), "Nonempty split with unique IDs required")
96
- return rows
97
-
98
-
99
- def load_splits(train_path, dev_path, selection):
100
- train, dev = read_jsonl(train_path), read_jsonl(dev_path)
101
- need(not ({r["id"] for r in train} & {r["id"] for r in dev}), "TRAIN/DEV IDs overlap")
102
- need(not ({input_signature(r) for r in train} & {input_signature(r) for r in dev}),
103
- "TRAIN/DEV normalized complete inputs overlap")
104
- # When both fields are supplied, also reject same-source dependency components.
105
- components = lambda rows: {(r["source_id"], r["component_id"]) for r in rows
106
- if r.get("source_id") and r.get("component_id")}
107
- need(not (components(train) & components(dev)), "TRAIN/DEV source components overlap")
108
- if selection == "hard-accuracy":
109
- need(all(r.get("hard_target_id") is not None for r in dev),
110
- "Hard-accuracy selection requires explicit hard_target_id on EVERY DEV row")
111
- return train, dev
112
-
113
-
114
- def make_schedule(rows, epochs, logical_batch, seed, max_steps=None):
115
- steps, epoch_ends = [], []
116
- for epoch in range(epochs):
117
- order = sorted(range(len(rows)), key=lambda i: (digest([seed, epoch, rows[i]["id"]]), rows[i]["id"]))
118
- steps.extend([order[i:i + logical_batch] for i in range(0, len(order), logical_batch)])
119
- epoch_ends.append(len(steps))
120
- if max_steps is not None:
121
- need(1 <= max_steps <= len(steps), "max-steps must fit the declared epochs; no implicit extra epochs")
122
- steps = steps[:max_steps]
123
- return steps, sorted(set([0, len(steps)] + [s for s in epoch_ends if s <= len(steps)]))
124
-
125
-
126
- def admit(collator, rows):
127
- need(collator.max_length == 1024 and collator.state_truncation == "error", "Full1K collator required")
128
- evidence = {}
129
- for row in rows:
130
- e = collator.encode(model_row(row), labeled=True)
131
- kind, options, values, target = semantics(row)
132
- need(e["id"] == row["id"] and e["kind"] == kind and e["target"] == target
133
- and e["candidate_ids"] == [o["id"] for o in options] and e["values"] == values,
134
- "CLI schema and loaded native collator disagree")
135
- need(e["state_tokens_original"] == e["state_tokens_kept"] and e["input_tokens"] == len(e["ids"]) <= 1024,
136
- "Complete input exceeds1K or was truncated; no rows are silently dropped")
137
- evidence[row["id"]] = {"encoded_sha256": digest(e), "input_tokens": len(e["ids"]),
138
- "token_ids_sha256": digest(e["ids"]), "record_sha256": digest(row)}
139
- return evidence
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_finetune/metrics.py DELETED
@@ -1,61 +0,0 @@
1
- """Explicit DEV metrics/selectors; no external benchmark or hidden retention gates."""
2
- import math
3
- from collections import defaultdict
4
- from .data import need, semantics
5
-
6
-
7
- def summarize(rows, predictions):
8
- need(len(rows) == len(predictions) > 0, "Complete DEV predictions required")
9
- groups = defaultdict(list)
10
- for row, p in zip(rows, predictions):
11
- kind, options, values, target = semantics(row)
12
- ids = [o["id"] for o in options]; z = p["logits"]; probs = p["probabilities"]
13
- need(p["id"] == row["id"] and p["type"].lower() == kind and p["candidate_ids"] == ids,
14
- "Prediction ID/type/order mismatch")
15
- need(len(z) == len(probs) == len(ids) and all(math.isfinite(v) for v in z + probs)
16
- and all(0 <= v <= 1 for v in probs) and abs(sum(probs) - 1) <= 1e-5, "Invalid probabilities/logits")
17
- peak = max(z); logsum = peak + math.log(sum(math.exp(v - peak) for v in z))
18
- need(max(abs(math.exp(v - logsum) - p) for v, p in zip(z, probs)) <= 2e-5,
19
- "Probabilities disagree with saved logits")
20
- # NLL from logits avoids an arbitrary probability floor.
21
- nll = sum(t * (logsum - v) for t, v in zip(target, z))
22
- native_hat = max(range(len(probs)), key=probs.__getitem__)
23
- # Original typed evaluation convention, distinct from native first argmax at exact ties.
24
- hat = int(probs[1] >= .5) if kind == "noul" else native_hat
25
- item = {"nll": nll, "hard_accuracy": None if row.get("hard_target_id") is None else float(ids[hat] == row["hard_target_id"]),
26
- "native_argmax_accuracy": None if row.get("hard_target_id") is None else float(ids[native_hat] == row["hard_target_id"])}
27
- if kind == "score":
28
- pc = tc = rps = 0.
29
- for a, b in zip(probs[:-1], target[:-1]):
30
- pc += a; tc += b; rps += (pc - tc) ** 2
31
- item["rps"] = rps / (len(ids) - 1)
32
- item["expected_value_squared_error"] = (sum(a * v for a, v in zip(probs, values)) - sum(a * v for a, v in zip(target, values))) ** 2
33
- groups[kind].append(item)
34
- def report(items):
35
- result = {"rows": len(items), "soft_nll": sum(x["nll"] for x in items) / len(items)}
36
- result["hard_accuracy"] = (sum(x["hard_accuracy"] for x in items) / len(items)
37
- if all(x["hard_accuracy"] is not None for x in items) else None)
38
- result["native_argmax_accuracy_diagnostic"] = (sum(x["native_argmax_accuracy"] for x in items) / len(items)
39
- if all(x["native_argmax_accuracy"] is not None for x in items) else None)
40
- if all("rps" in x for x in items):
41
- result.update(rps=sum(x["rps"] for x in items) / len(items),
42
- expected_value_rmse=math.sqrt(sum(x["expected_value_squared_error"] for x in items) / len(items)))
43
- return result
44
- by_type = {kind: report(items) for kind, items in groups.items()}
45
- return {"by_type": by_type, "row": report([x for items in groups.values() for x in items]),
46
- "macro_soft_nll": sum(x["soft_nll"] for x in by_type.values()) / len(by_type),
47
- "macro_weighting": "equal weight for each type present in DEV; row mean within type",
48
- "hard_rule": "Noul p_yes>=0.5; Choice/Score original-order first argmax",
49
- "native_argmax_diagnostic": "original-order first argmax for every type; Noul exact tie no"}
50
-
51
-
52
- def better(candidate, best, selection, tolerance=1e-8):
53
- if best is None:
54
- return True
55
- a, b = candidate["metrics"], best["metrics"]
56
- if selection == "macro-nll":
57
- return a["macro_soft_nll"] < b["macro_soft_nll"] - tolerance
58
- need(selection == "hard-accuracy" and a["row"]["hard_accuracy"] is not None
59
- and b["row"]["hard_accuracy"] is not None, "Explicit complete hard labels required")
60
- delta = a["row"]["hard_accuracy"] - b["row"]["hard_accuracy"]
61
- return delta > tolerance or (abs(delta) <= tolerance and a["row"]["soft_nll"] < b["row"]["soft_nll"] - tolerance)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_finetune/run.py DELETED
@@ -1,237 +0,0 @@
1
- """Single-device training orchestration over the public API, with no research paths."""
2
- import gc
3
- import importlib.metadata
4
- import json
5
- import os
6
- from pathlib import Path
7
- import random
8
- import sys
9
- import time
10
-
11
- from . import data, metrics, state
12
-
13
-
14
- def source_identity():
15
- import decision_runtime
16
- roots = {"decision_finetune": Path(__file__).parent,
17
- "decision_runtime": Path(decision_runtime.__file__).parent}
18
- return {package + "/" + p.name: data.file_sha(p) for package, root in roots.items()
19
- for p in sorted(root.glob("*.py")) if p.name not in ("checks.py", "bridge_checks.py")}
20
-
21
-
22
- def setup(config):
23
- import numpy as np
24
- import torch
25
- data.need(torch.version.hip is not None and torch.cuda.is_available() and torch.cuda.device_count() == 1,
26
- "Expose exactly one real AMD ROCm device; CPU/NVIDIA training is unsupported")
27
- torch.cuda.set_device(0); torch.set_num_threads(config["cpu_threads"])
28
- torch.backends.cuda.matmul.allow_tf32 = False; torch.backends.cudnn.allow_tf32 = False
29
- torch.backends.mha.set_fastpath_enabled(False)
30
- torch.use_deterministic_algorithms(config["deterministic_algorithms"])
31
- torch.cuda.set_per_process_memory_fraction(min(1., config["max_reserved_gib"] * 2**30 / torch.cuda.get_device_properties(0).total_memory))
32
- random.seed(config["seed"]); np.random.seed(config["seed"]); torch.manual_seed(config["seed"])
33
- environment = {"python": sys.version.split()[0], "rocm": torch.version.hip,
34
- "gpu": torch.cuda.get_device_name(0),
35
- "versions": {n: importlib.metadata.version(n) for n in ("torch", "transformers", "tokenizers", "safetensors", "numpy")}}
36
- return torch, environment
37
-
38
-
39
- def check_memory(torch, config):
40
- data.need(torch.cuda.max_memory_reserved() <= config["max_reserved_gib"] * 2**30, "Reserved GPU memory cap exceeded")
41
-
42
-
43
- def evaluation(api, native, dev, admission, out, step, torch):
44
- # Parameter objects and optimizer state survive policy restoration unchanged.
45
- api.restore_native_policy_for_export(native)
46
- predictions = api.predict(native, [data.model_row(r) for r in dev], batch_size=8)
47
- for row, p in zip(dev, predictions):
48
- data.need(p["input_tokens"] == admission[row["id"]]["input_tokens"] <= 1024
49
- and p["state_tokens_original"] == p["state_tokens_kept"], "DEV admission mismatch")
50
- point = {"step": step, "metrics": metrics.summarize(dev, predictions)}
51
- state.write_json(out / f"evaluation-{step:06d}.json", point)
52
- with (out / f"predictions-{step:06d}.jsonl").open("x", encoding="utf-8") as stream:
53
- for p in predictions:
54
- stream.write(json.dumps(p, ensure_ascii=False, allow_nan=False) + "\n")
55
- api.configure_training(native, max_input_tokens=1024)
56
- return point
57
-
58
-
59
- def update(api, native, optimizer, rows, admission, config, step, total, torch):
60
- api.set_training_mode(native, True); optimizer.zero_grad(set_to_none=True)
61
- scale = state.learning_rate_scale(step, total, config["warmup_steps"])
62
- for group in optimizer.param_groups:
63
- base = config["encoder_lr"] if group["name"].endswith("encoder") else config["head_lr"]
64
- group["lr"] = state.learning_rate(step, total, config["warmup_steps"], base, config["lr_min"])
65
- sums = [0., 0., 0.]; tokens = padded = 0
66
- for start in range(0, len(rows), config["micro_batch_size"]):
67
- micro = rows[start:start + config["micro_batch_size"]]
68
- batch, encoded = native.collator([data.model_row(r) for r in micro], labeled=True, device="cuda:0")
69
- data.need(all(data.digest(e) == admission[r["id"]]["encoded_sha256"] for r, e in zip(micro, encoded)),
70
- "Actual training tokens/labels changed after complete admission")
71
- tokens += sum(e["input_tokens"] for e in encoded); padded += len(micro) * batch["input_ids"].shape[1]
72
- with torch.autocast("cuda", dtype=torch.bfloat16):
73
- logits = api.forward_for_training(native, batch)
74
- losses = api.loss_for_training(native, logits, batch, score_rps_weight=config["score_rps_weight"])
75
- (losses[0].sum() / len(rows)).backward()
76
- for i, loss in enumerate(losses):
77
- sums[i] += float(loss.detach().sum())
78
- del logits, losses, batch
79
- check_memory(torch, config)
80
- present = {r["question"]["type"].lower() for r in rows}
81
- proof = api.gradient_report(native, present_types=present) if step == 1 else None
82
- # AdamW must never decay a path absent from the logical batch.
83
- for group in optimizer.param_groups:
84
- for p in group["params"]:
85
- if group["name"].split(".")[0] not in present:
86
- data.need(p.grad is None, "Absent type has stale gradient")
87
- elif p.grad is not None:
88
- data.need(torch.isfinite(p.grad).all().item(), "Nonfinite gradient")
89
- active = [p for g in optimizer.param_groups for p in g["params"]]
90
- norm = float(torch.nn.utils.clip_grad_norm_(active, config["clip_norm"], error_if_nonfinite=True))
91
- optimizer.step(); optimizer.zero_grad(set_to_none=True)
92
- check_memory(torch, config)
93
- return {"step": step, "record_ids": [r["id"] for r in rows], "logical_rows": len(rows),
94
- "denominator": len(rows), "micro_batches": (len(rows) + config["micro_batch_size"] - 1) // config["micro_batch_size"],
95
- "mean_loss": sums[0] / len(rows), "mean_ce": sums[1] / len(rows), "mean_rps_all_types_diagnostic": sums[2] / len(rows),
96
- "effective_tokens": tokens, "padded_tokens": padded, "gradient_norm_before_clip": norm,
97
- "clipped": norm > config["clip_norm"], "lr_scale": scale,
98
- "learning_rates": {g["name"]: g["lr"] for g in optimizer.param_groups}, "first_step_gradient_proof": proof}
99
-
100
-
101
- def execute(args, config):
102
- import decision_runtime as api
103
- train, dev = data.load_splits(args.train, args.dev, config["selection"])
104
- schedule, default_points = data.make_schedule(train, config["epochs"], config["logical_batch_size"], config["seed"], config["max_steps"])
105
- points = default_points if config["eval_steps"] is None else config["eval_steps"]
106
- data.need(points == sorted(set(points)) and points[0] == 0 and points[-1] == len(schedule)
107
- and all(0 <= s <= len(schedule) for s in points), "Evaluation steps must be ordered, unique, and include0/final")
108
- config = {**config, "eval_steps": points, "total_steps": len(schedule)}
109
- data.need(args.stop_after_step is None or 1 <= args.stop_after_step < len(schedule),
110
- "A scheduling pause must be an original update before final; it does not shorten training")
111
- config["warmup_steps"] = (int(len(schedule) * config["warmup_ratio"]) if config["warmup_steps"] is None else config["warmup_steps"])
112
- data.need(0 <= config["warmup_steps"] < len(schedule), "Warmup must be shorter than the original full schedule")
113
- train_sha, dev_sha = data.file_sha(args.train), data.file_sha(args.dev)
114
- torch, environment = setup(config)
115
- identity = {"config": config, "sources": source_identity(), "native_manifest_sha256": args.manifest_sha256,
116
- "train_sha256": train_sha, "dev_sha256": dev_sha, "schedule_sha256": data.digest(schedule), "environment": environment}
117
- out = Path(args.output).resolve()
118
- data.need(not (out / "COMPLETE.json").exists(), "Completed runs are immutable")
119
- if args.resume:
120
- data.need(out.is_dir() and state.read_json(out / "RUN.json")["identity"] == identity,
121
- "Resume must use the original output/config/data/source/environment")
122
- data.need(Path(args.resume).resolve().is_relative_to(out / "attempts"), "Resume checkpoint must belong to this run")
123
- else:
124
- data.need(not out.exists(), "Fresh output required; use explicit --resume for an interrupted run")
125
- out.mkdir(parents=True); (out / "attempts").mkdir()
126
- state.write_json(out / "RUN.json", {"identity": identity, "train_path": str(Path(args.train).resolve()),
127
- "dev_path": str(Path(args.dev).resolve()), "native_path": str(Path(args.native).resolve()),
128
- "model_validation": "CLI integration not yet externally validated"})
129
- state.write_json(out / "SCHEDULE.json", {"train_ids": [r["id"] for r in train], "row_indices": schedule})
130
- import fcntl
131
- with (out / "LOCK").open("a") as lock:
132
- fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
133
- attempt = out / "attempts" / f"{len(list((out / 'attempts').iterdir())) + 1:04d}"
134
- attempt.mkdir(); started = time.monotonic()
135
- try:
136
- native = api.load_native(args.native, expected_manifest_sha256=args.manifest_sha256)
137
- # Same native packing; the user profile lowers only the admission cap.
138
- native.collator = type(native.collator)(native.collator.tokenizer, max_length=1024, state_truncation="error")
139
- admitted_train, admitted_dev = data.admit(native.collator, train), data.admit(native.collator, dev)
140
- admission = {"train": admitted_train, "dev": admitted_dev, "truncated": 0, "dropped": 0,
141
- "full_input_limit": 1024, "identity_sha256": data.digest(identity)}
142
- state.write_json(attempt / "ADMISSION.json", admission)
143
- api.configure_training(native, max_input_tokens=1024)
144
- frozen = api.frozen_snapshot(native)
145
- optimizer = torch.optim.AdamW(api.optimizer_groups(native, encoder_lr=config["encoder_lr"], head_lr=config["head_lr"]),
146
- weight_decay=config["weight_decay"], fused=False, foreach=False)
147
- layout = state.optimizer_layout(native.model, optimizer)
148
- parameter_ids = [id(p) for g in optimizer.param_groups for p in g["params"]]
149
- curve, best, step = [], None, 0
150
- if args.resume:
151
- meta, rng = state.load_checkpoint(args.resume, args.resume_sha256, native, optimizer, identity, torch)
152
- api.assert_frozen(native, frozen, content=True, versions=False)
153
- data.need(meta["frozen_hashes"] == {n: x["sha256"] for n, x in frozen.items()}, "Checkpoint frozen content differs from parent")
154
- frozen = api.frozen_snapshot(native)
155
- curve, best, step = meta["curve"], meta["best"], meta["step"]
156
- data.need(0 <= step <= len(schedule) and all(p["step"] <= step for p in curve), "Resume position invalid")
157
- data.need(args.stop_after_step is None or args.stop_after_step > step, "Pause must follow the resumed checkpoint")
158
- if best is not None and best.get("checkpoint_manifest_sha256") is None:
159
- data.need(Path(best["checkpoint"]).resolve() == Path(args.resume).resolve()
160
- and best["step"] == step, "Missing previous best checkpoint identity")
161
- best = {**best, "checkpoint_manifest_sha256": args.resume_sha256}
162
- state.restore_rng(torch, rng) # Last potentially RNG-sensitive setup operation.
163
- def checkpoint(current):
164
- nonlocal best
165
- directory = attempt / f"checkpoint-{current:06d}"
166
- if best is not None and best.get("checkpoint") is None:
167
- best = {**best, "checkpoint": str(directory.resolve())}
168
- api.assert_frozen(native, frozen, content=True)
169
- receipt = state.save_checkpoint(directory, native, optimizer,
170
- {"identity": identity, "step": current, "curve": curve, "best": best,
171
- "frozen_hashes": {n: x["sha256"] for n, x in frozen.items()}}, torch)
172
- if best is not None and best["checkpoint"] == str(directory.resolve()):
173
- best = {**best, "checkpoint_manifest_sha256": receipt["manifest_sha256"]}
174
- state.write_json(out / "LATEST.json", receipt)
175
- return receipt
176
- def evaluate(current):
177
- nonlocal best
178
- api.assert_frozen(native, frozen, content=True)
179
- opt_states = {id(p): id(s) for p, s in optimizer.state.items()}
180
- point = evaluation(api, native, dev, admitted_dev, attempt, current, torch)
181
- data.need(parameter_ids == [id(p) for g in optimizer.param_groups for p in g["params"]]
182
- and layout == state.optimizer_layout(native.model, optimizer)
183
- and opt_states == {id(p): id(s) for p, s in optimizer.state.items()}, "Evaluation broke optimizer identity")
184
- curve.append(point)
185
- if metrics.better(point, best, config["selection"], config["selection_tolerance"]):
186
- best = {**point, "checkpoint": None}
187
- checkpoint(current)
188
- if not args.resume:
189
- evaluate(0)
190
- with (attempt / "TRAIN_LOG.jsonl").open("x", encoding="utf-8") as log:
191
- for current in range(step + 1, len(schedule) + 1):
192
- report = update(api, native, optimizer, [train[i] for i in schedule[current - 1]], admitted_train, config, current, len(schedule), torch)
193
- api.assert_frozen(native, frozen)
194
- log.write(json.dumps(report, allow_nan=False) + "\n"); log.flush()
195
- if current in points:
196
- evaluate(current)
197
- elif config["save_every"] and current % config["save_every"] == 0:
198
- checkpoint(current)
199
- if current == args.stop_after_step:
200
- if current not in points and not (config["save_every"] and current % config["save_every"] == 0):
201
- checkpoint(current)
202
- paused = {"status": "PAUSED_DECISION_FINETUNE", "identity": identity, "step": current,
203
- "checkpoint": state.read_json(out / "LATEST.json"), "total_steps": len(schedule),
204
- "scheduling_pause_not_selection": True, "native_exported": False}
205
- state.write_json(attempt / "PAUSED.json", paused)
206
- return paused
207
- data.need([p["step"] for p in curve] == points and best is not None, "Missing evaluation/selected checkpoint")
208
- # A resume after final checkpoint can finish export without replaying training.
209
- del optimizer; gc.collect(); torch.cuda.empty_cache()
210
- from safetensors.torch import load_file
211
- selected_dir = Path(best["checkpoint"])
212
- selected_sha = best["checkpoint_manifest_sha256"]
213
- _, selected_meta = state.verify_checkpoint(selected_dir, selected_sha)
214
- data.need(selected_meta["identity"] == identity and selected_meta["step"] == best["step"], "Selected checkpoint identity")
215
- native.model.load_state_dict(load_file(str(selected_dir / "model.safetensors"), device="cpu"), strict=True)
216
- api.assert_frozen(native, frozen, content=True, versions=False)
217
- api.restore_native_policy_for_export(native)
218
- export_path = attempt / "selected-native"
219
- api.export_native(native, export_path, provenance={"cli": "decision_finetune", "identity": identity,
220
- "selected_step": best["step"], "selection": best["metrics"], "total_steps": len(schedule),
221
- "selected0_no_adaptation": best["step"] == 0, "checkpoint_manifest_sha256": selected_sha,
222
- "release_or_quality_qualification": False})
223
- data.need(data.file_sha(args.train) == train_sha and data.file_sha(args.dev) == dev_sha
224
- and source_identity() == identity["sources"], "Source/data changed during run")
225
- check_memory(torch, config)
226
- result = {"status": "COMPLETE_DECISION_FINETUNE", "identity": identity, "curve": curve,
227
- "selected": best, "selected0_no_adaptation": best["step"] == 0,
228
- "native": str(export_path), "native_manifest_sha256": data.file_sha(export_path / "MANIFEST.json"),
229
- "attempt_elapsed_seconds": time.monotonic() - started,
230
- "peak_reserved_bytes": torch.cuda.max_memory_reserved(), "resumed": bool(args.resume),
231
- "independent_fresh_reload_performed": False, "release_qualified": False}
232
- state.write_json(out / "COMPLETE.json", result)
233
- return result
234
- except BaseException as error:
235
- state.write_json(attempt / "FAILURE.json", {"exception_type": type(error).__name__, "message": str(error),
236
- "resume_requires_explicit_checkpoint": True})
237
- raise
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_finetune/state.py DELETED
@@ -1,110 +0,0 @@
1
- """Atomic update-boundary checkpoints with named optimizer mapping and RNG state."""
2
- import json
3
- import os
4
- from pathlib import Path
5
- import random
6
- import tempfile
7
- from .data import need, file_sha
8
-
9
-
10
- def write_json(path, value):
11
- path = Path(path)
12
- fd, temporary = tempfile.mkstemp(prefix="." + path.name + ".", suffix=".tmp", dir=path.parent)
13
- try:
14
- with os.fdopen(fd, "w", encoding="utf-8") as stream:
15
- json.dump(value, stream, ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
16
- stream.write("\n"); stream.flush(); os.fsync(stream.fileno())
17
- os.replace(temporary, path)
18
- finally:
19
- if os.path.exists(temporary):
20
- os.unlink(temporary)
21
-
22
-
23
- def read_json(path):
24
- return json.loads(Path(path).read_text(encoding="utf-8"))
25
-
26
-
27
- def optimizer_layout(model, optimizer):
28
- names = {id(p): name for name, p in model.named_parameters()}
29
- layout = [{"name": g["name"], "parameters": [names[id(p)] for p in g["params"]]}
30
- for g in optimizer.param_groups]
31
- flattened = [n for group in layout for n in group["parameters"]]
32
- need(len(set(flattened)) == len(flattened), "Optimizer parameter alias")
33
- return layout
34
-
35
-
36
- def rng_state(torch):
37
- import numpy as np
38
- n = np.random.get_state()
39
- return {"python": random.getstate(), "numpy": [n[0], n[1].tolist(), n[2], n[3], n[4]],
40
- "torch_cpu": torch.get_rng_state(), "torch_cuda": torch.cuda.get_rng_state_all()}
41
-
42
-
43
- def restore_rng(torch, saved):
44
- import numpy as np
45
- need(len(saved["torch_cuda"]) == torch.cuda.device_count(), "CUDA RNG device count changed")
46
- random.setstate(saved["python"])
47
- n = saved["numpy"]; np.random.set_state((n[0], np.asarray(n[1], dtype=np.uint32), n[2], n[3], n[4]))
48
- torch.set_rng_state(saved["torch_cpu"]); torch.cuda.set_rng_state_all(saved["torch_cuda"])
49
-
50
-
51
- def save_checkpoint(path, native, optimizer, metadata, torch):
52
- from safetensors.torch import save_file
53
- path = Path(path); temporary = path.with_name(path.name + ".tmp")
54
- need(not path.exists() and not temporary.exists(), "Checkpoint destinations are immutable")
55
- temporary.mkdir()
56
- # CPU copies are serialization only; no CPU model or forward is constructed.
57
- save_file({n: t.detach().cpu().contiguous() for n, t in native.model.state_dict().items()}, str(temporary / "model.safetensors"))
58
- torch.save({"optimizer": optimizer.state_dict(), "rng": rng_state(torch)}, temporary / "state.pt")
59
- write_json(temporary / "metadata.json", {**metadata, "optimizer_layout": optimizer_layout(native.model, optimizer)})
60
- files = {p.name: {"sha256": file_sha(p), "bytes": p.stat().st_size} for p in sorted(temporary.iterdir())}
61
- write_json(temporary / "MANIFEST.json", {"schema": "decision.finetune.checkpoint.v1", "files": files})
62
- os.replace(temporary, path)
63
- return {"path": str(path.resolve()), "manifest_sha256": file_sha(path / "MANIFEST.json")}
64
-
65
-
66
- def verify_checkpoint(path, expected_sha):
67
- path = Path(path).resolve(strict=True)
68
- need(file_sha(path / "MANIFEST.json") == expected_sha, "Explicit checkpoint manifest SHA mismatch")
69
- manifest = read_json(path / "MANIFEST.json")
70
- need(manifest["schema"] == "decision.finetune.checkpoint.v1"
71
- and set(manifest["files"]) == {"model.safetensors", "state.pt", "metadata.json"}, "Checkpoint roster mismatch")
72
- need({p.name for p in path.iterdir()} == set(manifest["files"]) | {"MANIFEST.json"}, "Unexpected checkpoint file")
73
- for name, ref in manifest["files"].items():
74
- p = path / name
75
- need(not p.is_symlink() and p.is_file() and p.stat().st_size == ref["bytes"] and file_sha(p) == ref["sha256"], "Checkpoint file changed")
76
- return path, read_json(path / "metadata.json")
77
-
78
-
79
- def load_checkpoint(path, expected_sha, native, optimizer, identity, torch):
80
- from safetensors.torch import load_file
81
- path, meta = verify_checkpoint(path, expected_sha)
82
- need(meta["identity"] == identity, "Resume source/data/native/config/schedule/environment identity changed")
83
- need(meta["optimizer_layout"] == optimizer_layout(native.model, optimizer), "Named optimizer group/order mismatch")
84
- parameter_ids = [id(p) for p in native.model.parameters()]
85
- native.model.load_state_dict(load_file(str(path / "model.safetensors"), device="cpu"), strict=True)
86
- need([id(p) for p in native.model.parameters()] == parameter_ids, "Loading replaced parameters")
87
- # Only trust locally produced, explicitly hash-verified checkpoints. No arbitrary pickle globals.
88
- state = torch.load(path / "state.pt", map_location="cpu", weights_only=True)
89
- optimizer.load_state_dict(state["optimizer"])
90
- need(meta["optimizer_layout"] == optimizer_layout(native.model, optimizer), "Restored optimizer mapping drift")
91
- for p, item in optimizer.state.items():
92
- need(all(item[k].shape == p.shape and torch.isfinite(item[k]).all().item() for k in ("exp_avg", "exp_avg_sq")), "Invalid AdamW moments")
93
- need(0 < float(item["step"]) <= meta["step"], "Invalid per-parameter AdamW step")
94
- return meta, state["rng"]
95
-
96
-
97
- def learning_rate_scale(step, total, warmup):
98
- import math
99
- need(1 <= step <= total and 0 <= warmup < total, "Invalid original LR horizon")
100
- if warmup and step <= warmup:
101
- return step / warmup
102
- progress = ((step - warmup) / (total - warmup) if warmup
103
- else (step - 1) / (total - 1) if total > 1 else 0.)
104
- return .5 * (1 + math.cos(math.pi * progress))
105
-
106
-
107
- def learning_rate(step, total, warmup, base, minimum):
108
- need(0 <= minimum <= base, "LR floor must fit every optimizer group")
109
- scale = learning_rate_scale(step, total, warmup)
110
- return base * scale if warmup and step <= warmup else minimum + (base - minimum) * scale
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_finetune/system_one.py DELETED
@@ -1,34 +0,0 @@
1
- """Keep System One inference and fine-tuning input semantics identical."""
2
- import copy
3
- import json
4
-
5
-
6
- def system_one_training_rows(request, targets, *, request_id, source_id,
7
- component_id, hard_target_ids=None):
8
- """Convert a typed request plus separate native targets to training rows.
9
-
10
- targets is keyed by question ID. Noul uses {'probability': p_yes}, Choice
11
- {'choice_id': label} or {'probabilities': [...]}, Score {'probabilities': [...]}.
12
- The caller assigns shared source/component IDs across related examples so
13
- existing TRAIN/DEV validation can reject component overlap.
14
- """
15
- from decision_inference._system_one import system_one_records
16
- from .data import semantics
17
- for value in (request_id, source_id, component_id):
18
- if not isinstance(value, str) or not value.strip():
19
- raise ValueError("Explicit nonempty request/source/component IDs required")
20
- rows = system_one_records(request)
21
- qids = {row["question"]["id"] for row in rows}
22
- if not isinstance(targets, dict) or set(targets) != qids:
23
- raise ValueError("Exactly one target per question ID is required")
24
- if hard_target_ids is not None and (not isinstance(hard_target_ids, dict) or set(hard_target_ids) != qids):
25
- raise ValueError("Hard evaluation labels must cover exactly the question IDs")
26
- for i, row in enumerate(rows):
27
- qid = row["question"]["id"]
28
- row.update(id=json.dumps([request_id, i], ensure_ascii=False, separators=(",", ":")),
29
- source_id=source_id, component_id=component_id,
30
- target=copy.deepcopy(targets[qid]))
31
- if hard_target_ids is not None:
32
- row["hard_target_id"] = hard_target_ids[qid]
33
- semantics(row)
34
- return rows
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_inference/__init__.py DELETED
@@ -1,5 +0,0 @@
1
- from .profile import Complete1KCollator, MAX_INPUT_TOKENS, predict_1k
2
- from ._auto import predict_auto_1k
3
- from ._system_one import SystemOne, system_one_records
4
-
5
- __all__ = ["Complete1KCollator", "MAX_INPUT_TOKENS", "predict_1k", "predict_auto_1k", "SystemOne", "system_one_records"]
 
 
 
 
 
 
decision_inference/_auto.py DELETED
@@ -1,34 +0,0 @@
1
- """Optional homogeneous FP32 batching, with no increase in total padding."""
2
- from dataclasses import replace
3
- from ._request import request_collator, predict_1k
4
-
5
-
6
- def _padded_tokens(entries, size):
7
- return sum(len(chunk) * max(row['input_tokens'] for row in chunk)
8
- for start in range(0, len(entries), size)
9
- for chunk in [entries[start:start + size]])
10
-
11
-
12
- def _capacity(entries):
13
- if (len(entries) > 8 and len({row['kind'] for row in entries}) == 1
14
- and _padded_tokens(entries, 32) <= _padded_tokens(entries, 8)):
15
- return 32
16
- return 8
17
-
18
-
19
- def predict_auto_1k(native, records):
20
- """Opt in to at most32 consecutive rows; same complete1024 FP32 contract.
21
-
22
- Default predict_1k remains unchanged. Mixed requests or extra padding retain
23
- B8. Every input admits first; errors never trigger partial outputs/retries.
24
- This changes physical shapes and can change FP32 rounding, not task semantics.
25
- """
26
- records = list(records)
27
- if len(records) <= 8:
28
- return predict_1k(native, records, batch_size=8)
29
- guard = request_collator(native.collator, records)
30
- from decision_runtime import predict
31
- result = predict(replace(native, collator=guard), records,
32
- batch_size=_capacity(guard._encoded))
33
- guard.finish()
34
- return result
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_inference/_grouped.py DELETED
@@ -1,16 +0,0 @@
1
- """Stable typed scheduling behind SystemOne; preserve request and answer order."""
2
- from dataclasses import replace
3
-
4
- def predict_grouped_1k(native,records,*,batch_size=8):
5
- from decision_inference._request import request_collator
6
- from decision_runtime import predict
7
- if type(batch_size) is not int or batch_size!=8:raise ValueError('SystemOne typed scheduling uses physical batch size 8')
8
- records=list(records);guard=request_collator(native.collator,records)
9
- # Admit every original row before scheduling or any forward.
10
- order=sorted(range(len(records)),key=lambda i:records[i]['question']['type'])
11
- sorted_rows=[records[i]for i in order]
12
- guard._records=tuple(sorted_rows);guard._encoded=[guard._encoded[i]for i in order]
13
- predictions=predict(replace(native,collator=guard),sorted_rows,batch_size=8);guard.finish();restored=[None]*len(records)
14
- for i,result in zip(order,predictions):restored[i]=result
15
- if any(x is None for x in restored):raise RuntimeError('Incomplete scheduled predictions')
16
- return restored
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_inference/_request.py DELETED
@@ -1,58 +0,0 @@
1
- """Isolated complete-1K request-local encoding reuse; native arithmetic is untouched."""
2
- from dataclasses import replace
3
-
4
- MAX_INPUT_TOKENS=1024
5
-
6
- def request_collator(native_collator,records):
7
- """Encode each occurrence once before any model call; never cache by external ID."""
8
- class RequestEncodedCollator(type(native_collator)):
9
- def __init__(self):
10
- super().__init__(native_collator.tokenizer,max_length=MAX_INPUT_TOKENS,state_truncation='error')
11
- self._records=tuple(records);self._encoded=[];self._cursor=0;self._active=None
12
- for row in self._records:
13
- encoded=super().encode(row,labeled=False)
14
- if (encoded['input_tokens']>MAX_INPUT_TOKENS
15
- or encoded['state_tokens_original']!=encoded['state_tokens_kept']):
16
- raise ValueError('Native collator violated the complete-input 1K profile')
17
- self._encoded.append(encoded)
18
-
19
- def encode(self,record,labeled=False):
20
- # Only the unchanged native tensor assembler consumes this request-local queue.
21
- if labeled or self._active is None:
22
- raise ValueError('Request-local collator supports only its admitted inference batch')
23
- expected,encoded=next(self._active)
24
- if record is not expected:
25
- raise ValueError('Request-local record occurrence order changed')
26
- return encoded
27
-
28
- def __call__(self,records,labeled=False,device='cpu'):
29
- records=list(records);end=self._cursor+len(records)
30
- if labeled or end>len(self._records) or any(a is not b for a,b in zip(records,self._records[self._cursor:end])):
31
- raise ValueError('Request-local record occurrence order changed')
32
- self._active=iter(zip(records,self._encoded[self._cursor:end]))
33
- try:
34
- # Original allocation, padding, targets/values, kinds and device transfer.
35
- result=super().__call__(records,labeled=False,device=device)
36
- finally:
37
- self._active=None
38
- self._cursor=end
39
- return result
40
-
41
- def finish(self):
42
- if self._cursor!=len(self._records):
43
- raise ValueError('Native prediction did not consume the admitted request')
44
- return RequestEncodedCollator()
45
-
46
- def predict_1k(native,records,*,batch_size=8):
47
- """Same native prediction/output/errors; one encoding per occurrence per call.
48
-
49
- No model mode/inventory check is skipped. No result, token array or record is
50
- retained across API calls. Same IDs with different inputs remain distinct.
51
- """
52
- if type(batch_size) is not int or not 1<=batch_size<=8:
53
- raise ValueError('The product profile permits integer batch sizes 1..8')
54
- records=list(records);guard=request_collator(native.collator,records)
55
- from decision_runtime import predict
56
- result=predict(replace(native,collator=guard),records,batch_size=batch_size)
57
- guard.finish()
58
- return result
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_inference/_system_one.py DELETED
@@ -1,217 +0,0 @@
1
- """System One request/answer schema over complete-input Decision inference.
2
-
3
- Pure input conversion is shared with fine-tuning. Question IDs are bookkeeping;
4
- Choice labels are semantic. No chat prompts, generated JSON, or cross-call cache.
5
- """
6
- import copy
7
- import json
8
- import math
9
-
10
- MAX_QUESTIONS = 128
11
- MAX_REQUESTS = 128
12
- MAX_DECISIONS = 512
13
- MAX_REQUEST_BYTES = 2 * 1024 * 1024
14
- PUBLIC_MODELS = {
15
- "Decision-1.0-Kai": "da603662bc57e89ccfb51c972ed9c1f2825f267597353cf1337df9117a3dfabe",
16
- "Decision-1.0-Lex": "f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6",
17
- }
18
-
19
-
20
- def _identifier(value, label):
21
- if not isinstance(value, str) or not value.strip() or len(value) > 128:
22
- raise ValueError(label + " must be a nonempty string of at most 128 characters")
23
- return value
24
-
25
-
26
- def _json(value):
27
- # JSON objects must have string keys: never silently coerce Python keys.
28
- def check(item):
29
- if isinstance(item, dict):
30
- if not all(isinstance(k, str) for k in item):
31
- raise ValueError("JSON object keys must be strings")
32
- for v in item.values():
33
- check(v)
34
- elif isinstance(item, list):
35
- for v in item:
36
- check(v)
37
- elif item is not None and not isinstance(item, (str, bool, int, float)):
38
- raise ValueError("Only JSON values are supported")
39
- try:
40
- check(value)
41
- return json.dumps(value, ensure_ascii=False, sort_keys=True,
42
- separators=(",", ":"), allow_nan=False)
43
- except (TypeError, RecursionError, UnicodeError) as exc:
44
- raise ValueError("Invalid JSON content") from exc
45
-
46
-
47
- def _content(value, label):
48
- if isinstance(value, str):
49
- if not value.strip():
50
- raise ValueError(label + " must not be empty")
51
- return value
52
- if isinstance(value, (dict, list)):
53
- return _json(value)
54
- raise ValueError(label + " must be text, an object, or an array")
55
-
56
-
57
- def system_one_records(request):
58
- """Validate one wire request and return native rows, without loading a model.
59
-
60
- Full token admission occurs in predict_1k before the first model forward.
61
- External record IDs should be made unique when combining training examples.
62
- """
63
- if not isinstance(request, dict) or set(request) != {"model", "state", "questions"}:
64
- raise ValueError("A request contains exactly model, state and questions")
65
- _identifier(request["model"], "Model")
66
- if len(_json(request).encode("utf-8")) > MAX_REQUEST_BYTES:
67
- raise ValueError("Request exceeds 2 MiB; no input is truncated")
68
- state = _content(request["state"], "State")
69
- questions = request["questions"]
70
- if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS:
71
- raise ValueError("Provide 1..128 named questions")
72
- rows = []
73
- for index, (qid, item) in enumerate(questions.items()):
74
- _identifier(qid, "Question ID")
75
- if (not isinstance(item, dict) or set(item) - {"type", "instructions", "criteria"}
76
- or not {"type", "instructions"} <= set(item)):
77
- raise ValueError(qid + ": use type, instructions and optional criteria")
78
- kind = item["type"]
79
- if kind not in ("noul", "choice", "score"):
80
- raise ValueError(qid + ": type must be noul, choice or score")
81
- q = {"id": qid, "type": kind.capitalize(),
82
- "text": _content(item["instructions"], qid + ".instructions")}
83
- criteria = item.get("criteria")
84
- if kind == "choice":
85
- if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255:
86
- raise ValueError(qid + ": Choice requires 2..255 named options")
87
- q["options"] = []
88
- for name, description in criteria.items():
89
- _identifier(name, "Choice option")
90
- text = name if description is None else name + ": " + _content(description, qid + ".criteria")
91
- q["options"].append({"id": name, "text": text})
92
- elif kind == "score":
93
- if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
94
- raise ValueError(qid + ": Score requires 2..10 ordered levels")
95
- q["levels"] = [{"id": str(i), "value": i, "text": _content(v, qid + ".criteria")}
96
- for i, v in enumerate(criteria)]
97
- elif "criteria" in item:
98
- if not isinstance(criteria, dict) or set(criteria) - {"false", "true"}:
99
- raise ValueError(qid + ": Noul criteria accept false and true only")
100
- for key in ("false", "true"):
101
- if key in criteria:
102
- q[key + "_criterion"] = _content(criteria[key], qid + ".criteria." + key)
103
- rows.append({"id": "systemone:" + str(index), "state_text": state, "question": q})
104
- return rows
105
-
106
-
107
- def _answer(row, prediction):
108
- q = row["question"]
109
- kind = q["type"].lower()
110
- ids = (["no", "yes"] if kind == "noul" else
111
- [v["id"] for v in q["options" if kind == "choice" else "levels"]])
112
- if (prediction.get("id") != row["id"] or prediction.get("question_id") != q["id"]
113
- or prediction.get("type") != q["type"] or prediction.get("candidate_ids") != ids
114
- or type(prediction.get("input_tokens")) is not int
115
- or not 1 <= prediction["input_tokens"] <= 1024
116
- or type(prediction.get("state_tokens_original")) is not int
117
- or prediction["state_tokens_original"] < 0
118
- or prediction["state_tokens_original"] != prediction.get("state_tokens_kept")):
119
- raise RuntimeError("Prediction identity or complete-input profile mismatch")
120
- p = prediction.get("probabilities")
121
- if (not isinstance(p, list) or len(p) != len(ids)
122
- or not all(type(v) in (int, float) and math.isfinite(v) and 0 <= v <= 1 for v in p)
123
- or abs(sum(p) - 1) > 2e-5):
124
- raise RuntimeError("Invalid prediction probabilities")
125
- answer = {"type": kind}
126
- if kind == "noul":
127
- if prediction.get("probability") != p[1]:
128
- raise RuntimeError("Native Noul probability mismatch")
129
- answer["noul"] = p[1]
130
- return answer
131
- best = ids[max(range(len(p)), key=p.__getitem__)]
132
- if prediction.get("choice_id") != best or prediction.get("confidence") != max(p):
133
- raise RuntimeError("Native Choice/confidence mismatch")
134
- answer.update(probabilities=dict(zip(ids, p)), confidence=prediction["confidence"])
135
- if kind == "choice":
136
- answer["choice"] = best
137
- else:
138
- score = prediction.get("score")
139
- if (type(score) not in (int, float) or not math.isfinite(score)
140
- or abs(score - sum(i * v for i, v in enumerate(p))) > 2e-5):
141
- raise RuntimeError("Native ordinal Score mismatch")
142
- # Preserve native FP32 arithmetic, not a new CPU reduction.
143
- answer["score"] = score
144
- answer["legend"] = {v["id"]: v["text"] for v in q["levels"]}
145
- return answer
146
-
147
-
148
- class SystemOne:
149
- """Local System One API for a loaded Kai, Lex or compatible fine-tune.
150
-
151
- evaluate(request) accepts the HTTP body shape; system_one(**request) is its
152
- Python equivalent. batch(requests) flattens independent states into GPU
153
- batches and restores the original request/question order. Default B8 groups rows by decision type;
154
- batching='auto' opts into the published homogeneous padding-aware B32 path.
155
- """
156
- def __init__(self, native, *, model=None, batching="default"):
157
- if model is None:
158
- model = next((name for name, sha in PUBLIC_MODELS.items()
159
- if sha == native.manifest_sha256), None)
160
- _identifier(model, "Model (required for a custom fine-tune)")
161
- # Do not let a different loaded checkpoint claim a published identity.
162
- if model in PUBLIC_MODELS and native.manifest_sha256 != PUBLIC_MODELS[model]:
163
- raise ValueError("Loaded checkpoint does not match the public model name")
164
- if batching not in ("default", "auto"):
165
- raise ValueError("batching must be default or auto")
166
- self.native, self.model, self.batching = native, model, batching
167
-
168
- def system_one(self, *, state, questions, model=None):
169
- return self.evaluate({"model": self.model if model is None else model,
170
- "state": state, "questions": questions})
171
-
172
- def evaluate(self, request):
173
- return self.batch([request])[0]
174
-
175
- def batch(self, requests):
176
- if not isinstance(requests, list) or not 1 <= len(requests) <= MAX_REQUESTS:
177
- raise ValueError("Provide 1..128 request objects")
178
- if len(_json(requests).encode("utf-8")) > MAX_REQUEST_BYTES:
179
- raise ValueError("Combined request exceeds 2 MiB")
180
- # Detach mutable caller inputs before conversion/admission/inference.
181
- requests = copy.deepcopy(requests)
182
- groups = []
183
- for request in requests:
184
- rows = system_one_records(request)
185
- if request["model"] != self.model:
186
- raise ValueError("Request model does not match this loaded model")
187
- groups.append(rows)
188
- count = sum(map(len, groups))
189
- if count > MAX_DECISIONS:
190
- raise ValueError("Provide at most 512 decisions in one batch")
191
- records, slots = [], []
192
- # Question-major order permits the same question across many states to
193
- # share a physical batch; external IDs never decide caching or grouping.
194
- for qi in range(max(map(len, groups))):
195
- for ri, group in enumerate(groups):
196
- if qi < len(group):
197
- row = group[qi]
198
- row["id"] = f"systemone:{ri}:{qi}"
199
- records.append(row)
200
- slots.append((ri, qi))
201
- if self.batching == "auto":
202
- from ._auto import predict_auto_1k
203
- predictions = predict_auto_1k(self.native, records)
204
- else:
205
- from ._grouped import predict_grouped_1k
206
- predictions = predict_grouped_1k(self.native, records, batch_size=8)
207
- if len(predictions) != len(records):
208
- raise RuntimeError("Incomplete model result; no partial answers returned")
209
- values = [[None] * len(group) for group in groups]
210
- tokens = [0] * len(groups)
211
- for row, prediction, (ri, qi) in zip(records, predictions, slots):
212
- values[ri][qi] = _answer(row, prediction)
213
- tokens[ri] += prediction["input_tokens"]
214
- return [{"model": self.model,
215
- "answers": {row["question"]["id"]: answer for row, answer in zip(group, values[ri])},
216
- "usage": {"input_tokens": tokens[ri], "output_tokens": 0}}
217
- for ri, group in enumerate(groups)]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_inference/profile.py DELETED
@@ -1,51 +0,0 @@
1
- """Complete-input 1K product profile around an unchanged native runtime."""
2
- from dataclasses import replace
3
-
4
- MAX_INPUT_TOKENS = 1024
5
-
6
-
7
- class Complete1KCollator:
8
- """Use the native collator's actual length error, never truncate fields.
9
-
10
- Constructible with a tokenizer before any model is loaded. A whole request
11
- is admitted before predict, and every physical batch before tensor assembly.
12
- """
13
- def __init__(self, native_collator):
14
- self.base = type(native_collator)(native_collator.tokenizer,
15
- max_length=MAX_INPUT_TOKENS,
16
- state_truncation="error")
17
- self.tokenizer = self.base.tokenizer
18
- self.marker, self.pad = self.base.marker, self.base.pad
19
- self.tensor_batches = 0
20
-
21
- def tokens(self, text):
22
- return self.base.tokens(text)
23
-
24
- def encode(self, row, labeled=False):
25
- encoded = self.base.encode(row, labeled=labeled)
26
- if (encoded["input_tokens"] > MAX_INPUT_TOKENS
27
- or encoded["state_tokens_original"] != encoded["state_tokens_kept"]):
28
- raise ValueError("Native collator violated the complete-input 1K profile")
29
- return encoded
30
-
31
- def admit(self, records):
32
- return [self.encode(row, labeled=False) for row in records]
33
-
34
- def __call__(self, records, labeled=False, device="cpu"):
35
- records = list(records)
36
- # All encodes finish before delegating any tensor allocation.
37
- for row in records:
38
- self.encode(row, labeled=labeled)
39
- self.tensor_batches += 1
40
- return self.base(records, labeled=labeled, device=device)
41
-
42
-
43
-
44
- def predict_1k(native, records, *, batch_size=8):
45
- """Admit every complete input, then reuse its encoding within this request.
46
-
47
- Native prediction, batch boundaries and outputs are unchanged. Encodings
48
- are released with the synchronous call; no state activations are cached.
49
- """
50
- from ._request import predict_1k as predict_admitted
51
- return predict_admitted(native, records, batch_size=batch_size)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_runtime/__init__.py DELETED
@@ -1,11 +0,0 @@
1
- """Source-only portable training adapter; real AMD validation is still pending."""
2
- from .native import load_native, export_native, predict
3
- from .training import (configure_training, set_training_mode, forward_for_training,
4
- loss_for_training, optimizer_groups, inventory,
5
- frozen_snapshot, assert_frozen, gradient_report,
6
- restore_native_policy_for_export)
7
-
8
- __all__ = ["load_native", "export_native", "predict", "configure_training",
9
- "set_training_mode", "forward_for_training", "loss_for_training",
10
- "optimizer_groups", "inventory", "frozen_snapshot", "assert_frozen",
11
- "gradient_report", "restore_native_policy_for_export"]
 
 
 
 
 
 
 
 
 
 
 
 
decision_runtime/_compat.py DELETED
@@ -1,20 +0,0 @@
1
- """Supported self-contained runtime revision; weights may differ. No path bindings."""
2
- RUNTIME_SHA256 = {'__init__.py': '1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010',
3
- 'artifacts.py': 'c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3',
4
- 'contract.py': '51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726',
5
- 'infer.py': '9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315',
6
- 'model.py': '8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b',
7
- 'packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
8
- 'policy/__init__.py': '0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114',
9
- 'policy/artifacts.py': '5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2',
10
- 'policy/contract.py': '0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67',
11
- 'policy/infer.py': '672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896',
12
- 'policy/model.py': 'eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca',
13
- 'policy/packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
14
- 'policy/reference/__init__.py': 'c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28',
15
- 'policy/reference/artifacts.py': '3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159',
16
- 'policy/reference/infer.py': '672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896',
17
- 'policy/reference/model.py': '733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26',
18
- 'policy/reference/modernbert_sdpa_layout.py': '0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8',
19
- 'policy/reference/packing.py': 'f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819',
20
- 'training_policy.py': '6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5'}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_runtime/native.py DELETED
@@ -1,152 +0,0 @@
1
- """Load the export's own verified runtime; never import an experiment package."""
2
- from dataclasses import dataclass, field
3
- from pathlib import Path, PurePosixPath
4
- import copy
5
- import hashlib
6
- import importlib
7
- import importlib.util
8
- import json
9
- import re
10
- import sys
11
- import uuid
12
-
13
- from ._compat import RUNTIME_SHA256
14
-
15
-
16
- def sha256(path):
17
- digest = hashlib.sha256()
18
- with Path(path).open("rb") as stream:
19
- for chunk in iter(lambda: stream.read(1 << 20), b""):
20
- digest.update(chunk)
21
- return digest.hexdigest()
22
-
23
-
24
- def verify_files(directory, expected_manifest_sha256):
25
- """Stdlib pre-import verification, including every weight and runtime file."""
26
- root = Path(directory).resolve(strict=True)
27
- if not re.fullmatch(r"[0-9a-f]{64}", expected_manifest_sha256 or ""):
28
- raise ValueError("An explicit trusted MANIFEST SHA256 is required")
29
- if sha256(root / "MANIFEST.json") != expected_manifest_sha256:
30
- raise ValueError("Native manifest identity mismatch")
31
- manifest = json.loads((root / "MANIFEST.json").read_text())
32
- files = manifest.get("files")
33
- if manifest.get("schema") != "decision.files.v1" or not isinstance(files, dict):
34
- raise ValueError("Unsupported native manifest")
35
- for name, ref in files.items():
36
- if not isinstance(name, str) or not isinstance(ref, dict):
37
- raise ValueError("Invalid manifest entry")
38
- rel = PurePosixPath(name)
39
- if (rel.is_absolute() or ".." in rel.parts
40
- or rel.as_posix() != name or "\\" in name or not rel.parts):
41
- raise ValueError("Unsafe manifest path")
42
- path = root / name
43
- if (not path.is_file() or path.is_symlink()
44
- or set(ref) != {"bytes", "sha256"}
45
- or path.stat().st_size != ref["bytes"] or sha256(path) != ref["sha256"]):
46
- raise ValueError("Native file mismatch: " + name)
47
- actual = set()
48
- for path in root.rglob("*"):
49
- if path.is_symlink():
50
- raise ValueError("Materialized native required; symlinks are not supported")
51
- rel = path.relative_to(root)
52
- if path.is_file() and rel.as_posix() != "MANIFEST.json":
53
- if "__pycache__" in rel.parts and path.suffix == ".pyc":
54
- continue
55
- actual.add(rel.as_posix())
56
- if actual != set(files):
57
- raise ValueError("Native file roster mismatch")
58
- runtime = {name: ref["sha256"] for name, ref in files.items() if name.endswith(".py")}
59
- if runtime != RUNTIME_SHA256:
60
- raise ValueError("Unsupported runtime revision; architecture name alone is insufficient")
61
- return root, manifest
62
-
63
-
64
- @dataclass
65
- class Native:
66
- directory: Path
67
- manifest_sha256: str
68
- model: object
69
- collator: object
70
- config: dict
71
- model_api: object
72
- contract: object
73
- artifacts: object
74
- training_state: dict | None = None
75
- last_training_provenance: dict | None = None
76
- package_name: str = field(default="", repr=False)
77
-
78
-
79
- def _amd_device(device):
80
- import torch
81
- selected = torch.device(device)
82
- if selected.type != "cuda" or torch.version.hip is None or not torch.cuda.is_available():
83
- raise ValueError("This runtime requires a real ROCm CUDA device; no CPU fallback")
84
- if selected.index is None:
85
- selected = torch.device("cuda", torch.cuda.current_device())
86
- return selected
87
-
88
-
89
- def load_native(directory, *, expected_manifest_sha256, device="cuda:0"):
90
- """Load any compatible weight export with the pinned public runtime revision.
91
-
92
- Verification precedes executing bundled Python. This function loads a model;
93
- importing decision_runtime or running its stdlib checks does not.
94
- """
95
- root, manifest = verify_files(directory, expected_manifest_sha256)
96
- selected = _amd_device(device)
97
- namespace = "_decision_native_" + uuid.uuid4().hex
98
- spec = importlib.util.spec_from_file_location(namespace, root / "__init__.py",
99
- submodule_search_locations=[str(root)])
100
- package = importlib.util.module_from_spec(spec)
101
- sys.modules[namespace] = package
102
- try:
103
- spec.loader.exec_module(package)
104
- artifacts = importlib.import_module(namespace + ".artifacts")
105
- contract = importlib.import_module(namespace + ".contract")
106
- api = importlib.import_module(namespace + ".model")
107
- declared = contract.validate_config(json.loads((root / "decision_config.json").read_text()))
108
- if (declared["arm"], declared["training_arm"]) != ("all22", "S22"):
109
- raise ValueError("Only the complete three-path all22/S22 architecture is supported")
110
- model, collator, cfg = artifacts.load_export(root, device=str(selected))
111
- if (cfg["arm"], cfg["training_arm"]) != ("all22", "S22"):
112
- raise ValueError("Only the complete three-path all22/S22 architecture is supported")
113
- if type(model) is not api.DecisionModel or artifacts.verify_native(root) != (manifest, cfg):
114
- raise ValueError("Loaded class or native identity mismatch")
115
- return Native(root, expected_manifest_sha256, model, collator, cfg, api,
116
- contract, artifacts, package_name=namespace)
117
- except BaseException:
118
- for name in tuple(sys.modules):
119
- if name == namespace or name.startswith(namespace + "."):
120
- del sys.modules[name]
121
- raise
122
-
123
-
124
- def _native_mode(native):
125
- if native.training_state is not None:
126
- raise ValueError("Restore native policy before native inference or export")
127
- native.model.verify_inventory(check_values=False)
128
-
129
-
130
- def predict(native, records, *, batch_size=8):
131
- """Unchanged bundled inference, FP32; no training graph is used here."""
132
- _native_mode(native)
133
- device = next(native.model.parameters()).device
134
- _amd_device(device)
135
- return native.model_api.predict(native.model, native.collator, records,
136
- batch_size=batch_size, device=str(device), precision="fp32")
137
-
138
-
139
- def export_native(native, directory, *, provenance):
140
- """Use the original strict exporter after explicit policy restoration."""
141
- _native_mode(native)
142
- if native.last_training_provenance is None or not isinstance(provenance, dict):
143
- raise ValueError("Restored training provenance and an explicit user provenance dict required")
144
- if Path(directory).exists():
145
- raise ValueError("Export destination must be fresh")
146
- details = {"parent_native_manifest_sha256": native.manifest_sha256,
147
- "parent_provenance": copy.deepcopy(native.config.get("provenance", {})),
148
- "training_adapter": copy.deepcopy(native.last_training_provenance),
149
- "user": copy.deepcopy(provenance)}
150
- json.dumps(details, allow_nan=False)
151
- return native.artifacts.export_model(native.model, native.collator.tokenizer,
152
- directory, native.config, details)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
decision_runtime/training.py DELETED
@@ -1,310 +0,0 @@
1
- """Explicit all-types training graph for the unchanged three-path native."""
2
- import math
3
- from pathlib import Path
4
- from .native import sha256, _amd_device
5
-
6
- KINDS = ("choice", "noul", "score")
7
- FROZEN = ("encoder.embeddings.tok_embeddings.weight",
8
- "encoder.embeddings.norm.weight", "type_embedding.weight")
9
- COUNTS = {"tensors": 489, "parameters": 571909635, "active_tensors": 486,
10
- "active_parameters": 375298563, "frozen_tensors": 3,
11
- "frozen_parameters": 196611072}
12
-
13
-
14
- def parameter_group(name):
15
- if name in FROZEN:
16
- return "frozen"
17
- for kind, prefixes in {"choice": ("choice_blocks.", "choice_final_norm."),
18
- "noul": ("encoder.layers.", "encoder.final_norm."),
19
- "score": ("score_blocks.", "score_final_norm.")}.items():
20
- if name.startswith(prefixes):
21
- return kind + ".encoder"
22
- for kind in KINDS:
23
- for suffix, prefix in (("head", "heads."), ("scorer", "scorers.")):
24
- if name.startswith(prefix + kind + "."):
25
- return kind + "." + suffix
26
- raise ValueError("Unsupported parameter name: " + name)
27
-
28
-
29
- def _roots(model):
30
- return {"choice": (model.choice_blocks, model.choice_final_norm, model.heads["choice"], model.scorers["choice"]),
31
- "noul": (model.encoder.layers, model.encoder.final_norm, model.heads["noul"], model.scorers["noul"]),
32
- "score": (model.score_blocks, model.score_final_norm, model.heads["score"], model.scorers["score"])}
33
-
34
-
35
- def _identities(model):
36
- return {n: (id(p), p.untyped_storage().data_ptr()) for n, p in model.named_parameters()}
37
-
38
-
39
- def _api(native):
40
- m, c, api = native.model, native.contract, native.model_api
41
- if (type(m) is not api.DecisionModel or m.arm != "all22" or m.training_arm != "S22"
42
- or m.trainability_mode != c.training_policy("S22")):
43
- raise ValueError("Original all22/S22 native class and metadata required")
44
- api.runtime_gate()
45
- return api
46
-
47
-
48
- def inventory(native):
49
- import torch
50
- _api(native)
51
- m = native.model
52
- shapes = native.contract.all_shapes(m.arm, m.training_arm)
53
- raw = list(m.named_parameters(remove_duplicate=False))
54
- names = {n for n, _ in raw}
55
- if names != set(shapes) or set(m.state_dict()) != names or len(raw) != len(names):
56
- raise ValueError("Exact native parameter/state roster required")
57
- if len({id(p) for _, p in raw}) != len(raw) or len({p.untyped_storage().data_ptr() for _, p in raw}) != len(raw):
58
- raise ValueError("Aliased parameters/storage are unsupported")
59
- if any(tuple(p.shape) != tuple(shapes[n]) or p.dtype != torch.float32
60
- or p.requires_grad != (parameter_group(n) != "frozen") for n, p in raw):
61
- raise ValueError("Training shape, FP32 parameter dtype or policy drift")
62
- active = [p for n, p in raw if parameter_group(n) != "frozen"]
63
- counts = {"tensors": len(raw), "parameters": sum(p.numel() for _, p in raw),
64
- "active_tensors": len(active), "active_parameters": sum(p.numel() for p in active),
65
- "frozen_tensors": len(raw) - len(active),
66
- "frozen_parameters": sum(p.numel() for n, p in raw if parameter_group(n) == "frozen")}
67
- if counts != COUNTS:
68
- raise ValueError("Unsupported native geometry")
69
- groups = {g: [n for n, _ in raw if parameter_group(n) == g]
70
- for g in ("frozen", *(k + "." + part for k in KINDS for part in ("encoder", "head", "scorer")))}
71
- return {"counts": counts, "groups": groups}
72
-
73
-
74
- def configure_training(native, *, max_input_tokens=1024):
75
- """Enable all three existing encoder/head/scorer paths; allocate no parameters."""
76
- api = _api(native)
77
- if native.training_state is not None:
78
- raise ValueError("Restore before configuring again")
79
- if type(max_input_tokens) is not int or not 1 <= max_input_tokens <= native.config["packing"]["max_length"]:
80
- raise ValueError("Training token limit must fit the native complete-input cap")
81
- m = native.model
82
- m.verify_inventory(check_values=False)
83
- native.contract.validate_encoder_config(m.encoder.config)
84
- for blocks, _, _, _ in _roots(m).values():
85
- if len(blocks) != 22:
86
- raise ValueError("Three full 22-layer paths required")
87
- for index, layer in enumerate(blocks):
88
- api.layer_gate(layer, index)
89
- for name, p in m.named_parameters():
90
- p.requires_grad_(parameter_group(name) != "frozen")
91
- p.grad = None
92
- native.training_state = {"identities": _identities(m), "training": False,
93
- "max_input_tokens": max_input_tokens}
94
- set_training_mode(native, False)
95
- return inventory(native)
96
-
97
-
98
- def _assert_modes(native):
99
- m, state = native.model, native.training_state
100
- if state is None:
101
- raise ValueError("Call configure_training first")
102
- selected = {id(child) for roots in _roots(m).values() for root in roots for child in root.modules()}
103
- if m.training != state["training"] or _identities(m) != state["identities"]:
104
- raise ValueError("Mode or parameter/optimizer identity changed")
105
- for module in m.modules():
106
- if module is not m and module.training != (state["training"] and id(module) in selected):
107
- raise ValueError("Active/frozen module mode drift; use set_training_mode")
108
- if any(p.requires_grad != (parameter_group(n) != "frozen") for n, p in m.named_parameters()):
109
- raise ValueError("Training requires_grad drift")
110
-
111
-
112
- def set_training_mode(native, training=True):
113
- if type(training) is not bool or native.training_state is None:
114
- raise ValueError("Configured boolean training mode required")
115
- m = native.model
116
- m.eval()
117
- m.training = training
118
- for roots in _roots(m).values():
119
- for module in roots:
120
- module.train(training)
121
- native.training_state["training"] = training
122
- _assert_modes(native)
123
-
124
-
125
- def optimizer_groups(native, *, encoder_lr, head_lr):
126
- """Return disjoint complete groups; no optimizer or scheduler is created."""
127
- if any(not math.isfinite(x) or x <= 0 for x in (encoder_lr, head_lr)):
128
- raise ValueError("Positive finite learning rates required")
129
- info = inventory(native)
130
- params = dict(native.model.named_parameters())
131
- return [{"name": group, "params": [params[n] for n in names],
132
- "lr": encoder_lr if group.endswith("encoder") else head_lr}
133
- for group, names in info["groups"].items() if group != "frozen"]
134
-
135
-
136
- def _validate_batch(native, batch):
137
- import torch
138
- required = {"input_ids", "attention_mask", "marker_positions", "valid_candidates", "targets", "values", "kind_ids"}
139
- if set(batch) != required or any(not isinstance(t, torch.Tensor) for t in batch.values()):
140
- raise ValueError("Use the native collator's complete batch dictionary")
141
- ids, mask, kinds = batch["input_ids"], batch["attention_mask"], batch["kind_ids"]
142
- pos, valid = batch["marker_positions"], batch["valid_candidates"]
143
- _amd_device(ids.device)
144
- if any(p.device != ids.device for p in native.model.parameters()) or any(t.device != ids.device for t in batch.values()):
145
- raise ValueError("All parameters and batch tensors must share the actual AMD device")
146
- if (ids.dtype != torch.long or ids.ndim != 2 or ids.shape[0] < 1
147
- or not 1 <= ids.shape[1] <= native.training_state["max_input_tokens"]
148
- or mask.dtype != torch.bool or mask.shape != ids.shape
149
- or kinds.dtype != torch.long or kinds.shape != (ids.shape[0],)):
150
- raise ValueError("Invalid complete B x L input")
151
- if (pos.dtype != torch.long or valid.dtype != torch.bool or pos.ndim != 2
152
- or pos.shape != valid.shape or pos.shape[0] != ids.shape[0]
153
- or not 2 <= pos.shape[1] <= 255 or (pos < 0).any().item()
154
- or (pos >= ids.shape[1]).any().item()):
155
- raise ValueError("Invalid candidate markers")
156
- counts = valid.sum(-1)
157
- prefix = torch.arange(valid.shape[1], device=ids.device)[None] < counts[:, None]
158
- if (counts < 2).any().item() or not torch.equal(prefix, valid):
159
- raise ValueError("At least two candidates, then contiguous padding required")
160
- if not mask.gather(1, pos)[valid].all().item() or not mask.any(-1).all().item():
161
- raise ValueError("Candidate marker points into padding")
162
- if not torch.equal(mask, torch.arange(ids.shape[1], device=ids.device)[None] < mask.sum(-1)[:, None]):
163
- raise ValueError("Native right-padding layout required")
164
- if ((pos[:, 1:] <= pos[:, :-1]) & valid[:, 1:]).any().item():
165
- raise ValueError("Candidate markers must retain their original sequence order")
166
- if ((kinds < 0) | (kinds > 2)).any().item() or (counts[kinds == 1] != 2).any().item():
167
- raise ValueError("Unknown type or non-binary Noul")
168
- if batch["targets"].shape != valid.shape or batch["values"].shape != valid.shape:
169
- raise ValueError("Target/value shape mismatch")
170
- values = batch["values"]
171
- if values.dtype != torch.float32 or not torch.isfinite(values).all().item():
172
- raise ValueError("Finite FP32 values required")
173
- scored = (kinds == 2)[:, None] & valid[:, 1:]
174
- if ((values[:, 1:] <= values[:, :-1]) & scored).any().item():
175
- raise ValueError("Score values must remain strictly increasing")
176
-
177
-
178
- def forward_for_training(native, batch):
179
- """Full-B×L suffixes, typed row gathering only at heads; no native monkeypatch."""
180
- import torch
181
- from torch.utils.checkpoint import checkpoint
182
- api = _api(native)
183
- _assert_modes(native)
184
- _validate_batch(native, batch)
185
- if native.training_state["training"] and not torch.is_grad_enabled():
186
- raise ValueError("Training forward requires autograd")
187
- m = native.model
188
- ids, mask = batch["input_ids"], batch["attention_mask"]
189
- kinds, valid, positions = batch["kind_ids"], batch["valid_candidates"], batch["marker_positions"]
190
- routes = api.route_indices(kinds)
191
- m.encoder._maybe_set_compile()
192
- position_ids = torch.arange(ids.shape[1], device=ids.device).unsqueeze(0)
193
- global_mask, local_mask = m.encoder._update_attention_mask(mask, output_attentions=False)
194
- kwargs = dict(attention_mask=global_mask, sliding_window_mask=local_mask,
195
- position_ids=position_ids, cu_seqlens=None, max_seqlen=None, output_attentions=False)
196
- with torch.no_grad():
197
- embedded = m.encoder.embeddings(input_ids=ids, inputs_embeds=None)
198
- hidden = {}
199
- roots = _roots(m)
200
- for name, indices in routes:
201
- if indices.numel() == 0:
202
- continue
203
- blocks, norm, _, _ = roots[name]
204
- h = m._suffix(embedded, blocks, norm, kwargs)
205
- with torch.no_grad():
206
- type_value = m.type_embedding(kinds)[:, None, :]
207
- hidden[name] = h + type_value.to(h.dtype)
208
- pad = ~mask.bool()
209
- out = torch.empty(positions.shape, device=ids.device, dtype=torch.float32)
210
- for name, indices in routes:
211
- if indices.numel() == 0:
212
- continue
213
- h = hidden[name].index_select(0, indices)
214
- branch_pad = pad.index_select(0, indices)
215
- for layer in m.heads[name]:
216
- h = (checkpoint(layer, h, src_key_padding_mask=branch_pad, use_reentrant=False)
217
- if layer.training and m.head_config["gradient_checkpointing"]
218
- else layer(h, src_key_padding_mask=branch_pad))
219
- loc = positions.index_select(0, indices)
220
- markers = torch.gather(h, 1, loc[:, :, None].expand(-1, -1, h.shape[-1]))
221
- logits = m.scorers[name](markers).squeeze(-1).float()
222
- out = api.scatter_rows(out, indices, logits)
223
- return out.masked_fill(~valid, torch.finfo(torch.float32).min)
224
-
225
-
226
- def loss_for_training(native, logits, batch, *, score_rps_weight=0.0):
227
- """Return (per-row total, CE, RPS); caller chooses the logical-batch denominator."""
228
- import torch
229
- if not math.isfinite(score_rps_weight) or score_rps_weight < 0:
230
- raise ValueError("Nonnegative finite Score RPS weight required")
231
- target, valid = batch["targets"], batch["valid_candidates"]
232
- if (logits.shape != target.shape or target.shape != valid.shape
233
- or target.dtype != torch.float32 or not torch.isfinite(logits).all().item()
234
- or not torch.isfinite(target).all().item() or ((target < 0) | (target > 1)).any().item()
235
- or target[~valid].count_nonzero().item() or not torch.allclose(target.sum(-1), torch.ones_like(target[:, 0]), atol=1e-6, rtol=0)):
236
- raise ValueError("Explicit normalized hard/soft targets and finite masked logits required")
237
- return native.model_api.typed_loss(logits, batch, score_rps_weight=score_rps_weight)
238
-
239
-
240
- def _tensor_sha(tensor):
241
- import hashlib
242
- return hashlib.sha256(tensor.detach().cpu().contiguous().numpy().tobytes()).hexdigest()
243
-
244
-
245
- def frozen_snapshot(native):
246
- inventory(native)
247
- _assert_modes(native)
248
- ps = dict(native.model.named_parameters())
249
- return {n: {"version": int(ps[n]._version), "sha256": _tensor_sha(ps[n])} for n in FROZEN}
250
-
251
-
252
- def assert_frozen(native, snapshot, *, content=False, versions=True):
253
- _assert_modes(native)
254
- ps = dict(native.model.named_parameters())
255
- if set(snapshot) != set(FROZEN):
256
- raise ValueError("Expected exactly the three shared frozen parameters")
257
- for name, saved in snapshot.items():
258
- p = ps[name]
259
- if (p.requires_grad or p.grad is not None or (versions and int(p._version) != saved["version"])
260
- or (content and _tensor_sha(p) != saved["sha256"])):
261
- raise ValueError("Frozen parameter changed: " + name)
262
- return True
263
-
264
-
265
- def gradient_report(native, *, present_types):
266
- """Check groups actually present in the logical batch; absent grads must be None."""
267
- import torch
268
- if not present_types or not set(present_types) <= set(KINDS):
269
- raise ValueError("Explicit nonempty logical-batch type set required")
270
- info = inventory(native)
271
- ps = dict(native.model.named_parameters())
272
- report = {}
273
- for group, names in info["groups"].items():
274
- active = group != "frozen" and group.split(".")[0] in present_types
275
- gradients = [ps[n].grad for n in names]
276
- if not active:
277
- if any(g is not None for g in gradients):
278
- raise ValueError("Frozen/absent group has stale gradients: " + group)
279
- continue
280
- if any(g is None or not torch.isfinite(g).all().item() for g in gradients):
281
- raise ValueError("Missing or nonfinite gradient: " + group)
282
- norms = [float(g.float().norm().item()) for g in gradients]
283
- if not any(n > 0 for n in norms):
284
- raise ValueError("No nonzero gradient: " + group)
285
- report[group] = {"tensors": len(names), "nonzero_tensors": sum(n > 0 for n in norms),
286
- "l2": math.sqrt(sum(n * n for n in norms))}
287
- return report
288
-
289
-
290
- def restore_native_policy_for_export(native):
291
- """Clear gradients and restore the original strict metadata, without replacing parameters."""
292
- _assert_modes(native)
293
- inventory(native)
294
- m, state, contract = native.model, native.training_state, native.contract
295
- policy = contract.training_policy(m.training_arm)
296
- sources = {p.name: sha256(p) for p in sorted(Path(__file__).parent.glob("*.py")) if p.name != "checks.py"}
297
- provenance = {"actual_training_policy": "all_three_full22_paths_and_heads",
298
- "counts": dict(COUNTS), "complete_input_limit": state["max_input_tokens"],
299
- "native_policy_is_not_actual_training_provenance": True,
300
- "restored_native_trainability_mode": policy, "adapter_sources_sha256": sources}
301
- for n, p in m.named_parameters():
302
- p.requires_grad_(contract.trainable_name(n, policy))
303
- p.grad = None
304
- m.eval()
305
- m.verify_inventory(check_values=False)
306
- if _identities(m) != state["identities"]:
307
- raise ValueError("Restoration replaced parameter objects")
308
- native.training_state = None
309
- native.last_training_provenance = provenance
310
- return provenance
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
EVALUATION.md → evaluation/EVALUATION.md RENAMED
File without changes
METHODS.md → evaluation/METHODS.md RENAMED
@@ -6,6 +6,6 @@ Development used a state-component split of the original official TRAIN: 4,800 d
6
 
7
  For the original normalized soft label vector q and original semantic hard-label one-hot h, the target is 0.5q + 0.5h. The loss is cross-entropy summed over 64 decisions and divided by 64, with no RPS, consistency, distillation or calibration fitting. Candidate order and semantic descriptions are retained. Each logical batch uses eight physical batches of eight. Training uses BF16 autocast with FP32 parameters and loss, AdamW (weight decay 0.01), gradient clipping at 1, and seed 20260921. There is no warmup. Encoder/head starting learning rates are 2.5e-5/1e-4 and each follows `1e-6 + (base - 1e-6) * (1 + cos(pi * (step - 1) / 749)) / 2` for steps 1–750.
8
 
9
- All complete packed inputs fit 1,024 tokens. The final native manifest is `f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6`. Native metadata retains the original research export provenance; its native Score training-policy metadata is an inference/export compatibility contract, not a claim that only Score was trained. The actual all-types update is recorded in the export provenance and [TRAINING_PROVENANCE.json](TRAINING_PROVENANCE.json).
10
 
11
  The final TEST was opened only after this checkpoint and recipe were frozen. It was evaluated once against the official typed specialist on identical complete inputs. Neither TEST checkpoint selection nor temperature fitting was performed. [EVALUATION.md](EVALUATION.md) records the point improvement and its limits. This package does not transfer Kai's multilingual quality or latency measurements to Lex.
 
6
 
7
  For the original normalized soft label vector q and original semantic hard-label one-hot h, the target is 0.5q + 0.5h. The loss is cross-entropy summed over 64 decisions and divided by 64, with no RPS, consistency, distillation or calibration fitting. Candidate order and semantic descriptions are retained. Each logical batch uses eight physical batches of eight. Training uses BF16 autocast with FP32 parameters and loss, AdamW (weight decay 0.01), gradient clipping at 1, and seed 20260921. There is no warmup. Encoder/head starting learning rates are 2.5e-5/1e-4 and each follows `1e-6 + (base - 1e-6) * (1 + cos(pi * (step - 1) / 749)) / 2` for steps 1–750.
8
 
9
+ All complete packed inputs fit 1,024 tokens. The original native manifest was `f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6`. That historical export included training-policy metadata; the model-only release retains the model weights and configuration. The actual all-types update is recorded in the export provenance and [TRAINING_PROVENANCE.json](../TRAINING_PROVENANCE.json).
10
 
11
  The final TEST was opened only after this checkpoint and recipe were frozen. It was evaluated once against the official typed specialist on identical complete inputs. Neither TEST checkpoint selection nor temperature fitting was performed. [EVALUATION.md](EVALUATION.md) records the point improvement and its limits. This package does not transfer Kai's multilingual quality or latency measurements to Lex.
MIXED_QUESTION_SCALING.json → evaluation/MIXED_QUESTION_SCALING.json RENAMED
File without changes
evaluation/PERFORMANCE.md ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Lex: measured Decision runtime latency
2
+
3
+ On the evaluated AMD ROCm runtime, 128 mixed Choice, Noul and Score questions took **154.41 ms median**, 56.6% less than the previous default runtime on the same fixed workload.
4
+
5
+ ![Lex latency as mixed question count increases](../assets/mixed-question-scaling.png)
6
+
7
+ | Questions | Previous p50 / p95 (ms) | Typed scheduling p50 / p95 (ms) |
8
+ |---:|---:|---:|
9
+ | 1 | 13.22 / 13.32 | 13.20 / 13.40 |
10
+ | 8 | 28.07 / 28.32 | 28.02 / 28.31 |
11
+ | 32 | 93.76 / 94.06 | 53.43 / 53.77 |
12
+ | 64 | 181.78 / 184.53 | 87.39 / 88.09 |
13
+ | 128 | 355.97 / 360.25 | 154.41 / 155.93 |
14
+
15
+ The request repeats three fixed questions over one context; only question count and bookkeeping IDs change. Each question still receives its own contextual computation. Measurements used the exact released weights, FP32 inference, physical batch size eight, 10 warmup pairs and 30 alternating AB/BA pairs per point, with GPU synchronization. They include local request conversion, tokenization, model execution and answer assembly; transport and Studio are excluded.
16
+
17
+ These results describe the evaluated runtime and workload, not a latency guarantee for a separately distributed runtime. The model-only repository contains no serving code. [Samples and validation details](MIXED_QUESTION_SCALING.json) · [SVG](../assets/mixed-question-scaling.svg) · [PDF](../assets/mixed-question-scaling.pdf)
TECHNICAL_VALIDATION.json → evaluation/TECHNICAL_VALIDATION.json RENAMED
File without changes
VALIDATION.md → evaluation/VALIDATION.md RENAMED
@@ -4,6 +4,6 @@ The final refit completed 750 updates and exported a native bundle with 571,909,
4
 
5
  After releasing the training model and optimizer, a separate process loaded the final native and repeated those 64 TRAIN decisions in eight FP32 batches. Logits and probabilities matched the saved final predictions exactly, with zero hard-label flips. This was a technical reload check, not another quality evaluation.
6
 
7
- The public Python inference and fine-tuning modules are copied unchanged from the tested Kai distribution. Their strict loader verifies every native file and the runtime source allowlist before loading. The native runtime/tokenizer/geometry match the supported architecture; no research-directory import is required. The completed Lex package-isolation check loaded the bundled native through public `load_native` and evaluated the same 64 TRAIN decisions through `predict_1k` in eight FP32 batches. It ran with isolated Python from an external working directory, with all application imports confined to this package. Maximum logit/probability differences from the independent final-refit fresh reference were 0.0/0.0, with zero hard flips. All 489 parameter contents and versions were unchanged; no gradients, updates or TEST evaluation occurred. See [TECHNICAL_VALIDATION.json](TECHNICAL_VALIDATION.json).
8
 
9
- The included fine-tuning CLI's earlier bounded continuous-versus-resume and independent-reload checks are interface evidence, not new Lex training results. Lex's final training used the fixed research recipe described in [METHODS.md](METHODS.md). The supported product profile is complete packed length ≤1,024 and FP32 inference on compatible AMD ROCm. No browser end-to-end latency, CPU/NVIDIA inference, multilingual-specialist quality or larger-window quality is claimed.
 
4
 
5
  After releasing the training model and optimizer, a separate process loaded the final native and repeated those 64 TRAIN decisions in eight FP32 batches. Logits and probabilities matched the saved final predictions exactly, with zero hard-label flips. This was a technical reload check, not another quality evaluation.
6
 
7
+ The original release included Python inference and fine-tuning modules copied from the tested Kai distribution. Its package-isolation check loaded the bundled native and evaluated the same 64 TRAIN decisions in eight FP32 batches from an external working directory. Maximum logit/probability differences from the independent final-refit fresh reference were 0.0/0.0, with zero hard flips. All 489 parameter contents and versions were unchanged; no gradients, updates or TEST evaluation occurred. This is historical validation of those model objects and runtime; executable code is no longer part of the model-only repository. See [TECHNICAL_VALIDATION.json](TECHNICAL_VALIDATION.json).
8
 
9
+ The earlier fine-tuning CLI's bounded continuous-versus-resume and independent-reload checks are interface evidence, not new Lex training results. Lex's final training used the fixed research recipe described in [METHODS.md](METHODS.md). The evaluated runtime profile used complete packed length ≤1,024 and FP32 inference on compatible AMD ROCm. No browser end-to-end latency, CPU/NVIDIA inference, multilingual-specialist quality or larger-window quality is claimed.
examples/decisions.jsonl DELETED
@@ -1,3 +0,0 @@
1
- {"id":"route","state_text":"The customer asks for a refund for a duplicate charge.","question":{"id":"route-q","type":"Choice","text":"Which team should handle this request?","options":[{"id":"billing","text":"Billing and payments"},{"id":"technical","text":"Technical support"},{"id":"sales","text":"Sales enquiries"}]}}
2
- {"id":"evidence","state_text":"The package was delivered on Monday. A signed receipt is available.","question":{"id":"evidence-q","type":"Noul","text":"Does the evidence confirm that the package was delivered?"}}
3
- {"id":"sentiment","state_text":"The replacement arrived quickly and works perfectly.","question":{"id":"sentiment-q","type":"Score","text":"How positive is the customer's sentiment?","levels":[{"id":"negative","text":"Negative","value":0},{"id":"neutral","text":"Neutral","value":1},{"id":"positive","text":"Positive","value":2}]}}
 
 
 
 
examples/finetune.sh DELETED
@@ -1,21 +0,0 @@
1
- #!/usr/bin/env bash
2
- set -euo pipefail
3
- # Run from the package root in an existing compatible AMD ROCm environment.
4
- # Replace these paths with your own isolated TRAIN and DEV JSONL files.
5
- : "${TRAIN:?Set TRAIN to your training JSONL}"
6
- : "${DEV:?Set DEV to your development JSONL}"
7
- : "${OUTPUT:?Set OUTPUT to a fresh run directory}"
8
- NATIVE="${NATIVE:-./native}"
9
- MANIFEST_SHA256="${MANIFEST_SHA256:-f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6}"
10
- export PYTHONPATH="${PWD}${PYTHONPATH:+:${PYTHONPATH}}"
11
- export PYTHONDONTWRITEBYTECODE=1 HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 TOKENIZERS_PARALLELISM=false
12
- python -m decision_finetune validate-jsonl --train "$TRAIN" --dev "$DEV" --selection macro-nll
13
- # Device visibility is supplied by the caller, e.g. ROCR_VISIBLE_DEVICES=0.
14
- python -m decision_finetune train \
15
- --native "$NATIVE" --manifest-sha256 "$MANIFEST_SHA256" \
16
- --train "$TRAIN" --dev "$DEV" --output "$OUTPUT" \
17
- --epochs 4 --logical-batch-size 64 --micro-batch-size 8 --seed 20260921 \
18
- --encoder-lr 2.5e-5 --head-lr 1e-4 --lr-min 1e-6 \
19
- --weight-decay 0.01 --clip-norm 1 --score-rps-weight 0.1 \
20
- --warmup-ratio 0.1 --selection macro-nll --selection-tolerance 1e-8 \
21
- --cpu-threads 2 --max-reserved-gib 64
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
examples/system-one.json DELETED
@@ -1,29 +0,0 @@
1
- {
2
- "model": "Decision-1.0-Lex",
3
- "state": {
4
- "message": "Please refund the duplicate charge. I need this fixed today."
5
- },
6
- "questions": {
7
- "refund_requested": {
8
- "type": "noul",
9
- "instructions": "Does the customer explicitly request a refund?"
10
- },
11
- "team": {
12
- "type": "choice",
13
- "instructions": "Which team should handle this request?",
14
- "criteria": {
15
- "Billing": "Charges and refunds",
16
- "Support": "Technical problems"
17
- }
18
- },
19
- "urgency": {
20
- "type": "score",
21
- "instructions": "How urgent is the request?",
22
- "criteria": [
23
- "No deadline",
24
- "Needed soon",
25
- "Needed today"
26
- ]
27
- }
28
- }
29
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
infer.py DELETED
@@ -1,46 +0,0 @@
1
- """Complete-input 1K inference using the bundled, unchanged native runtime."""
2
- import argparse
3
- import json
4
- from pathlib import Path
5
-
6
- NATIVE_MANIFEST_SHA256 = 'f288d873999832a3f37c6a7c4268c2ab309691e621794dbf7acab891acbbb7e6'
7
-
8
-
9
- def main():
10
- p = argparse.ArgumentParser(description='Decision Lex: complete1024, FP32 AMD inference')
11
- p.add_argument('--native', default=str(Path(__file__).resolve().parent / 'native'))
12
- p.add_argument('--manifest-sha256', default=NATIVE_MANIFEST_SHA256)
13
- p.add_argument('--input', required=True, help='JSONL with id, state_text, question')
14
- p.add_argument('--output', required=True, help='Fresh prediction JSONL; existing files are not overwritten')
15
- p.add_argument('--batch-size', type=int, choices=range(1, 9), default=8)
16
- args = p.parse_args()
17
- records = []
18
- with Path(args.input).open(encoding='utf-8') as stream:
19
- for line in stream:
20
- row = json.loads(line)
21
- if set(row) != {'id', 'state_text', 'question'}:
22
- raise ValueError('Inference rows must contain exactly id, state_text and question')
23
- records.append(row)
24
- if not records or len({r['id'] for r in records}) != len(records):
25
- raise ValueError('Nonempty input with unique record IDs required')
26
- if Path(args.output).exists():
27
- raise ValueError('Prediction output must be fresh')
28
- import torch
29
- from decision_runtime import load_native
30
- from decision_inference import predict_1k
31
- if torch.version.hip is None or not torch.cuda.is_available() or torch.cuda.device_count() != 1:
32
- raise RuntimeError('Expose exactly one AMD ROCm GPU; this package has no CPU/NVIDIA fallback')
33
- torch.cuda.set_device(0); torch.set_num_threads(2)
34
- torch.backends.cuda.matmul.allow_tf32 = False
35
- torch.backends.cudnn.allow_tf32 = False
36
- torch.backends.mha.set_fastpath_enabled(False)
37
- native = load_native(args.native, expected_manifest_sha256=args.manifest_sha256, device='cuda:0')
38
- answers = predict_1k(native, records, batch_size=args.batch_size)
39
- if len(answers) != len(records):
40
- raise RuntimeError('Incomplete inference result')
41
- with Path(args.output).open('x', encoding='utf-8') as stream:
42
- for answer in answers:
43
- stream.write(json.dumps(answer, ensure_ascii=False, allow_nan=False) + '\n')
44
-
45
-
46
- if __name__ == '__main__': main()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
native/INVENTORY.json DELETED
@@ -1,1786 +0,0 @@
1
- {
2
- "choice_suffix_shapes": {
3
- "choice_blocks.0.attn.Wo.weight": [
4
- 768,
5
- 768
6
- ],
7
- "choice_blocks.0.attn.Wqkv.weight": [
8
- 2304,
9
- 768
10
- ],
11
- "choice_blocks.0.mlp.Wi.weight": [
12
- 2304,
13
- 768
14
- ],
15
- "choice_blocks.0.mlp.Wo.weight": [
16
- 768,
17
- 1152
18
- ],
19
- "choice_blocks.0.mlp_norm.weight": [
20
- 768
21
- ],
22
- "choice_blocks.1.attn.Wo.weight": [
23
- 768,
24
- 768
25
- ],
26
- "choice_blocks.1.attn.Wqkv.weight": [
27
- 2304,
28
- 768
29
- ],
30
- "choice_blocks.1.attn_norm.weight": [
31
- 768
32
- ],
33
- "choice_blocks.1.mlp.Wi.weight": [
34
- 2304,
35
- 768
36
- ],
37
- "choice_blocks.1.mlp.Wo.weight": [
38
- 768,
39
- 1152
40
- ],
41
- "choice_blocks.1.mlp_norm.weight": [
42
- 768
43
- ],
44
- "choice_blocks.10.attn.Wo.weight": [
45
- 768,
46
- 768
47
- ],
48
- "choice_blocks.10.attn.Wqkv.weight": [
49
- 2304,
50
- 768
51
- ],
52
- "choice_blocks.10.attn_norm.weight": [
53
- 768
54
- ],
55
- "choice_blocks.10.mlp.Wi.weight": [
56
- 2304,
57
- 768
58
- ],
59
- "choice_blocks.10.mlp.Wo.weight": [
60
- 768,
61
- 1152
62
- ],
63
- "choice_blocks.10.mlp_norm.weight": [
64
- 768
65
- ],
66
- "choice_blocks.11.attn.Wo.weight": [
67
- 768,
68
- 768
69
- ],
70
- "choice_blocks.11.attn.Wqkv.weight": [
71
- 2304,
72
- 768
73
- ],
74
- "choice_blocks.11.attn_norm.weight": [
75
- 768
76
- ],
77
- "choice_blocks.11.mlp.Wi.weight": [
78
- 2304,
79
- 768
80
- ],
81
- "choice_blocks.11.mlp.Wo.weight": [
82
- 768,
83
- 1152
84
- ],
85
- "choice_blocks.11.mlp_norm.weight": [
86
- 768
87
- ],
88
- "choice_blocks.12.attn.Wo.weight": [
89
- 768,
90
- 768
91
- ],
92
- "choice_blocks.12.attn.Wqkv.weight": [
93
- 2304,
94
- 768
95
- ],
96
- "choice_blocks.12.attn_norm.weight": [
97
- 768
98
- ],
99
- "choice_blocks.12.mlp.Wi.weight": [
100
- 2304,
101
- 768
102
- ],
103
- "choice_blocks.12.mlp.Wo.weight": [
104
- 768,
105
- 1152
106
- ],
107
- "choice_blocks.12.mlp_norm.weight": [
108
- 768
109
- ],
110
- "choice_blocks.13.attn.Wo.weight": [
111
- 768,
112
- 768
113
- ],
114
- "choice_blocks.13.attn.Wqkv.weight": [
115
- 2304,
116
- 768
117
- ],
118
- "choice_blocks.13.attn_norm.weight": [
119
- 768
120
- ],
121
- "choice_blocks.13.mlp.Wi.weight": [
122
- 2304,
123
- 768
124
- ],
125
- "choice_blocks.13.mlp.Wo.weight": [
126
- 768,
127
- 1152
128
- ],
129
- "choice_blocks.13.mlp_norm.weight": [
130
- 768
131
- ],
132
- "choice_blocks.14.attn.Wo.weight": [
133
- 768,
134
- 768
135
- ],
136
- "choice_blocks.14.attn.Wqkv.weight": [
137
- 2304,
138
- 768
139
- ],
140
- "choice_blocks.14.attn_norm.weight": [
141
- 768
142
- ],
143
- "choice_blocks.14.mlp.Wi.weight": [
144
- 2304,
145
- 768
146
- ],
147
- "choice_blocks.14.mlp.Wo.weight": [
148
- 768,
149
- 1152
150
- ],
151
- "choice_blocks.14.mlp_norm.weight": [
152
- 768
153
- ],
154
- "choice_blocks.15.attn.Wo.weight": [
155
- 768,
156
- 768
157
- ],
158
- "choice_blocks.15.attn.Wqkv.weight": [
159
- 2304,
160
- 768
161
- ],
162
- "choice_blocks.15.attn_norm.weight": [
163
- 768
164
- ],
165
- "choice_blocks.15.mlp.Wi.weight": [
166
- 2304,
167
- 768
168
- ],
169
- "choice_blocks.15.mlp.Wo.weight": [
170
- 768,
171
- 1152
172
- ],
173
- "choice_blocks.15.mlp_norm.weight": [
174
- 768
175
- ],
176
- "choice_blocks.16.attn.Wo.weight": [
177
- 768,
178
- 768
179
- ],
180
- "choice_blocks.16.attn.Wqkv.weight": [
181
- 2304,
182
- 768
183
- ],
184
- "choice_blocks.16.attn_norm.weight": [
185
- 768
186
- ],
187
- "choice_blocks.16.mlp.Wi.weight": [
188
- 2304,
189
- 768
190
- ],
191
- "choice_blocks.16.mlp.Wo.weight": [
192
- 768,
193
- 1152
194
- ],
195
- "choice_blocks.16.mlp_norm.weight": [
196
- 768
197
- ],
198
- "choice_blocks.17.attn.Wo.weight": [
199
- 768,
200
- 768
201
- ],
202
- "choice_blocks.17.attn.Wqkv.weight": [
203
- 2304,
204
- 768
205
- ],
206
- "choice_blocks.17.attn_norm.weight": [
207
- 768
208
- ],
209
- "choice_blocks.17.mlp.Wi.weight": [
210
- 2304,
211
- 768
212
- ],
213
- "choice_blocks.17.mlp.Wo.weight": [
214
- 768,
215
- 1152
216
- ],
217
- "choice_blocks.17.mlp_norm.weight": [
218
- 768
219
- ],
220
- "choice_blocks.18.attn.Wo.weight": [
221
- 768,
222
- 768
223
- ],
224
- "choice_blocks.18.attn.Wqkv.weight": [
225
- 2304,
226
- 768
227
- ],
228
- "choice_blocks.18.attn_norm.weight": [
229
- 768
230
- ],
231
- "choice_blocks.18.mlp.Wi.weight": [
232
- 2304,
233
- 768
234
- ],
235
- "choice_blocks.18.mlp.Wo.weight": [
236
- 768,
237
- 1152
238
- ],
239
- "choice_blocks.18.mlp_norm.weight": [
240
- 768
241
- ],
242
- "choice_blocks.19.attn.Wo.weight": [
243
- 768,
244
- 768
245
- ],
246
- "choice_blocks.19.attn.Wqkv.weight": [
247
- 2304,
248
- 768
249
- ],
250
- "choice_blocks.19.attn_norm.weight": [
251
- 768
252
- ],
253
- "choice_blocks.19.mlp.Wi.weight": [
254
- 2304,
255
- 768
256
- ],
257
- "choice_blocks.19.mlp.Wo.weight": [
258
- 768,
259
- 1152
260
- ],
261
- "choice_blocks.19.mlp_norm.weight": [
262
- 768
263
- ],
264
- "choice_blocks.2.attn.Wo.weight": [
265
- 768,
266
- 768
267
- ],
268
- "choice_blocks.2.attn.Wqkv.weight": [
269
- 2304,
270
- 768
271
- ],
272
- "choice_blocks.2.attn_norm.weight": [
273
- 768
274
- ],
275
- "choice_blocks.2.mlp.Wi.weight": [
276
- 2304,
277
- 768
278
- ],
279
- "choice_blocks.2.mlp.Wo.weight": [
280
- 768,
281
- 1152
282
- ],
283
- "choice_blocks.2.mlp_norm.weight": [
284
- 768
285
- ],
286
- "choice_blocks.20.attn.Wo.weight": [
287
- 768,
288
- 768
289
- ],
290
- "choice_blocks.20.attn.Wqkv.weight": [
291
- 2304,
292
- 768
293
- ],
294
- "choice_blocks.20.attn_norm.weight": [
295
- 768
296
- ],
297
- "choice_blocks.20.mlp.Wi.weight": [
298
- 2304,
299
- 768
300
- ],
301
- "choice_blocks.20.mlp.Wo.weight": [
302
- 768,
303
- 1152
304
- ],
305
- "choice_blocks.20.mlp_norm.weight": [
306
- 768
307
- ],
308
- "choice_blocks.21.attn.Wo.weight": [
309
- 768,
310
- 768
311
- ],
312
- "choice_blocks.21.attn.Wqkv.weight": [
313
- 2304,
314
- 768
315
- ],
316
- "choice_blocks.21.attn_norm.weight": [
317
- 768
318
- ],
319
- "choice_blocks.21.mlp.Wi.weight": [
320
- 2304,
321
- 768
322
- ],
323
- "choice_blocks.21.mlp.Wo.weight": [
324
- 768,
325
- 1152
326
- ],
327
- "choice_blocks.21.mlp_norm.weight": [
328
- 768
329
- ],
330
- "choice_blocks.3.attn.Wo.weight": [
331
- 768,
332
- 768
333
- ],
334
- "choice_blocks.3.attn.Wqkv.weight": [
335
- 2304,
336
- 768
337
- ],
338
- "choice_blocks.3.attn_norm.weight": [
339
- 768
340
- ],
341
- "choice_blocks.3.mlp.Wi.weight": [
342
- 2304,
343
- 768
344
- ],
345
- "choice_blocks.3.mlp.Wo.weight": [
346
- 768,
347
- 1152
348
- ],
349
- "choice_blocks.3.mlp_norm.weight": [
350
- 768
351
- ],
352
- "choice_blocks.4.attn.Wo.weight": [
353
- 768,
354
- 768
355
- ],
356
- "choice_blocks.4.attn.Wqkv.weight": [
357
- 2304,
358
- 768
359
- ],
360
- "choice_blocks.4.attn_norm.weight": [
361
- 768
362
- ],
363
- "choice_blocks.4.mlp.Wi.weight": [
364
- 2304,
365
- 768
366
- ],
367
- "choice_blocks.4.mlp.Wo.weight": [
368
- 768,
369
- 1152
370
- ],
371
- "choice_blocks.4.mlp_norm.weight": [
372
- 768
373
- ],
374
- "choice_blocks.5.attn.Wo.weight": [
375
- 768,
376
- 768
377
- ],
378
- "choice_blocks.5.attn.Wqkv.weight": [
379
- 2304,
380
- 768
381
- ],
382
- "choice_blocks.5.attn_norm.weight": [
383
- 768
384
- ],
385
- "choice_blocks.5.mlp.Wi.weight": [
386
- 2304,
387
- 768
388
- ],
389
- "choice_blocks.5.mlp.Wo.weight": [
390
- 768,
391
- 1152
392
- ],
393
- "choice_blocks.5.mlp_norm.weight": [
394
- 768
395
- ],
396
- "choice_blocks.6.attn.Wo.weight": [
397
- 768,
398
- 768
399
- ],
400
- "choice_blocks.6.attn.Wqkv.weight": [
401
- 2304,
402
- 768
403
- ],
404
- "choice_blocks.6.attn_norm.weight": [
405
- 768
406
- ],
407
- "choice_blocks.6.mlp.Wi.weight": [
408
- 2304,
409
- 768
410
- ],
411
- "choice_blocks.6.mlp.Wo.weight": [
412
- 768,
413
- 1152
414
- ],
415
- "choice_blocks.6.mlp_norm.weight": [
416
- 768
417
- ],
418
- "choice_blocks.7.attn.Wo.weight": [
419
- 768,
420
- 768
421
- ],
422
- "choice_blocks.7.attn.Wqkv.weight": [
423
- 2304,
424
- 768
425
- ],
426
- "choice_blocks.7.attn_norm.weight": [
427
- 768
428
- ],
429
- "choice_blocks.7.mlp.Wi.weight": [
430
- 2304,
431
- 768
432
- ],
433
- "choice_blocks.7.mlp.Wo.weight": [
434
- 768,
435
- 1152
436
- ],
437
- "choice_blocks.7.mlp_norm.weight": [
438
- 768
439
- ],
440
- "choice_blocks.8.attn.Wo.weight": [
441
- 768,
442
- 768
443
- ],
444
- "choice_blocks.8.attn.Wqkv.weight": [
445
- 2304,
446
- 768
447
- ],
448
- "choice_blocks.8.attn_norm.weight": [
449
- 768
450
- ],
451
- "choice_blocks.8.mlp.Wi.weight": [
452
- 2304,
453
- 768
454
- ],
455
- "choice_blocks.8.mlp.Wo.weight": [
456
- 768,
457
- 1152
458
- ],
459
- "choice_blocks.8.mlp_norm.weight": [
460
- 768
461
- ],
462
- "choice_blocks.9.attn.Wo.weight": [
463
- 768,
464
- 768
465
- ],
466
- "choice_blocks.9.attn.Wqkv.weight": [
467
- 2304,
468
- 768
469
- ],
470
- "choice_blocks.9.attn_norm.weight": [
471
- 768
472
- ],
473
- "choice_blocks.9.mlp.Wi.weight": [
474
- 2304,
475
- 768
476
- ],
477
- "choice_blocks.9.mlp.Wo.weight": [
478
- 768,
479
- 1152
480
- ],
481
- "choice_blocks.9.mlp_norm.weight": [
482
- 768
483
- ],
484
- "choice_final_norm.weight": [
485
- 768
486
- ]
487
- },
488
- "counts": {
489
- "frozen_parameters": 446810114,
490
- "frozen_tensors": 327,
491
- "integrated_choice_suffix_tensors": 132,
492
- "score_suffix_parameters": 110330880,
493
- "score_suffix_tensors": 132,
494
- "shared_encoder_tensors": 134,
495
- "trainable_parameters": 125099521,
496
- "trainable_tensors": 162,
497
- "type_and_head_tensors": 91,
498
- "unique_parameters": 571909635,
499
- "unique_tensors": 489
500
- },
501
- "encoder_shapes": {
502
- "embeddings.norm.weight": [
503
- 768
504
- ],
505
- "embeddings.tok_embeddings.weight": [
506
- 256000,
507
- 768
508
- ],
509
- "final_norm.weight": [
510
- 768
511
- ],
512
- "layers.0.attn.Wo.weight": [
513
- 768,
514
- 768
515
- ],
516
- "layers.0.attn.Wqkv.weight": [
517
- 2304,
518
- 768
519
- ],
520
- "layers.0.mlp.Wi.weight": [
521
- 2304,
522
- 768
523
- ],
524
- "layers.0.mlp.Wo.weight": [
525
- 768,
526
- 1152
527
- ],
528
- "layers.0.mlp_norm.weight": [
529
- 768
530
- ],
531
- "layers.1.attn.Wo.weight": [
532
- 768,
533
- 768
534
- ],
535
- "layers.1.attn.Wqkv.weight": [
536
- 2304,
537
- 768
538
- ],
539
- "layers.1.attn_norm.weight": [
540
- 768
541
- ],
542
- "layers.1.mlp.Wi.weight": [
543
- 2304,
544
- 768
545
- ],
546
- "layers.1.mlp.Wo.weight": [
547
- 768,
548
- 1152
549
- ],
550
- "layers.1.mlp_norm.weight": [
551
- 768
552
- ],
553
- "layers.10.attn.Wo.weight": [
554
- 768,
555
- 768
556
- ],
557
- "layers.10.attn.Wqkv.weight": [
558
- 2304,
559
- 768
560
- ],
561
- "layers.10.attn_norm.weight": [
562
- 768
563
- ],
564
- "layers.10.mlp.Wi.weight": [
565
- 2304,
566
- 768
567
- ],
568
- "layers.10.mlp.Wo.weight": [
569
- 768,
570
- 1152
571
- ],
572
- "layers.10.mlp_norm.weight": [
573
- 768
574
- ],
575
- "layers.11.attn.Wo.weight": [
576
- 768,
577
- 768
578
- ],
579
- "layers.11.attn.Wqkv.weight": [
580
- 2304,
581
- 768
582
- ],
583
- "layers.11.attn_norm.weight": [
584
- 768
585
- ],
586
- "layers.11.mlp.Wi.weight": [
587
- 2304,
588
- 768
589
- ],
590
- "layers.11.mlp.Wo.weight": [
591
- 768,
592
- 1152
593
- ],
594
- "layers.11.mlp_norm.weight": [
595
- 768
596
- ],
597
- "layers.12.attn.Wo.weight": [
598
- 768,
599
- 768
600
- ],
601
- "layers.12.attn.Wqkv.weight": [
602
- 2304,
603
- 768
604
- ],
605
- "layers.12.attn_norm.weight": [
606
- 768
607
- ],
608
- "layers.12.mlp.Wi.weight": [
609
- 2304,
610
- 768
611
- ],
612
- "layers.12.mlp.Wo.weight": [
613
- 768,
614
- 1152
615
- ],
616
- "layers.12.mlp_norm.weight": [
617
- 768
618
- ],
619
- "layers.13.attn.Wo.weight": [
620
- 768,
621
- 768
622
- ],
623
- "layers.13.attn.Wqkv.weight": [
624
- 2304,
625
- 768
626
- ],
627
- "layers.13.attn_norm.weight": [
628
- 768
629
- ],
630
- "layers.13.mlp.Wi.weight": [
631
- 2304,
632
- 768
633
- ],
634
- "layers.13.mlp.Wo.weight": [
635
- 768,
636
- 1152
637
- ],
638
- "layers.13.mlp_norm.weight": [
639
- 768
640
- ],
641
- "layers.14.attn.Wo.weight": [
642
- 768,
643
- 768
644
- ],
645
- "layers.14.attn.Wqkv.weight": [
646
- 2304,
647
- 768
648
- ],
649
- "layers.14.attn_norm.weight": [
650
- 768
651
- ],
652
- "layers.14.mlp.Wi.weight": [
653
- 2304,
654
- 768
655
- ],
656
- "layers.14.mlp.Wo.weight": [
657
- 768,
658
- 1152
659
- ],
660
- "layers.14.mlp_norm.weight": [
661
- 768
662
- ],
663
- "layers.15.attn.Wo.weight": [
664
- 768,
665
- 768
666
- ],
667
- "layers.15.attn.Wqkv.weight": [
668
- 2304,
669
- 768
670
- ],
671
- "layers.15.attn_norm.weight": [
672
- 768
673
- ],
674
- "layers.15.mlp.Wi.weight": [
675
- 2304,
676
- 768
677
- ],
678
- "layers.15.mlp.Wo.weight": [
679
- 768,
680
- 1152
681
- ],
682
- "layers.15.mlp_norm.weight": [
683
- 768
684
- ],
685
- "layers.16.attn.Wo.weight": [
686
- 768,
687
- 768
688
- ],
689
- "layers.16.attn.Wqkv.weight": [
690
- 2304,
691
- 768
692
- ],
693
- "layers.16.attn_norm.weight": [
694
- 768
695
- ],
696
- "layers.16.mlp.Wi.weight": [
697
- 2304,
698
- 768
699
- ],
700
- "layers.16.mlp.Wo.weight": [
701
- 768,
702
- 1152
703
- ],
704
- "layers.16.mlp_norm.weight": [
705
- 768
706
- ],
707
- "layers.17.attn.Wo.weight": [
708
- 768,
709
- 768
710
- ],
711
- "layers.17.attn.Wqkv.weight": [
712
- 2304,
713
- 768
714
- ],
715
- "layers.17.attn_norm.weight": [
716
- 768
717
- ],
718
- "layers.17.mlp.Wi.weight": [
719
- 2304,
720
- 768
721
- ],
722
- "layers.17.mlp.Wo.weight": [
723
- 768,
724
- 1152
725
- ],
726
- "layers.17.mlp_norm.weight": [
727
- 768
728
- ],
729
- "layers.18.attn.Wo.weight": [
730
- 768,
731
- 768
732
- ],
733
- "layers.18.attn.Wqkv.weight": [
734
- 2304,
735
- 768
736
- ],
737
- "layers.18.attn_norm.weight": [
738
- 768
739
- ],
740
- "layers.18.mlp.Wi.weight": [
741
- 2304,
742
- 768
743
- ],
744
- "layers.18.mlp.Wo.weight": [
745
- 768,
746
- 1152
747
- ],
748
- "layers.18.mlp_norm.weight": [
749
- 768
750
- ],
751
- "layers.19.attn.Wo.weight": [
752
- 768,
753
- 768
754
- ],
755
- "layers.19.attn.Wqkv.weight": [
756
- 2304,
757
- 768
758
- ],
759
- "layers.19.attn_norm.weight": [
760
- 768
761
- ],
762
- "layers.19.mlp.Wi.weight": [
763
- 2304,
764
- 768
765
- ],
766
- "layers.19.mlp.Wo.weight": [
767
- 768,
768
- 1152
769
- ],
770
- "layers.19.mlp_norm.weight": [
771
- 768
772
- ],
773
- "layers.2.attn.Wo.weight": [
774
- 768,
775
- 768
776
- ],
777
- "layers.2.attn.Wqkv.weight": [
778
- 2304,
779
- 768
780
- ],
781
- "layers.2.attn_norm.weight": [
782
- 768
783
- ],
784
- "layers.2.mlp.Wi.weight": [
785
- 2304,
786
- 768
787
- ],
788
- "layers.2.mlp.Wo.weight": [
789
- 768,
790
- 1152
791
- ],
792
- "layers.2.mlp_norm.weight": [
793
- 768
794
- ],
795
- "layers.20.attn.Wo.weight": [
796
- 768,
797
- 768
798
- ],
799
- "layers.20.attn.Wqkv.weight": [
800
- 2304,
801
- 768
802
- ],
803
- "layers.20.attn_norm.weight": [
804
- 768
805
- ],
806
- "layers.20.mlp.Wi.weight": [
807
- 2304,
808
- 768
809
- ],
810
- "layers.20.mlp.Wo.weight": [
811
- 768,
812
- 1152
813
- ],
814
- "layers.20.mlp_norm.weight": [
815
- 768
816
- ],
817
- "layers.21.attn.Wo.weight": [
818
- 768,
819
- 768
820
- ],
821
- "layers.21.attn.Wqkv.weight": [
822
- 2304,
823
- 768
824
- ],
825
- "layers.21.attn_norm.weight": [
826
- 768
827
- ],
828
- "layers.21.mlp.Wi.weight": [
829
- 2304,
830
- 768
831
- ],
832
- "layers.21.mlp.Wo.weight": [
833
- 768,
834
- 1152
835
- ],
836
- "layers.21.mlp_norm.weight": [
837
- 768
838
- ],
839
- "layers.3.attn.Wo.weight": [
840
- 768,
841
- 768
842
- ],
843
- "layers.3.attn.Wqkv.weight": [
844
- 2304,
845
- 768
846
- ],
847
- "layers.3.attn_norm.weight": [
848
- 768
849
- ],
850
- "layers.3.mlp.Wi.weight": [
851
- 2304,
852
- 768
853
- ],
854
- "layers.3.mlp.Wo.weight": [
855
- 768,
856
- 1152
857
- ],
858
- "layers.3.mlp_norm.weight": [
859
- 768
860
- ],
861
- "layers.4.attn.Wo.weight": [
862
- 768,
863
- 768
864
- ],
865
- "layers.4.attn.Wqkv.weight": [
866
- 2304,
867
- 768
868
- ],
869
- "layers.4.attn_norm.weight": [
870
- 768
871
- ],
872
- "layers.4.mlp.Wi.weight": [
873
- 2304,
874
- 768
875
- ],
876
- "layers.4.mlp.Wo.weight": [
877
- 768,
878
- 1152
879
- ],
880
- "layers.4.mlp_norm.weight": [
881
- 768
882
- ],
883
- "layers.5.attn.Wo.weight": [
884
- 768,
885
- 768
886
- ],
887
- "layers.5.attn.Wqkv.weight": [
888
- 2304,
889
- 768
890
- ],
891
- "layers.5.attn_norm.weight": [
892
- 768
893
- ],
894
- "layers.5.mlp.Wi.weight": [
895
- 2304,
896
- 768
897
- ],
898
- "layers.5.mlp.Wo.weight": [
899
- 768,
900
- 1152
901
- ],
902
- "layers.5.mlp_norm.weight": [
903
- 768
904
- ],
905
- "layers.6.attn.Wo.weight": [
906
- 768,
907
- 768
908
- ],
909
- "layers.6.attn.Wqkv.weight": [
910
- 2304,
911
- 768
912
- ],
913
- "layers.6.attn_norm.weight": [
914
- 768
915
- ],
916
- "layers.6.mlp.Wi.weight": [
917
- 2304,
918
- 768
919
- ],
920
- "layers.6.mlp.Wo.weight": [
921
- 768,
922
- 1152
923
- ],
924
- "layers.6.mlp_norm.weight": [
925
- 768
926
- ],
927
- "layers.7.attn.Wo.weight": [
928
- 768,
929
- 768
930
- ],
931
- "layers.7.attn.Wqkv.weight": [
932
- 2304,
933
- 768
934
- ],
935
- "layers.7.attn_norm.weight": [
936
- 768
937
- ],
938
- "layers.7.mlp.Wi.weight": [
939
- 2304,
940
- 768
941
- ],
942
- "layers.7.mlp.Wo.weight": [
943
- 768,
944
- 1152
945
- ],
946
- "layers.7.mlp_norm.weight": [
947
- 768
948
- ],
949
- "layers.8.attn.Wo.weight": [
950
- 768,
951
- 768
952
- ],
953
- "layers.8.attn.Wqkv.weight": [
954
- 2304,
955
- 768
956
- ],
957
- "layers.8.attn_norm.weight": [
958
- 768
959
- ],
960
- "layers.8.mlp.Wi.weight": [
961
- 2304,
962
- 768
963
- ],
964
- "layers.8.mlp.Wo.weight": [
965
- 768,
966
- 1152
967
- ],
968
- "layers.8.mlp_norm.weight": [
969
- 768
970
- ],
971
- "layers.9.attn.Wo.weight": [
972
- 768,
973
- 768
974
- ],
975
- "layers.9.attn.Wqkv.weight": [
976
- 2304,
977
- 768
978
- ],
979
- "layers.9.attn_norm.weight": [
980
- 768
981
- ],
982
- "layers.9.mlp.Wi.weight": [
983
- 2304,
984
- 768
985
- ],
986
- "layers.9.mlp.Wo.weight": [
987
- 768,
988
- 1152
989
- ],
990
- "layers.9.mlp_norm.weight": [
991
- 768
992
- ]
993
- },
994
- "head_shapes": {
995
- "heads.choice.0.linear1.bias": [
996
- 3072
997
- ],
998
- "heads.choice.0.linear1.weight": [
999
- 3072,
1000
- 768
1001
- ],
1002
- "heads.choice.0.linear2.bias": [
1003
- 768
1004
- ],
1005
- "heads.choice.0.linear2.weight": [
1006
- 768,
1007
- 3072
1008
- ],
1009
- "heads.choice.0.norm1.bias": [
1010
- 768
1011
- ],
1012
- "heads.choice.0.norm1.weight": [
1013
- 768
1014
- ],
1015
- "heads.choice.0.norm2.bias": [
1016
- 768
1017
- ],
1018
- "heads.choice.0.norm2.weight": [
1019
- 768
1020
- ],
1021
- "heads.choice.0.self_attn.in_proj_bias": [
1022
- 2304
1023
- ],
1024
- "heads.choice.0.self_attn.in_proj_weight": [
1025
- 2304,
1026
- 768
1027
- ],
1028
- "heads.choice.0.self_attn.out_proj.bias": [
1029
- 768
1030
- ],
1031
- "heads.choice.0.self_attn.out_proj.weight": [
1032
- 768,
1033
- 768
1034
- ],
1035
- "heads.choice.1.linear1.bias": [
1036
- 3072
1037
- ],
1038
- "heads.choice.1.linear1.weight": [
1039
- 3072,
1040
- 768
1041
- ],
1042
- "heads.choice.1.linear2.bias": [
1043
- 768
1044
- ],
1045
- "heads.choice.1.linear2.weight": [
1046
- 768,
1047
- 3072
1048
- ],
1049
- "heads.choice.1.norm1.bias": [
1050
- 768
1051
- ],
1052
- "heads.choice.1.norm1.weight": [
1053
- 768
1054
- ],
1055
- "heads.choice.1.norm2.bias": [
1056
- 768
1057
- ],
1058
- "heads.choice.1.norm2.weight": [
1059
- 768
1060
- ],
1061
- "heads.choice.1.self_attn.in_proj_bias": [
1062
- 2304
1063
- ],
1064
- "heads.choice.1.self_attn.in_proj_weight": [
1065
- 2304,
1066
- 768
1067
- ],
1068
- "heads.choice.1.self_attn.out_proj.bias": [
1069
- 768
1070
- ],
1071
- "heads.choice.1.self_attn.out_proj.weight": [
1072
- 768,
1073
- 768
1074
- ],
1075
- "heads.noul.0.linear1.bias": [
1076
- 3072
1077
- ],
1078
- "heads.noul.0.linear1.weight": [
1079
- 3072,
1080
- 768
1081
- ],
1082
- "heads.noul.0.linear2.bias": [
1083
- 768
1084
- ],
1085
- "heads.noul.0.linear2.weight": [
1086
- 768,
1087
- 3072
1088
- ],
1089
- "heads.noul.0.norm1.bias": [
1090
- 768
1091
- ],
1092
- "heads.noul.0.norm1.weight": [
1093
- 768
1094
- ],
1095
- "heads.noul.0.norm2.bias": [
1096
- 768
1097
- ],
1098
- "heads.noul.0.norm2.weight": [
1099
- 768
1100
- ],
1101
- "heads.noul.0.self_attn.in_proj_bias": [
1102
- 2304
1103
- ],
1104
- "heads.noul.0.self_attn.in_proj_weight": [
1105
- 2304,
1106
- 768
1107
- ],
1108
- "heads.noul.0.self_attn.out_proj.bias": [
1109
- 768
1110
- ],
1111
- "heads.noul.0.self_attn.out_proj.weight": [
1112
- 768,
1113
- 768
1114
- ],
1115
- "heads.noul.1.linear1.bias": [
1116
- 3072
1117
- ],
1118
- "heads.noul.1.linear1.weight": [
1119
- 3072,
1120
- 768
1121
- ],
1122
- "heads.noul.1.linear2.bias": [
1123
- 768
1124
- ],
1125
- "heads.noul.1.linear2.weight": [
1126
- 768,
1127
- 3072
1128
- ],
1129
- "heads.noul.1.norm1.bias": [
1130
- 768
1131
- ],
1132
- "heads.noul.1.norm1.weight": [
1133
- 768
1134
- ],
1135
- "heads.noul.1.norm2.bias": [
1136
- 768
1137
- ],
1138
- "heads.noul.1.norm2.weight": [
1139
- 768
1140
- ],
1141
- "heads.noul.1.self_attn.in_proj_bias": [
1142
- 2304
1143
- ],
1144
- "heads.noul.1.self_attn.in_proj_weight": [
1145
- 2304,
1146
- 768
1147
- ],
1148
- "heads.noul.1.self_attn.out_proj.bias": [
1149
- 768
1150
- ],
1151
- "heads.noul.1.self_attn.out_proj.weight": [
1152
- 768,
1153
- 768
1154
- ],
1155
- "heads.score.0.linear1.bias": [
1156
- 3072
1157
- ],
1158
- "heads.score.0.linear1.weight": [
1159
- 3072,
1160
- 768
1161
- ],
1162
- "heads.score.0.linear2.bias": [
1163
- 768
1164
- ],
1165
- "heads.score.0.linear2.weight": [
1166
- 768,
1167
- 3072
1168
- ],
1169
- "heads.score.0.norm1.bias": [
1170
- 768
1171
- ],
1172
- "heads.score.0.norm1.weight": [
1173
- 768
1174
- ],
1175
- "heads.score.0.norm2.bias": [
1176
- 768
1177
- ],
1178
- "heads.score.0.norm2.weight": [
1179
- 768
1180
- ],
1181
- "heads.score.0.self_attn.in_proj_bias": [
1182
- 2304
1183
- ],
1184
- "heads.score.0.self_attn.in_proj_weight": [
1185
- 2304,
1186
- 768
1187
- ],
1188
- "heads.score.0.self_attn.out_proj.bias": [
1189
- 768
1190
- ],
1191
- "heads.score.0.self_attn.out_proj.weight": [
1192
- 768,
1193
- 768
1194
- ],
1195
- "heads.score.1.linear1.bias": [
1196
- 3072
1197
- ],
1198
- "heads.score.1.linear1.weight": [
1199
- 3072,
1200
- 768
1201
- ],
1202
- "heads.score.1.linear2.bias": [
1203
- 768
1204
- ],
1205
- "heads.score.1.linear2.weight": [
1206
- 768,
1207
- 3072
1208
- ],
1209
- "heads.score.1.norm1.bias": [
1210
- 768
1211
- ],
1212
- "heads.score.1.norm1.weight": [
1213
- 768
1214
- ],
1215
- "heads.score.1.norm2.bias": [
1216
- 768
1217
- ],
1218
- "heads.score.1.norm2.weight": [
1219
- 768
1220
- ],
1221
- "heads.score.1.self_attn.in_proj_bias": [
1222
- 2304
1223
- ],
1224
- "heads.score.1.self_attn.in_proj_weight": [
1225
- 2304,
1226
- 768
1227
- ],
1228
- "heads.score.1.self_attn.out_proj.bias": [
1229
- 768
1230
- ],
1231
- "heads.score.1.self_attn.out_proj.weight": [
1232
- 768,
1233
- 768
1234
- ],
1235
- "scorers.choice.0.bias": [
1236
- 768
1237
- ],
1238
- "scorers.choice.0.weight": [
1239
- 768
1240
- ],
1241
- "scorers.choice.1.bias": [
1242
- 768
1243
- ],
1244
- "scorers.choice.1.weight": [
1245
- 768,
1246
- 768
1247
- ],
1248
- "scorers.choice.3.bias": [
1249
- 1
1250
- ],
1251
- "scorers.choice.3.weight": [
1252
- 1,
1253
- 768
1254
- ],
1255
- "scorers.noul.0.bias": [
1256
- 768
1257
- ],
1258
- "scorers.noul.0.weight": [
1259
- 768
1260
- ],
1261
- "scorers.noul.1.bias": [
1262
- 768
1263
- ],
1264
- "scorers.noul.1.weight": [
1265
- 768,
1266
- 768
1267
- ],
1268
- "scorers.noul.3.bias": [
1269
- 1
1270
- ],
1271
- "scorers.noul.3.weight": [
1272
- 1,
1273
- 768
1274
- ],
1275
- "scorers.score.0.bias": [
1276
- 768
1277
- ],
1278
- "scorers.score.0.weight": [
1279
- 768
1280
- ],
1281
- "scorers.score.1.bias": [
1282
- 768
1283
- ],
1284
- "scorers.score.1.weight": [
1285
- 768,
1286
- 768
1287
- ],
1288
- "scorers.score.3.bias": [
1289
- 1
1290
- ],
1291
- "scorers.score.3.weight": [
1292
- 1,
1293
- 768
1294
- ],
1295
- "type_embedding.weight": [
1296
- 3,
1297
- 768
1298
- ]
1299
- },
1300
- "score_suffix_shapes": {
1301
- "score_blocks.0.attn.Wo.weight": [
1302
- 768,
1303
- 768
1304
- ],
1305
- "score_blocks.0.attn.Wqkv.weight": [
1306
- 2304,
1307
- 768
1308
- ],
1309
- "score_blocks.0.mlp.Wi.weight": [
1310
- 2304,
1311
- 768
1312
- ],
1313
- "score_blocks.0.mlp.Wo.weight": [
1314
- 768,
1315
- 1152
1316
- ],
1317
- "score_blocks.0.mlp_norm.weight": [
1318
- 768
1319
- ],
1320
- "score_blocks.1.attn.Wo.weight": [
1321
- 768,
1322
- 768
1323
- ],
1324
- "score_blocks.1.attn.Wqkv.weight": [
1325
- 2304,
1326
- 768
1327
- ],
1328
- "score_blocks.1.attn_norm.weight": [
1329
- 768
1330
- ],
1331
- "score_blocks.1.mlp.Wi.weight": [
1332
- 2304,
1333
- 768
1334
- ],
1335
- "score_blocks.1.mlp.Wo.weight": [
1336
- 768,
1337
- 1152
1338
- ],
1339
- "score_blocks.1.mlp_norm.weight": [
1340
- 768
1341
- ],
1342
- "score_blocks.10.attn.Wo.weight": [
1343
- 768,
1344
- 768
1345
- ],
1346
- "score_blocks.10.attn.Wqkv.weight": [
1347
- 2304,
1348
- 768
1349
- ],
1350
- "score_blocks.10.attn_norm.weight": [
1351
- 768
1352
- ],
1353
- "score_blocks.10.mlp.Wi.weight": [
1354
- 2304,
1355
- 768
1356
- ],
1357
- "score_blocks.10.mlp.Wo.weight": [
1358
- 768,
1359
- 1152
1360
- ],
1361
- "score_blocks.10.mlp_norm.weight": [
1362
- 768
1363
- ],
1364
- "score_blocks.11.attn.Wo.weight": [
1365
- 768,
1366
- 768
1367
- ],
1368
- "score_blocks.11.attn.Wqkv.weight": [
1369
- 2304,
1370
- 768
1371
- ],
1372
- "score_blocks.11.attn_norm.weight": [
1373
- 768
1374
- ],
1375
- "score_blocks.11.mlp.Wi.weight": [
1376
- 2304,
1377
- 768
1378
- ],
1379
- "score_blocks.11.mlp.Wo.weight": [
1380
- 768,
1381
- 1152
1382
- ],
1383
- "score_blocks.11.mlp_norm.weight": [
1384
- 768
1385
- ],
1386
- "score_blocks.12.attn.Wo.weight": [
1387
- 768,
1388
- 768
1389
- ],
1390
- "score_blocks.12.attn.Wqkv.weight": [
1391
- 2304,
1392
- 768
1393
- ],
1394
- "score_blocks.12.attn_norm.weight": [
1395
- 768
1396
- ],
1397
- "score_blocks.12.mlp.Wi.weight": [
1398
- 2304,
1399
- 768
1400
- ],
1401
- "score_blocks.12.mlp.Wo.weight": [
1402
- 768,
1403
- 1152
1404
- ],
1405
- "score_blocks.12.mlp_norm.weight": [
1406
- 768
1407
- ],
1408
- "score_blocks.13.attn.Wo.weight": [
1409
- 768,
1410
- 768
1411
- ],
1412
- "score_blocks.13.attn.Wqkv.weight": [
1413
- 2304,
1414
- 768
1415
- ],
1416
- "score_blocks.13.attn_norm.weight": [
1417
- 768
1418
- ],
1419
- "score_blocks.13.mlp.Wi.weight": [
1420
- 2304,
1421
- 768
1422
- ],
1423
- "score_blocks.13.mlp.Wo.weight": [
1424
- 768,
1425
- 1152
1426
- ],
1427
- "score_blocks.13.mlp_norm.weight": [
1428
- 768
1429
- ],
1430
- "score_blocks.14.attn.Wo.weight": [
1431
- 768,
1432
- 768
1433
- ],
1434
- "score_blocks.14.attn.Wqkv.weight": [
1435
- 2304,
1436
- 768
1437
- ],
1438
- "score_blocks.14.attn_norm.weight": [
1439
- 768
1440
- ],
1441
- "score_blocks.14.mlp.Wi.weight": [
1442
- 2304,
1443
- 768
1444
- ],
1445
- "score_blocks.14.mlp.Wo.weight": [
1446
- 768,
1447
- 1152
1448
- ],
1449
- "score_blocks.14.mlp_norm.weight": [
1450
- 768
1451
- ],
1452
- "score_blocks.15.attn.Wo.weight": [
1453
- 768,
1454
- 768
1455
- ],
1456
- "score_blocks.15.attn.Wqkv.weight": [
1457
- 2304,
1458
- 768
1459
- ],
1460
- "score_blocks.15.attn_norm.weight": [
1461
- 768
1462
- ],
1463
- "score_blocks.15.mlp.Wi.weight": [
1464
- 2304,
1465
- 768
1466
- ],
1467
- "score_blocks.15.mlp.Wo.weight": [
1468
- 768,
1469
- 1152
1470
- ],
1471
- "score_blocks.15.mlp_norm.weight": [
1472
- 768
1473
- ],
1474
- "score_blocks.16.attn.Wo.weight": [
1475
- 768,
1476
- 768
1477
- ],
1478
- "score_blocks.16.attn.Wqkv.weight": [
1479
- 2304,
1480
- 768
1481
- ],
1482
- "score_blocks.16.attn_norm.weight": [
1483
- 768
1484
- ],
1485
- "score_blocks.16.mlp.Wi.weight": [
1486
- 2304,
1487
- 768
1488
- ],
1489
- "score_blocks.16.mlp.Wo.weight": [
1490
- 768,
1491
- 1152
1492
- ],
1493
- "score_blocks.16.mlp_norm.weight": [
1494
- 768
1495
- ],
1496
- "score_blocks.17.attn.Wo.weight": [
1497
- 768,
1498
- 768
1499
- ],
1500
- "score_blocks.17.attn.Wqkv.weight": [
1501
- 2304,
1502
- 768
1503
- ],
1504
- "score_blocks.17.attn_norm.weight": [
1505
- 768
1506
- ],
1507
- "score_blocks.17.mlp.Wi.weight": [
1508
- 2304,
1509
- 768
1510
- ],
1511
- "score_blocks.17.mlp.Wo.weight": [
1512
- 768,
1513
- 1152
1514
- ],
1515
- "score_blocks.17.mlp_norm.weight": [
1516
- 768
1517
- ],
1518
- "score_blocks.18.attn.Wo.weight": [
1519
- 768,
1520
- 768
1521
- ],
1522
- "score_blocks.18.attn.Wqkv.weight": [
1523
- 2304,
1524
- 768
1525
- ],
1526
- "score_blocks.18.attn_norm.weight": [
1527
- 768
1528
- ],
1529
- "score_blocks.18.mlp.Wi.weight": [
1530
- 2304,
1531
- 768
1532
- ],
1533
- "score_blocks.18.mlp.Wo.weight": [
1534
- 768,
1535
- 1152
1536
- ],
1537
- "score_blocks.18.mlp_norm.weight": [
1538
- 768
1539
- ],
1540
- "score_blocks.19.attn.Wo.weight": [
1541
- 768,
1542
- 768
1543
- ],
1544
- "score_blocks.19.attn.Wqkv.weight": [
1545
- 2304,
1546
- 768
1547
- ],
1548
- "score_blocks.19.attn_norm.weight": [
1549
- 768
1550
- ],
1551
- "score_blocks.19.mlp.Wi.weight": [
1552
- 2304,
1553
- 768
1554
- ],
1555
- "score_blocks.19.mlp.Wo.weight": [
1556
- 768,
1557
- 1152
1558
- ],
1559
- "score_blocks.19.mlp_norm.weight": [
1560
- 768
1561
- ],
1562
- "score_blocks.2.attn.Wo.weight": [
1563
- 768,
1564
- 768
1565
- ],
1566
- "score_blocks.2.attn.Wqkv.weight": [
1567
- 2304,
1568
- 768
1569
- ],
1570
- "score_blocks.2.attn_norm.weight": [
1571
- 768
1572
- ],
1573
- "score_blocks.2.mlp.Wi.weight": [
1574
- 2304,
1575
- 768
1576
- ],
1577
- "score_blocks.2.mlp.Wo.weight": [
1578
- 768,
1579
- 1152
1580
- ],
1581
- "score_blocks.2.mlp_norm.weight": [
1582
- 768
1583
- ],
1584
- "score_blocks.20.attn.Wo.weight": [
1585
- 768,
1586
- 768
1587
- ],
1588
- "score_blocks.20.attn.Wqkv.weight": [
1589
- 2304,
1590
- 768
1591
- ],
1592
- "score_blocks.20.attn_norm.weight": [
1593
- 768
1594
- ],
1595
- "score_blocks.20.mlp.Wi.weight": [
1596
- 2304,
1597
- 768
1598
- ],
1599
- "score_blocks.20.mlp.Wo.weight": [
1600
- 768,
1601
- 1152
1602
- ],
1603
- "score_blocks.20.mlp_norm.weight": [
1604
- 768
1605
- ],
1606
- "score_blocks.21.attn.Wo.weight": [
1607
- 768,
1608
- 768
1609
- ],
1610
- "score_blocks.21.attn.Wqkv.weight": [
1611
- 2304,
1612
- 768
1613
- ],
1614
- "score_blocks.21.attn_norm.weight": [
1615
- 768
1616
- ],
1617
- "score_blocks.21.mlp.Wi.weight": [
1618
- 2304,
1619
- 768
1620
- ],
1621
- "score_blocks.21.mlp.Wo.weight": [
1622
- 768,
1623
- 1152
1624
- ],
1625
- "score_blocks.21.mlp_norm.weight": [
1626
- 768
1627
- ],
1628
- "score_blocks.3.attn.Wo.weight": [
1629
- 768,
1630
- 768
1631
- ],
1632
- "score_blocks.3.attn.Wqkv.weight": [
1633
- 2304,
1634
- 768
1635
- ],
1636
- "score_blocks.3.attn_norm.weight": [
1637
- 768
1638
- ],
1639
- "score_blocks.3.mlp.Wi.weight": [
1640
- 2304,
1641
- 768
1642
- ],
1643
- "score_blocks.3.mlp.Wo.weight": [
1644
- 768,
1645
- 1152
1646
- ],
1647
- "score_blocks.3.mlp_norm.weight": [
1648
- 768
1649
- ],
1650
- "score_blocks.4.attn.Wo.weight": [
1651
- 768,
1652
- 768
1653
- ],
1654
- "score_blocks.4.attn.Wqkv.weight": [
1655
- 2304,
1656
- 768
1657
- ],
1658
- "score_blocks.4.attn_norm.weight": [
1659
- 768
1660
- ],
1661
- "score_blocks.4.mlp.Wi.weight": [
1662
- 2304,
1663
- 768
1664
- ],
1665
- "score_blocks.4.mlp.Wo.weight": [
1666
- 768,
1667
- 1152
1668
- ],
1669
- "score_blocks.4.mlp_norm.weight": [
1670
- 768
1671
- ],
1672
- "score_blocks.5.attn.Wo.weight": [
1673
- 768,
1674
- 768
1675
- ],
1676
- "score_blocks.5.attn.Wqkv.weight": [
1677
- 2304,
1678
- 768
1679
- ],
1680
- "score_blocks.5.attn_norm.weight": [
1681
- 768
1682
- ],
1683
- "score_blocks.5.mlp.Wi.weight": [
1684
- 2304,
1685
- 768
1686
- ],
1687
- "score_blocks.5.mlp.Wo.weight": [
1688
- 768,
1689
- 1152
1690
- ],
1691
- "score_blocks.5.mlp_norm.weight": [
1692
- 768
1693
- ],
1694
- "score_blocks.6.attn.Wo.weight": [
1695
- 768,
1696
- 768
1697
- ],
1698
- "score_blocks.6.attn.Wqkv.weight": [
1699
- 2304,
1700
- 768
1701
- ],
1702
- "score_blocks.6.attn_norm.weight": [
1703
- 768
1704
- ],
1705
- "score_blocks.6.mlp.Wi.weight": [
1706
- 2304,
1707
- 768
1708
- ],
1709
- "score_blocks.6.mlp.Wo.weight": [
1710
- 768,
1711
- 1152
1712
- ],
1713
- "score_blocks.6.mlp_norm.weight": [
1714
- 768
1715
- ],
1716
- "score_blocks.7.attn.Wo.weight": [
1717
- 768,
1718
- 768
1719
- ],
1720
- "score_blocks.7.attn.Wqkv.weight": [
1721
- 2304,
1722
- 768
1723
- ],
1724
- "score_blocks.7.attn_norm.weight": [
1725
- 768
1726
- ],
1727
- "score_blocks.7.mlp.Wi.weight": [
1728
- 2304,
1729
- 768
1730
- ],
1731
- "score_blocks.7.mlp.Wo.weight": [
1732
- 768,
1733
- 1152
1734
- ],
1735
- "score_blocks.7.mlp_norm.weight": [
1736
- 768
1737
- ],
1738
- "score_blocks.8.attn.Wo.weight": [
1739
- 768,
1740
- 768
1741
- ],
1742
- "score_blocks.8.attn.Wqkv.weight": [
1743
- 2304,
1744
- 768
1745
- ],
1746
- "score_blocks.8.attn_norm.weight": [
1747
- 768
1748
- ],
1749
- "score_blocks.8.mlp.Wi.weight": [
1750
- 2304,
1751
- 768
1752
- ],
1753
- "score_blocks.8.mlp.Wo.weight": [
1754
- 768,
1755
- 1152
1756
- ],
1757
- "score_blocks.8.mlp_norm.weight": [
1758
- 768
1759
- ],
1760
- "score_blocks.9.attn.Wo.weight": [
1761
- 768,
1762
- 768
1763
- ],
1764
- "score_blocks.9.attn.Wqkv.weight": [
1765
- 2304,
1766
- 768
1767
- ],
1768
- "score_blocks.9.attn_norm.weight": [
1769
- 768
1770
- ],
1771
- "score_blocks.9.mlp.Wi.weight": [
1772
- 2304,
1773
- 768
1774
- ],
1775
- "score_blocks.9.mlp.Wo.weight": [
1776
- 768,
1777
- 1152
1778
- ],
1779
- "score_blocks.9.mlp_norm.weight": [
1780
- 768
1781
- ],
1782
- "score_final_norm.weight": [
1783
- 768
1784
- ]
1785
- }
1786
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
native/MANIFEST.json DELETED
@@ -1,125 +0,0 @@
1
- {
2
- "files": {
3
- "INVENTORY.json": {
4
- "bytes": 31007,
5
- "sha256": "06aed00ac4571582b992a995a793875804a0546bdae1461c4161b30fba2143a9"
6
- },
7
- "STATE_LAYOUT.json": {
8
- "bytes": 10971,
9
- "sha256": "a6b24716e2b21240d327ba8763a73f1b54862bc86d5e1f92bb9e670be968e494"
10
- },
11
- "__init__.py": {
12
- "bytes": 78,
13
- "sha256": "1afb9dbcfc379f28049486fe6acb7acfdaa8d16b6773d6084ca4b82ead26f010"
14
- },
15
- "artifacts.py": {
16
- "bytes": 7958,
17
- "sha256": "c98adaf6d782e9fc592cd78b1307d63faffa9ba622b9d508c63337c29156e4d3"
18
- },
19
- "choice_encoder.safetensors": {
20
- "bytes": 441337216,
21
- "sha256": "9516cc841c485c98b27b4f63d2ea8e604fe8064121173064b113da8a6bf57ef6"
22
- },
23
- "contract.py": {
24
- "bytes": 10886,
25
- "sha256": "51a24800792bb3e5bf2a11f50f7bc384770641f4f44ed46277dc01e891bf4726"
26
- },
27
- "decision_config.json": {
28
- "bytes": 14308,
29
- "sha256": "1fefb4ad7dede00c4633bb2ce5fc44fb04237207b9e9ce90fcf8ea8b01a1328d"
30
- },
31
- "decision_heads.safetensors": {
32
- "bytes": 177241884,
33
- "sha256": "bce3ee658a978a19c48c605b921cff994f892e7fc17abd6f8645f07428fbc36f"
34
- },
35
- "encoder/config.json": {
36
- "bytes": 2769,
37
- "sha256": "7aff915e9f159305e0bef3eb0206416f99b8560260b8969b35f1dcd54aaad1a5"
38
- },
39
- "encoder/model.safetensors": {
40
- "bytes": 1227771752,
41
- "sha256": "daaafd81c4ed767d203e24226d78e792800068a7dac344aa043844b82a6306c7"
42
- },
43
- "infer.py": {
44
- "bytes": 1095,
45
- "sha256": "9d14841935c836a0765c705d5a24e8fa97437ae35fff65bc6e690b397f0d0315"
46
- },
47
- "model.py": {
48
- "bytes": 11098,
49
- "sha256": "8fe91e2a77f372d32117e62281ccb58a054c144228b73b4987f1e1a3fca9ee5b"
50
- },
51
- "packing.py": {
52
- "bytes": 6748,
53
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
54
- },
55
- "policy/__init__.py": {
56
- "bytes": 131,
57
- "sha256": "0cb0bad7d3d0f258ecf95f52297ee8733a75f8504cabb86b9aea4b8256885114"
58
- },
59
- "policy/artifacts.py": {
60
- "bytes": 7877,
61
- "sha256": "5838d2ea747d912789f3fc813a212af919cbc24ee185073576d9fbe32ed31cf2"
62
- },
63
- "policy/contract.py": {
64
- "bytes": 4334,
65
- "sha256": "0b8eeeeebe9e367564a3c57f48364c5e94ae060f759e220b1326caf152276b67"
66
- },
67
- "policy/infer.py": {
68
- "bytes": 1545,
69
- "sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
70
- },
71
- "policy/model.py": {
72
- "bytes": 7678,
73
- "sha256": "eaeebac8fd6bd96243a5c4b6225c86ee3359c067784f1595979ee603cba9acca"
74
- },
75
- "policy/packing.py": {
76
- "bytes": 6748,
77
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
78
- },
79
- "policy/reference/__init__.py": {
80
- "bytes": 192,
81
- "sha256": "c8d6fd86207752407f94437544a97266b9529f88d2b341d35aafcd911c942b28"
82
- },
83
- "policy/reference/artifacts.py": {
84
- "bytes": 9342,
85
- "sha256": "3f59428b7a05608ac01c73c7db89c38fcc3e2c5030219df8d053af92b32e6159"
86
- },
87
- "policy/reference/infer.py": {
88
- "bytes": 1545,
89
- "sha256": "672bfae798060d51fcccac03143ac881fe7512893ffdacdb36b0618d4b160896"
90
- },
91
- "policy/reference/model.py": {
92
- "bytes": 5060,
93
- "sha256": "733ea4a48ea03089dac8c3e6a175920704fac50f93313b0214cec4052f25bd26"
94
- },
95
- "policy/reference/modernbert_sdpa_layout.py": {
96
- "bytes": 3365,
97
- "sha256": "0fb3a22db93ad76e30dfbfb3011de442149d97c565e139a3a55738287a1fbbc8"
98
- },
99
- "policy/reference/packing.py": {
100
- "bytes": 6748,
101
- "sha256": "f1dab7f032f4aaab27118606397ca67d71b1f4a2ba49e65bebcfd83a64b27819"
102
- },
103
- "score_encoder.safetensors": {
104
- "bytes": 441337080,
105
- "sha256": "4f45795977846ef4c31b34e4f35bf95dabf5951d71358bab32c9a94814a84a53"
106
- },
107
- "tokenizer/special_tokens_map.json": {
108
- "bytes": 1051,
109
- "sha256": "6e3204c5a4004719c185007a745d1940471a6fb02521cfa0a00306dbb6bc5903"
110
- },
111
- "tokenizer/tokenizer.json": {
112
- "bytes": 34363188,
113
- "sha256": "609d8f4c067cd3950f88594c5a802616cea245823836ef5848ee4fc40aab5b6f"
114
- },
115
- "tokenizer/tokenizer_config.json": {
116
- "bytes": 46470,
117
- "sha256": "74a259bb1a3811a7e3028adcd07a65765d866d5e66e0e883f0994ccfa67e8455"
118
- },
119
- "training_policy.py": {
120
- "bytes": 4761,
121
- "sha256": "6f7fad91c5089b304a1d637b594027a9379a63600792766ce0abca6b24bd2ec5"
122
- }
123
- },
124
- "schema": "decision.files.v1"
125
- }